#!/usr/bin/env python3 """EPUB -> audio cu una din vocile din tools/tts.py. CLI: python3 tools/epub_to_audio.py --voice "Marius 4" --out-dir DIR [--chapters N] [--start-chapter N] Per capitol: extrage text curat (ebooklib + BeautifulSoup), îl împarte în bucăți sigure pentru TTS (pe propoziții, sub _CHUNK_CHARS), sintetizează secvențial cu tools.tts.synthesize, apoi concatenează cu ffmpeg într-un singur WAV per capitol. """ import argparse import json import re import subprocess import sys from pathlib import Path from bs4 import BeautifulSoup from ebooklib import epub, ITEM_DOCUMENT REPO_ROOT = Path(__file__).resolve().parent.parent if str(REPO_ROOT) not in sys.path: sys.path.insert(0, str(REPO_ROOT)) from tools.tts import synthesize # noqa: E402 # pocket-tts n-are un hard limit documentat ca Supertonic, dar bucăți mari cresc # riscul de timeout (60s) și degradarea calității pe propoziții foarte lungi. _CHUNK_CHARS = 600 def extract_chapters(epub_path: Path) -> list[dict]: book = epub.read_epub(str(epub_path)) chapters = [] for item in book.get_items_of_type(ITEM_DOCUMENT): soup = BeautifulSoup(item.get_content(), "html.parser") for tag in soup(["script", "style"]): tag.decompose() text = soup.get_text(separator=" ", strip=True) text = re.sub(r"\s+", " ", text).strip() if len(text) < 200: continue # sar peste pagini de copertă/titlu/ToC, prea puțin conținut title_tag = soup.find(["h1", "h2", "title"]) title = title_tag.get_text(strip=True) if title_tag else item.get_name() chapters.append({"name": item.get_name(), "title": title, "text": text}) return chapters def chunk_text(text: str, max_chars: int = _CHUNK_CHARS) -> list[str]: sentences = re.split(r"(?<=[.!?])\s+", text) chunks = [] current = "" for sentence in sentences: if current and len(current) + 1 + len(sentence) > max_chars: chunks.append(current) current = sentence else: current = f"{current} {sentence}".strip() if current: chunks.append(current) return chunks def synthesize_chapter(text: str, voice: str, out_path: Path, mp3: bool = True) -> dict: chunks = chunk_text(text) wav_paths = [] for i, chunk in enumerate(chunks): result = synthesize(chunk, voice=voice, lang="en") if not result.get("ok"): return {"ok": False, "error": f"chunk {i}/{len(chunks)}: {result.get('error')}"} wav_paths.append(result["path"]) print(f" chunk {i + 1}/{len(chunks)} ok ({result['size_bytes']} bytes)", file=sys.stderr) if not wav_paths: return {"ok": False, "error": "niciun chunk de sintetizat"} wav_path = out_path.with_suffix(".wav") if len(wav_paths) == 1: Path(wav_paths[0]).replace(wav_path) else: list_file = out_path.with_suffix(".concat.txt") list_file.write_text("\n".join(f"file '{p}'" for p in wav_paths)) subprocess.run( ["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", str(list_file), "-c", "copy", str(wav_path)], check=True, capture_output=True, ) list_file.unlink() for p in wav_paths: Path(p).unlink(missing_ok=True) if not mp3: return {"ok": True, "path": str(wav_path), "chunks": len(chunks)} subprocess.run( ["ffmpeg", "-y", "-i", str(wav_path), "-ac", "1", "-b:a", "96k", str(out_path)], check=True, capture_output=True, ) wav_path.unlink() return {"ok": True, "path": str(out_path), "chunks": len(chunks)} _SKIP_TITLES = {"contents"} _MIN_CHAPTER_CHARS = 500 # sub asta e de regulă pagină de titlu, nu conținut de citit def main(): parser = argparse.ArgumentParser() parser.add_argument("epub_path") parser.add_argument("--voice", default="Marius 4") parser.add_argument("--out-dir", default="/home/moltbot/workspace/epub-audio") parser.add_argument("--chapters", type=int, default=None, help="limitează la primele N capitole") parser.add_argument("--start-chapter", type=int, default=0) parser.add_argument("--wav", action="store_true", help="păstrează WAV în loc de MP3") args = parser.parse_args() out_dir = Path(args.out_dir) out_dir.mkdir(parents=True, exist_ok=True) ext = "wav" if args.wav else "mp3" chapters = extract_chapters(Path(args.epub_path)) chapters = [ ch for ch in chapters if len(ch["text"]) >= _MIN_CHAPTER_CHARS and ch["title"].strip().lower() not in _SKIP_TITLES ] print(f"{len(chapters)} capitole de citit (după filtrare titlu/ToC)", file=sys.stderr) end = args.start_chapter + args.chapters if args.chapters else len(chapters) selected = chapters[args.start_chapter:end] results = [] for n, ch in enumerate(selected, start=args.start_chapter + 1): out_path = out_dir / f"{n:02d}_{re.sub(r'[^a-zA-Z0-9]+', '_', ch['title'])[:40]}.{ext}" print(f"[{n}] {ch['title']} ({len(ch['text'])} chars) -> {out_path.name}", file=sys.stderr) result = synthesize_chapter(ch["text"], args.voice, out_path, mp3=not args.wav) results.append({"chapter": ch["title"], **result}) print(json.dumps(results[-1]), file=sys.stderr) print(json.dumps({"ok": True, "results": results})) if __name__ == "__main__": main()