""" Transcribe audio//*.mp3 -> transcripts//*.txt using faster-whisper (CPU, CTranslate2). Deletes the source audio after a successful transcript (nobody wants the audio kept — text only). Resumable: skips lectures with an existing transcript. """ import argparse import logging import sys from pathlib import Path from faster_whisper import WhisperModel logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") log = logging.getLogger(__name__) # Vorbirea normală dă ~100-150 cuvinte/minut. Sub 15% din asta înseamnă # halucinație, tăcere, sau o scriere care a eșuat — nu o transcriere bună. MIN_WORDS_PER_SECOND = 100 / 60 * 0.15 def transcribe_file(model: WhisperModel, audio_path: Path, txt_path: Path) -> bool: """Întoarce True doar dacă transcrierea rezultată pare completă. Apelantul șterge audio-ul DOAR pe True — altfel un fișier pierdut înseamnă re-descărcare, nu doar re-transcriere.""" segments, info = model.transcribe(str(audio_path), language="ro", beam_size=5, vad_filter=True) txt_path.parent.mkdir(parents=True, exist_ok=True) with open(txt_path, "w", encoding="utf-8") as f: for seg in segments: f.write(seg.text.strip() + "\n") if not txt_path.exists(): log.error(f" {txt_path} a dispărut în timpul scrierii — păstrez audio-ul") return False words = len(txt_path.read_text(encoding="utf-8").split()) expected = info.duration * MIN_WORDS_PER_SECOND if words < expected: log.error(f" {txt_path.name}: doar {words} cuvinte pentru {info.duration:.0f}s " f"(minim așteptat {expected:.0f}) — păstrez audio-ul") return False log.info(f" Transcribed: {txt_path.name} (lang={info.language}, duration={info.duration:.0f}s, {words} cuvinte)") return True def main(): p = argparse.ArgumentParser(description="Transcribe ATM course audio") p.add_argument("--audio-dir", default="audio") p.add_argument("--out", default="transcripts") p.add_argument("--model", default="medium", help="faster-whisper model size") p.add_argument("--only", default=None, help="Limit to one module dir name (for piloting)") p.add_argument("--keep-audio", action="store_true", help="Don't delete source mp3 after transcribing") args = p.parse_args() audio_dir = Path(args.audio_dir) out_dir = Path(args.out) mp3_files = sorted(audio_dir.glob("*/*.mp3")) if args.only: mp3_files = [f for f in mp3_files if f.parent.name == args.only] if not mp3_files: log.error(f"No audio files found under {audio_dir}") sys.exit(1) log.info(f"Loading faster-whisper model '{args.model}' (CPU, int8)...") model = WhisperModel(args.model, device="cpu", compute_type="int8") done = skipped = 0 for audio_path in mp3_files: txt_path = out_dir / audio_path.parent.name / (audio_path.stem + ".txt") if txt_path.exists() and txt_path.stat().st_size > 50: log.info(f" Skipping (exists): {txt_path.name}") skipped += 1 continue if not audio_path.exists(): # lista de fișiere se face o singură dată la start; dacă un fișier # e mutat între timp (rebalansare între mașini), nu opri toată coada log.warning(f" A dispărut între timp, sar peste: {audio_path}") continue log.info(f"Transcribing: {audio_path}") ok = transcribe_file(model, audio_path, txt_path) if ok and not args.keep_audio: audio_path.unlink() done += 1 log.info("=" * 60) log.info(f"Transcribed {done}, skipped {skipped}.") if __name__ == "__main__": main()