Files
atm-curs-dl/transcribe.py
Claude Agent fffbcf4f57 Q&A 25.08.2025 sumarizata (13/28) + garda pe fisiere disparute in transcribe.py
Sesiunea: "4 candele in spate" depinde de time frame-ul DECIZIEI (pe weekly
inseamna 4 candele weekly, deci un stop care pare absurd pe daily), metoda de
proiectie a tintei plecand de la un nivel de volum comparabil din trecut (cu
limitele ei, adaugate de mine), strategia MICI cu retest la 1/2 sau 1/3 pe
alte time frame-uri, si o analiza ghidata transformata in lista de verificare
in 7 pasi.

Fix: transcribe.py sare peste fisierele care au disparut intre momentul in
care isi face lista si momentul in care ajunge la ele. Un fisier mutat la
rebalansare a oprit toata coada locala dupa 13 ore.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MtSTyTmt6AbCL9j5ajEDm1
2026-09-13 04:23:16 +00:00

94 lines
3.7 KiB
Python

"""
Transcribe audio/<module>/*.mp3 -> transcripts/<module>/*.txt using
faster-whisper (CPU, CTranslate2). Deletes the source audio after a
successful transcript (nobody wants the audio kept — text only).
Resumable: skips lectures with an existing transcript.
"""
import argparse
import logging
import sys
from pathlib import Path
from faster_whisper import WhisperModel
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
log = logging.getLogger(__name__)
# Vorbirea normală dă ~100-150 cuvinte/minut. Sub 15% din asta înseamnă
# halucinație, tăcere, sau o scriere care a eșuat — nu o transcriere bună.
MIN_WORDS_PER_SECOND = 100 / 60 * 0.15
def transcribe_file(model: WhisperModel, audio_path: Path, txt_path: Path) -> bool:
"""Întoarce True doar dacă transcrierea rezultată pare completă.
Apelantul șterge audio-ul DOAR pe True — altfel un fișier pierdut
înseamnă re-descărcare, nu doar re-transcriere."""
segments, info = model.transcribe(str(audio_path), language="ro", beam_size=5, vad_filter=True)
txt_path.parent.mkdir(parents=True, exist_ok=True)
with open(txt_path, "w", encoding="utf-8") as f:
for seg in segments:
f.write(seg.text.strip() + "\n")
if not txt_path.exists():
log.error(f" {txt_path} a dispărut în timpul scrierii — păstrez audio-ul")
return False
words = len(txt_path.read_text(encoding="utf-8").split())
expected = info.duration * MIN_WORDS_PER_SECOND
if words < expected:
log.error(f" {txt_path.name}: doar {words} cuvinte pentru {info.duration:.0f}s "
f"(minim așteptat {expected:.0f}) — păstrez audio-ul")
return False
log.info(f" Transcribed: {txt_path.name} (lang={info.language}, duration={info.duration:.0f}s, {words} cuvinte)")
return True
def main():
p = argparse.ArgumentParser(description="Transcribe ATM course audio")
p.add_argument("--audio-dir", default="audio")
p.add_argument("--out", default="transcripts")
p.add_argument("--model", default="medium", help="faster-whisper model size")
p.add_argument("--only", default=None, help="Limit to one module dir name (for piloting)")
p.add_argument("--keep-audio", action="store_true", help="Don't delete source mp3 after transcribing")
args = p.parse_args()
audio_dir = Path(args.audio_dir)
out_dir = Path(args.out)
mp3_files = sorted(audio_dir.glob("*/*.mp3"))
if args.only:
mp3_files = [f for f in mp3_files if f.parent.name == args.only]
if not mp3_files:
log.error(f"No audio files found under {audio_dir}")
sys.exit(1)
log.info(f"Loading faster-whisper model '{args.model}' (CPU, int8)...")
model = WhisperModel(args.model, device="cpu", compute_type="int8")
done = skipped = 0
for audio_path in mp3_files:
txt_path = out_dir / audio_path.parent.name / (audio_path.stem + ".txt")
if txt_path.exists() and txt_path.stat().st_size > 50:
log.info(f" Skipping (exists): {txt_path.name}")
skipped += 1
continue
if not audio_path.exists():
# lista de fișiere se face o singură dată la start; dacă un fișier
# e mutat între timp (rebalansare între mașini), nu opri toată coada
log.warning(f" A dispărut între timp, sar peste: {audio_path}")
continue
log.info(f"Transcribing: {audio_path}")
ok = transcribe_file(model, audio_path, txt_path)
if ok and not args.keep_audio:
audio_path.unlink()
done += 1
log.info("=" * 60)
log.info(f"Transcribed {done}, skipped {skipped}.")
if __name__ == "__main__":
main()