Fix: transcribe.py nu mai sterge audio daca transcrierea nu pare completa

Cauza: am rulat `find transcripts -size -1k -delete` cat transcribe.py scria
in acelasi director. faster_whisper scrie bufferat, deci o transcriere de o
ora sta sub 1KB minute in sir; find a sters fisierul de sub proces, Python a
logat "Transcribed" pe un inode disparut, si audio-ul a fost sters. Lectia
11 (25.11.2025) pierduta complet, re-descarcata.

- transcribe_file() intoarce bool, verifica nr. cuvinte vs durata audio
- pull_moltbot.sh: staging + mutare doar peste 1KB, in loc de find -delete
- PROGRESS.md: documentat, ca sa nu se repete

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MtSTyTmt6AbCL9j5ajEDm1
This commit is contained in:
Claude Agent
2026-09-12 18:35:06 +00:00
parent bde96649e9
commit c28e08b3aa
4 changed files with 68 additions and 4 deletions

View File

@@ -16,13 +16,32 @@ logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(me
log = logging.getLogger(__name__)
def transcribe_file(model: WhisperModel, audio_path: Path, txt_path: Path) -> None:
# Vorbirea normală dă ~100-150 cuvinte/minut. Sub 15% din asta înseamnă
# halucinație, tăcere, sau o scriere care a eșuat — nu o transcriere bună.
MIN_WORDS_PER_SECOND = 100 / 60 * 0.15
def transcribe_file(model: WhisperModel, audio_path: Path, txt_path: Path) -> bool:
"""Întoarce True doar dacă transcrierea rezultată pare completă.
Apelantul șterge audio-ul DOAR pe True — altfel un fișier pierdut
înseamnă re-descărcare, nu doar re-transcriere."""
segments, info = model.transcribe(str(audio_path), language="ro", beam_size=5, vad_filter=True)
txt_path.parent.mkdir(parents=True, exist_ok=True)
with open(txt_path, "w", encoding="utf-8") as f:
for seg in segments:
f.write(seg.text.strip() + "\n")
log.info(f" Transcribed: {txt_path.name} (lang={info.language}, duration={info.duration:.0f}s)")
if not txt_path.exists():
log.error(f" {txt_path} a dispărut în timpul scrierii — păstrez audio-ul")
return False
words = len(txt_path.read_text(encoding="utf-8").split())
expected = info.duration * MIN_WORDS_PER_SECOND
if words < expected:
log.error(f" {txt_path.name}: doar {words} cuvinte pentru {info.duration:.0f}s "
f"(minim așteptat {expected:.0f}) — păstrez audio-ul")
return False
log.info(f" Transcribed: {txt_path.name} (lang={info.language}, duration={info.duration:.0f}s, {words} cuvinte)")
return True
def main():
@@ -55,8 +74,8 @@ def main():
continue
log.info(f"Transcribing: {audio_path}")
transcribe_file(model, audio_path, txt_path)
if not args.keep_audio:
ok = transcribe_file(model, audio_path, txt_path)
if ok and not args.keep_audio:
audio_path.unlink()
done += 1