Proiect descărcare + transcriere curs ATM (trading, Bogdan Jinga)
Pipeline audio-only (fără video pe disc) prin ffmpeg direct din URL CloudFront, transcriere cu faster-whisper (CPU), sumarizări structurate cu diagrame SVG la scară reală (STYLE.md documentează toate cerințele de format).
This commit is contained in:
68
transcribe.py
Normal file
68
transcribe.py
Normal file
@@ -0,0 +1,68 @@
|
||||
"""
|
||||
Transcribe audio/<module>/*.mp3 -> transcripts/<module>/*.txt using
|
||||
faster-whisper (CPU, CTranslate2). Deletes the source audio after a
|
||||
successful transcript (nobody wants the audio kept — text only).
|
||||
Resumable: skips lectures with an existing transcript.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def transcribe_file(model: WhisperModel, audio_path: Path, txt_path: Path) -> None:
|
||||
segments, info = model.transcribe(str(audio_path), language="ro", beam_size=5, vad_filter=True)
|
||||
txt_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(txt_path, "w", encoding="utf-8") as f:
|
||||
for seg in segments:
|
||||
f.write(seg.text.strip() + "\n")
|
||||
log.info(f" Transcribed: {txt_path.name} (lang={info.language}, duration={info.duration:.0f}s)")
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description="Transcribe ATM course audio")
|
||||
p.add_argument("--audio-dir", default="audio")
|
||||
p.add_argument("--out", default="transcripts")
|
||||
p.add_argument("--model", default="medium", help="faster-whisper model size")
|
||||
p.add_argument("--only", default=None, help="Limit to one module dir name (for piloting)")
|
||||
p.add_argument("--keep-audio", action="store_true", help="Don't delete source mp3 after transcribing")
|
||||
args = p.parse_args()
|
||||
|
||||
audio_dir = Path(args.audio_dir)
|
||||
out_dir = Path(args.out)
|
||||
mp3_files = sorted(audio_dir.glob("*/*.mp3"))
|
||||
if args.only:
|
||||
mp3_files = [f for f in mp3_files if f.parent.name == args.only]
|
||||
if not mp3_files:
|
||||
log.error(f"No audio files found under {audio_dir}")
|
||||
sys.exit(1)
|
||||
|
||||
log.info(f"Loading faster-whisper model '{args.model}' (CPU, int8)...")
|
||||
model = WhisperModel(args.model, device="cpu", compute_type="int8")
|
||||
|
||||
done = skipped = 0
|
||||
for audio_path in mp3_files:
|
||||
txt_path = out_dir / audio_path.parent.name / (audio_path.stem + ".txt")
|
||||
if txt_path.exists() and txt_path.stat().st_size > 50:
|
||||
log.info(f" Skipping (exists): {txt_path.name}")
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
log.info(f"Transcribing: {audio_path}")
|
||||
transcribe_file(model, audio_path, txt_path)
|
||||
if not args.keep_audio:
|
||||
audio_path.unlink()
|
||||
done += 1
|
||||
|
||||
log.info("=" * 60)
|
||||
log.info(f"Transcribed {done}, skipped {skipped}.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user