#!/usr/bin/env bash # Descarcă un video (Facebook, YouTube etc.), extrage audio, transcrie cu Whisper. # # Usage: ./transcribe_video.sh [language] [--save-kb] [--bg] [--notify ] # # language cod ISO (ro, en, ...) sau "auto" (implicit) pentru detecție automată. # NU forța limba dacă nu ești sigur: `ro` pe audio englezesc face # Whisper să halucineze text incoerent (incident 2026-08-31). # --save-kb scrie notița în memory/kb// și reindexează. # --bg rulează detașat și întoarce imediat controlul. Obligatoriu când # scriptul e chemat dintr-un turn de chat: turnul are timeout de # 300s (DEFAULT_TIMEOUT, src/claude_session.py) iar un video lung # îl depășește, procesul e omorât la mijloc și userul nu primește # nimic. Cu --bg turnul răspunde imediat, jobul continuă singur. # --notify channel ID Discord unde se raportează finalizarea (util cu --bg). # # Exemple: # ./transcribe_video.sh "https://www.facebook.com/share/v/1EdPt3q2sq/" # ./transcribe_video.sh "https://www.facebook.com/share/r/1akfPJYvTw/" ro --save-kb # ./transcribe_video.sh "https://youtu.be/xyz" auto --save-kb --bg --notify 1471916752119009432 set -euo pipefail PROJECT_ROOT="$(cd "$(dirname "$0")/.." && pwd)" # Whisper trăiește în venv, nu în python-ul de sistem. Scriptul mergea înainte # doar din noroc, când era chemat dintr-un shell cu venv-ul deja activat. PY="$PROJECT_ROOT/.venv/bin/python3" KB_DIR="$PROJECT_ROOT/memory/kb" LOG_DIR="$PROJECT_ROOT/logs/transcribe" URL="" LANG="auto" SAVE_KB=0 BG=0 NOTIFY="" # --- parse argumente (poziționale: URL, apoi limba; restul flag-uri) --- POSITIONAL=() while [[ $# -gt 0 ]]; do case "$1" in --save-kb) SAVE_KB=1; shift ;; --bg) BG=1; shift ;; --notify) NOTIFY="${2:-}"; shift 2 ;; -h|--help) sed -n '2,20p' "$0"; exit 0 ;; *) POSITIONAL+=("$1"); shift ;; esac done URL="${POSITIONAL[0]:-}" LANG="${POSITIONAL[1]:-auto}" if [[ -z "$URL" ]]; then echo "Usage: $0 [language|auto] [--save-kb] [--bg] [--notify ]" exit 1 fi export PATH="/home/moltbot/bin:$PATH" # --- mod background: re-exec detașat, întoarce imediat --- if [[ "$BG" == "1" ]]; then mkdir -p "$LOG_DIR" JOB_ID="$(date +%Y%m%d-%H%M%S)-$$" JOB_LOG="$LOG_DIR/$JOB_ID.log" ARGS=("$URL" "$LANG") [[ "$SAVE_KB" == "1" ]] && ARGS+=(--save-kb) [[ -n "$NOTIFY" ]] && ARGS+=(--notify "$NOTIFY") setsid nohup "$0" "${ARGS[@]}" > "$JOB_LOG" 2>&1 & echo "→ Job pornit în background: $JOB_ID (pid $!)" echo " Log: $JOB_LOG" [[ -n "$NOTIFY" ]] && echo " Raportez pe canalul $NOTIFY la final." exit 0 fi # --- curăță workdir-uri orfane (trap-ul EXIT nu rulează la SIGKILL) --- find /tmp -maxdepth 1 -type d -name 'transcribe_*' -mmin +60 -exec rm -rf {} + 2>/dev/null || true WORKDIR="/tmp/transcribe_$$" mkdir -p "$WORKDIR" trap 'rm -rf "$WORKDIR"' EXIT notify() { [[ -z "$NOTIFY" ]] && return 0 "$PY" "$PROJECT_ROOT/tools/discord_send_file.py" \ --channel "$NOTIFY" --text "$1" >/dev/null 2>&1 || true } fail() { echo "Eroare: $1" notify "❌ Transcriere eșuată pentru $URL — $1" exit 1 } echo "→ Obțin informații video..." # JSON-ul merge pe disc, nu printr-o variabilă de shell: descrierile Facebook conțin # caractere de control și emoji care corup variabila (zsh: "character not in range"), # iar `|| echo Unknown` din versiunea veche masca eșecul în loc să-l semnaleze. yt-dlp "$URL" --dump-json --no-download -q > "$WORKDIR/info.json" 2>/dev/null || echo '{}' > "$WORKDIR/info.json" # Titlul, creatorul și durata într-o singură trecere. # Facebook împachetează titlul ca "830K views · 15K reactions | Titlul real | Pagina | ..."; # luarea oarbă a primului segment producea notițe numite "807k-views-15k-reactions" # (incident 2026-08-31), deci sărim segmentele de statistici și pe cel egal cu numele paginii. eval "$("$PY" - "$WORKDIR/info.json" <<'PYMETA' import json, re, shlex, sys try: d = json.load(open(sys.argv[1], encoding="utf-8")) except Exception: d = {} creator = (d.get("uploader") or d.get("channel") or "").strip() secs = int(d.get("duration") or 0) duration = f"{secs // 60}:{secs % 60:02d}" if secs else "?" raw = (d.get("title") or "Unknown").strip() STATS = re.compile(r"^[\d.,]+\s*[KMB]?\s*(views|reactions|comments|shares|likes)\b", re.I) parts = [p.strip() for p in raw.split("|") if p.strip()] title = next( (p for p in parts if not STATS.match(p) and p.lower() != creator.lower() and len(p) > 3), parts[0] if parts else raw, ) title = " ".join(title.split()) if len(title) > 80: title = title[:77].rstrip() + "..." for name, val in (("TITLE_SHORT", title), ("CREATOR", creator), ("DURATION", duration)): print(f"{name}={shlex.quote(val)}") PYMETA )" echo "→ Descarc video: $TITLE_SHORT..." yt-dlp "$URL" -o "$WORKDIR/video.%(ext)s" --no-playlist -q || fail "descărcarea a eșuat" VIDEO_FILE=$(ls "$WORKDIR"/video.* 2>/dev/null | head -1) [[ -z "$VIDEO_FILE" ]] && fail "descărcarea a eșuat (niciun fișier)" echo "→ Extrag audio..." ffmpeg -i "$VIDEO_FILE" -vn -acodec pcm_s16le -ar 16000 -ac 1 "$WORKDIR/audio.wav" -y -loglevel error \ || fail "extragerea audio a eșuat" # cpu_threads = core-uri FIZICE; hyperthread-urile încetinesc int8 compute-bound. THREADS=$(lscpu -p=CORE 2>/dev/null | grep -v '^#' | sort -u | wc -l) [[ -z "$THREADS" || "$THREADS" -lt 1 ]] && THREADS=4 echo "→ Transcriu cu faster-whisper (model: small, int8, ${THREADS} threads, limbă: $LANG)..." # faster-whisper (declarat în requirements.txt) în locul openai-whisper: ~5x realtime # pe acest CPU vs. peste 300s pentru 9 minute de audio — exact ce depășea timeout-ul. "$PY" - "$WORKDIR/audio.wav" "$LANG" "$THREADS" > "$WORKDIR/transcript.txt" 2>"$WORKDIR/whisper.err" <<'PYEOF' || fail "transcrierea a eșuat ($(tail -1 "$WORKDIR/whisper.err" 2>/dev/null))" import sys from faster_whisper import WhisperModel wav, lang, threads = sys.argv[1], sys.argv[2], int(sys.argv[3]) model = WhisperModel("small", device="cpu", compute_type="int8", cpu_threads=threads) segments, info = model.transcribe(wav, language=None if lang == "auto" else lang, vad_filter=True) text = " ".join(s.text.strip() for s in segments) print(text) print(f"[detected_language={info.language}]", file=sys.stderr) PYEOF TRANSCRIPT=$(cat "$WORKDIR/transcript.txt") DETECTED=$(grep -oP 'detected_language=\K\w+' "$WORKDIR/whisper.err" 2>/dev/null || echo "$LANG") [[ -z "$TRANSCRIPT" ]] && fail "transcrierea a ieșit goală" echo "" echo "=== $TITLE_SHORT ===" echo "$TRANSCRIPT" echo "" echo "✓ Transcriere completă (limbă: $DETECTED)." if [[ "$SAVE_KB" == "1" ]]; then DATE=$(date +%Y-%m-%d) SLUG=$(echo "$TITLE_SHORT" | "$PY" -c " import sys, re, unicodedata s = unicodedata.normalize('NFD', sys.stdin.read().strip()) s = ''.join(c for c in s if unicodedata.category(c) != 'Mn').lower() print(re.sub(r'[^a-z0-9]+', '-', s).strip('-')[:50]) ") [[ -z "$SLUG" ]] && SLUG="video-$(date +%H%M%S)" if echo "$URL" | grep -qi "facebook\.com"; then CATEGORY="facebook"; FORMAT="Reel (~${DURATION} min)" elif echo "$URL" | grep -qi "youtube\.com\|youtu\.be"; then CATEGORY="youtube"; FORMAT="Video (~${DURATION} min)" else CATEGORY="media"; FORMAT="Video (~${DURATION} min)" fi NOTE_DIR="$KB_DIR/$CATEGORY" mkdir -p "$NOTE_DIR" NOTE_FILE="$NOTE_DIR/${DATE}_${SLUG}.md" cat > "$NOTE_FILE" < --- ## Transcrierea $TRANSCRIPT NOTEEOF echo "" echo "→ Notiță salvată: $NOTE_FILE" echo "→ Reindexez KB..." "$PY" "$PROJECT_ROOT/tools/update_notes_index.py" >/dev/null KB_LINK="https://moltbot.tailf7372d.ts.net/echo/files.html#memory/kb/$CATEGORY/${DATE}_${SLUG}.md" echo "✓ KB actualizat. Link: $KB_LINK" notify "✅ Transcriere gata: **$TITLE_SHORT** (${DURATION} min, $DETECTED) $KB_LINK ⚠️ TL;DR-ul e gol — completează-l." else notify "✅ Transcriere gata: **$TITLE_SHORT** (${DURATION} min, $DETECTED). Rulat fără --save-kb." fi