diff --git a/cron/jobs.json b/cron/jobs.json index be8eafc..fee9dc8 100644 --- a/cron/jobs.json +++ b/cron/jobs.json @@ -11,9 +11,9 @@ "report_on": "changes", "timeout": 120, "enabled": true, - "last_run": "2026-08-21T16:00:00.000757+00:00", + "last_run": "2026-08-27T16:00:00.002423+00:00", "last_status": "ok", - "next_run": "2026-08-22T10:00:00+00:00" + "next_run": "2026-08-28T10:00:00+00:00" }, { "name": "security-audit-daily", @@ -43,9 +43,9 @@ "report_on": "never", "timeout": 120, "enabled": true, - "last_run": "2026-08-22T03:30:00.002627+00:00", + "last_run": "2026-08-27T03:30:00.002353+00:00", "last_status": "ok", - "next_run": "2026-08-23T03:30:00+00:00" + "next_run": "2026-08-28T03:30:00+00:00" }, { "name": "archive-tasks-daily", @@ -59,9 +59,9 @@ "report_on": "changes", "timeout": 60, "enabled": true, - "last_run": "2026-08-22T03:00:00.001220+00:00", + "last_run": "2026-08-27T03:00:00.001621+00:00", "last_status": "ok", - "next_run": "2026-08-23T03:00:00+00:00" + "next_run": "2026-08-28T03:00:00+00:00" }, { "name": "backup-config", @@ -75,9 +75,9 @@ "report_on": "never", "timeout": 120, "enabled": true, - "last_run": "2026-08-22T02:00:00.001599+00:00", + "last_run": "2026-08-27T02:00:00.000911+00:00", "last_status": "ok", - "next_run": "2026-08-23T02:00:00+00:00" + "next_run": "2026-08-28T02:00:00+00:00" }, { "name": "insights-extract", @@ -243,9 +243,9 @@ "prompt": "Heartbeat check. Rulează src/heartbeat.py printr-un scurt raport de status.\nDacă nu e nimic de raportat (email=0, calendar nu are evenimente <2h, kb ok), răspunde doar cu HEARTBEAT_OK și oprește-te — nu trimite mesaj.\nDacă e ceva: raport scurt pe Discord #echo-work.", "allowed_tools": [], "enabled": true, - "last_run": "2026-08-22T06:00:00.001316+00:00", + "last_run": "2026-08-27T14:00:00.002158+00:00", "last_status": "ok", - "next_run": "2026-08-22T10:00:00+00:00" + "next_run": "2026-08-27T18:00:00+00:00" }, { "name": "night-execute", diff --git a/dashboard/api.py b/dashboard/api.py index 3b5dff1..6c7aab2 100644 --- a/dashboard/api.py +++ b/dashboard/api.py @@ -173,6 +173,8 @@ class TaskBoardHandler( self.handle_cron_status() elif self.path == '/api/habits': self.handle_habits_get() + elif self.path.startswith('/api/files/download'): + self.handle_files_download() elif self.path.startswith('/api/files'): self.handle_files_get() elif self.path.startswith('/api/diff'): diff --git a/dashboard/handlers/files.py b/dashboard/handlers/files.py index 6ab7d4b..6928a4c 100644 --- a/dashboard/handlers/files.py +++ b/dashboard/handlers/files.py @@ -72,6 +72,42 @@ class FilesHandlers: except Exception as e: self.send_json({'error': str(e)}, 500) + def handle_files_download(self): + """Stream a binary file (audio, zip, etc.) from WORKSPACE_DIR as a download. + + Dedicated resolution (not `_resolve_sandboxed`, which always resolves + against ALLOWED_WORKSPACES[0]/BASE_DIR and never reaches WORKSPACE_DIR) + so this stays isolated from the shared file-browser sandbox logic. + """ + params = parse_qs(urlparse(self.path).query) + path = params.get('path', [''])[0] + + try: + target = (constants.WORKSPACE_DIR / path).resolve() + target.relative_to(constants.WORKSPACE_DIR.resolve()) + except (ValueError, OSError): + self.send_json({'error': 'Access denied'}, 403) + return + if not target.is_file(): + self.send_json({'error': 'Not found'}, 404) + return + + ext = target.suffix.lstrip('.').lower() + ctype = { + 'mp3': 'audio/mpeg', + 'wav': 'audio/wav', + 'zip': 'application/zip', + }.get(ext, 'application/octet-stream') + + data = target.read_bytes() + self.send_response(200) + self.send_header('Content-Type', ctype) + self.send_header('Content-Length', str(len(data))) + self.send_header('Content-Disposition', f'attachment; filename="{target.name}"') + self.send_header('Cache-Control', 'public, max-age=3600') + self.end_headers() + self.wfile.write(data) + def handle_files_post(self): """Save file content.""" try: diff --git a/memory/kb/index.json b/memory/kb/index.json index 7b192ad..4519c6d 100644 --- a/memory/kb/index.json +++ b/memory/kb/index.json @@ -1,5 +1,18 @@ { "notes": [ + { + "file": "notes-data/tools/infrastructure.md", + "title": "Infrastructură (Proxmox + Docker)", + "date": "2026-08-23", + "tags": [], + "domains": [], + "types": [], + "category": "tools", + "project": null, + "subdir": null, + "video": "", + "tldr": "- Orice operație distructivă" + }, { "file": "notes-data/projects/lfm2.5-230m-summarization-eval.md", "title": "LFM2.5-230M — evaluare ca preprocesor de rezumat / fallback conversațional (2026-08-21)", @@ -1444,19 +1457,6 @@ "video": "", "tldr": "*Instrumentale clasice și cinematic — Einaudi, Vangelis, Hisaishi, Tiersen, Secret Garden.*" }, - { - "file": "notes-data/tools/infrastructure.md", - "title": "Infrastructură (Proxmox + Docker)", - "date": "2026-04-26", - "tags": [], - "domains": [], - "types": [], - "category": "tools", - "project": null, - "subdir": null, - "video": "", - "tldr": "- Orice operație distructivă" - }, { "file": "notes-data/coaching/2026-04-25-negativity-bias-reframing.md", "title": "Negativity Bias & Positive Reframing", diff --git a/tools/epub_to_audio.py b/tools/epub_to_audio.py new file mode 100644 index 0000000..94c591f --- /dev/null +++ b/tools/epub_to_audio.py @@ -0,0 +1,146 @@ +#!/usr/bin/env python3 +"""EPUB -> audio cu una din vocile din tools/tts.py. + +CLI: + python3 tools/epub_to_audio.py --voice "Marius 4" --out-dir DIR + [--chapters N] [--start-chapter N] + +Per capitol: extrage text curat (ebooklib + BeautifulSoup), îl împarte în bucăți +sigure pentru TTS (pe propoziții, sub _CHUNK_CHARS), sintetizează secvențial cu +tools.tts.synthesize, apoi concatenează cu ffmpeg într-un singur WAV per capitol. +""" + +import argparse +import json +import re +import subprocess +import sys +from pathlib import Path + +from bs4 import BeautifulSoup +from ebooklib import epub, ITEM_DOCUMENT + +REPO_ROOT = Path(__file__).resolve().parent.parent +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from tools.tts import synthesize # noqa: E402 + +# pocket-tts n-are un hard limit documentat ca Supertonic, dar bucăți mari cresc +# riscul de timeout (60s) și degradarea calității pe propoziții foarte lungi. +_CHUNK_CHARS = 600 + + +def extract_chapters(epub_path: Path) -> list[dict]: + book = epub.read_epub(str(epub_path)) + chapters = [] + for item in book.get_items_of_type(ITEM_DOCUMENT): + soup = BeautifulSoup(item.get_content(), "html.parser") + for tag in soup(["script", "style"]): + tag.decompose() + text = soup.get_text(separator=" ", strip=True) + text = re.sub(r"\s+", " ", text).strip() + if len(text) < 200: + continue # sar peste pagini de copertă/titlu/ToC, prea puțin conținut + title_tag = soup.find(["h1", "h2", "title"]) + title = title_tag.get_text(strip=True) if title_tag else item.get_name() + chapters.append({"name": item.get_name(), "title": title, "text": text}) + return chapters + + +def chunk_text(text: str, max_chars: int = _CHUNK_CHARS) -> list[str]: + sentences = re.split(r"(?<=[.!?])\s+", text) + chunks = [] + current = "" + for sentence in sentences: + if current and len(current) + 1 + len(sentence) > max_chars: + chunks.append(current) + current = sentence + else: + current = f"{current} {sentence}".strip() + if current: + chunks.append(current) + return chunks + + +def synthesize_chapter(text: str, voice: str, out_path: Path, mp3: bool = True) -> dict: + chunks = chunk_text(text) + wav_paths = [] + for i, chunk in enumerate(chunks): + result = synthesize(chunk, voice=voice, lang="en") + if not result.get("ok"): + return {"ok": False, "error": f"chunk {i}/{len(chunks)}: {result.get('error')}"} + wav_paths.append(result["path"]) + print(f" chunk {i + 1}/{len(chunks)} ok ({result['size_bytes']} bytes)", file=sys.stderr) + + if not wav_paths: + return {"ok": False, "error": "niciun chunk de sintetizat"} + + wav_path = out_path.with_suffix(".wav") + if len(wav_paths) == 1: + Path(wav_paths[0]).replace(wav_path) + else: + list_file = out_path.with_suffix(".concat.txt") + list_file.write_text("\n".join(f"file '{p}'" for p in wav_paths)) + subprocess.run( + ["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", str(list_file), "-c", "copy", str(wav_path)], + check=True, + capture_output=True, + ) + list_file.unlink() + for p in wav_paths: + Path(p).unlink(missing_ok=True) + + if not mp3: + return {"ok": True, "path": str(wav_path), "chunks": len(chunks)} + + subprocess.run( + ["ffmpeg", "-y", "-i", str(wav_path), "-ac", "1", "-b:a", "96k", str(out_path)], + check=True, + capture_output=True, + ) + wav_path.unlink() + return {"ok": True, "path": str(out_path), "chunks": len(chunks)} + + +_SKIP_TITLES = {"contents"} +_MIN_CHAPTER_CHARS = 500 # sub asta e de regulă pagină de titlu, nu conținut de citit + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("epub_path") + parser.add_argument("--voice", default="Marius 4") + parser.add_argument("--out-dir", default="/home/moltbot/workspace/epub-audio") + parser.add_argument("--chapters", type=int, default=None, help="limitează la primele N capitole") + parser.add_argument("--start-chapter", type=int, default=0) + parser.add_argument("--wav", action="store_true", help="păstrează WAV în loc de MP3") + args = parser.parse_args() + + out_dir = Path(args.out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + ext = "wav" if args.wav else "mp3" + + chapters = extract_chapters(Path(args.epub_path)) + chapters = [ + ch for ch in chapters + if len(ch["text"]) >= _MIN_CHAPTER_CHARS and ch["title"].strip().lower() not in _SKIP_TITLES + ] + print(f"{len(chapters)} capitole de citit (după filtrare titlu/ToC)", file=sys.stderr) + + end = args.start_chapter + args.chapters if args.chapters else len(chapters) + selected = chapters[args.start_chapter:end] + + results = [] + for n, ch in enumerate(selected, start=args.start_chapter + 1): + out_path = out_dir / f"{n:02d}_{re.sub(r'[^a-zA-Z0-9]+', '_', ch['title'])[:40]}.{ext}" + print(f"[{n}] {ch['title']} ({len(ch['text'])} chars) -> {out_path.name}", file=sys.stderr) + result = synthesize_chapter(ch["text"], args.voice, out_path, mp3=not args.wav) + results.append({"chapter": ch["title"], **result}) + print(json.dumps(results[-1]), file=sys.stderr) + + print(json.dumps({"ok": True, "results": results})) + + +if __name__ == "__main__": + main()