feat(dashboard): endpoint de download fișiere binare + tool epub-to-audio

- dashboard/handlers/files.py: handle_files_download() servește mp3/wav/zip
  din WORKSPACE_DIR cu Content-Disposition attachment, resolve dedicat
  (nu _resolve_sandboxed, care nu ajunge niciodată la WORKSPACE_DIR)
- dashboard/api.py: rutează /api/files/download
- tools/epub_to_audio.py: tool nou de conversie EPUB → audio
- cron/jobs.json, memory/kb/index.json: stare auto-generată (job runs,
  regenerare index KB)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-27 17:50:27 +00:00
parent 164ff32031
commit bf62d6fb4b
5 changed files with 207 additions and 23 deletions

View File

@@ -11,9 +11,9 @@
"report_on": "changes",
"timeout": 120,
"enabled": true,
"last_run": "2026-08-21T16:00:00.000757+00:00",
"last_run": "2026-08-27T16:00:00.002423+00:00",
"last_status": "ok",
"next_run": "2026-08-22T10:00:00+00:00"
"next_run": "2026-08-28T10:00:00+00:00"
},
{
"name": "security-audit-daily",
@@ -43,9 +43,9 @@
"report_on": "never",
"timeout": 120,
"enabled": true,
"last_run": "2026-08-22T03:30:00.002627+00:00",
"last_run": "2026-08-27T03:30:00.002353+00:00",
"last_status": "ok",
"next_run": "2026-08-23T03:30:00+00:00"
"next_run": "2026-08-28T03:30:00+00:00"
},
{
"name": "archive-tasks-daily",
@@ -59,9 +59,9 @@
"report_on": "changes",
"timeout": 60,
"enabled": true,
"last_run": "2026-08-22T03:00:00.001220+00:00",
"last_run": "2026-08-27T03:00:00.001621+00:00",
"last_status": "ok",
"next_run": "2026-08-23T03:00:00+00:00"
"next_run": "2026-08-28T03:00:00+00:00"
},
{
"name": "backup-config",
@@ -75,9 +75,9 @@
"report_on": "never",
"timeout": 120,
"enabled": true,
"last_run": "2026-08-22T02:00:00.001599+00:00",
"last_run": "2026-08-27T02:00:00.000911+00:00",
"last_status": "ok",
"next_run": "2026-08-23T02:00:00+00:00"
"next_run": "2026-08-28T02:00:00+00:00"
},
{
"name": "insights-extract",
@@ -243,9 +243,9 @@
"prompt": "Heartbeat check. Rulează src/heartbeat.py printr-un scurt raport de status.\nDacă nu e nimic de raportat (email=0, calendar nu are evenimente <2h, kb ok), răspunde doar cu HEARTBEAT_OK și oprește-te — nu trimite mesaj.\nDacă e ceva: raport scurt pe Discord #echo-work.",
"allowed_tools": [],
"enabled": true,
"last_run": "2026-08-22T06:00:00.001316+00:00",
"last_run": "2026-08-27T14:00:00.002158+00:00",
"last_status": "ok",
"next_run": "2026-08-22T10:00:00+00:00"
"next_run": "2026-08-27T18:00:00+00:00"
},
{
"name": "night-execute",

View File

@@ -173,6 +173,8 @@ class TaskBoardHandler(
self.handle_cron_status()
elif self.path == '/api/habits':
self.handle_habits_get()
elif self.path.startswith('/api/files/download'):
self.handle_files_download()
elif self.path.startswith('/api/files'):
self.handle_files_get()
elif self.path.startswith('/api/diff'):

View File

@@ -72,6 +72,42 @@ class FilesHandlers:
except Exception as e:
self.send_json({'error': str(e)}, 500)
def handle_files_download(self):
"""Stream a binary file (audio, zip, etc.) from WORKSPACE_DIR as a download.
Dedicated resolution (not `_resolve_sandboxed`, which always resolves
against ALLOWED_WORKSPACES[0]/BASE_DIR and never reaches WORKSPACE_DIR)
so this stays isolated from the shared file-browser sandbox logic.
"""
params = parse_qs(urlparse(self.path).query)
path = params.get('path', [''])[0]
try:
target = (constants.WORKSPACE_DIR / path).resolve()
target.relative_to(constants.WORKSPACE_DIR.resolve())
except (ValueError, OSError):
self.send_json({'error': 'Access denied'}, 403)
return
if not target.is_file():
self.send_json({'error': 'Not found'}, 404)
return
ext = target.suffix.lstrip('.').lower()
ctype = {
'mp3': 'audio/mpeg',
'wav': 'audio/wav',
'zip': 'application/zip',
}.get(ext, 'application/octet-stream')
data = target.read_bytes()
self.send_response(200)
self.send_header('Content-Type', ctype)
self.send_header('Content-Length', str(len(data)))
self.send_header('Content-Disposition', f'attachment; filename="{target.name}"')
self.send_header('Cache-Control', 'public, max-age=3600')
self.end_headers()
self.wfile.write(data)
def handle_files_post(self):
"""Save file content."""
try:

View File

@@ -1,5 +1,18 @@
{
"notes": [
{
"file": "notes-data/tools/infrastructure.md",
"title": "Infrastructură (Proxmox + Docker)",
"date": "2026-08-23",
"tags": [],
"domains": [],
"types": [],
"category": "tools",
"project": null,
"subdir": null,
"video": "",
"tldr": "- Orice operație distructivă"
},
{
"file": "notes-data/projects/lfm2.5-230m-summarization-eval.md",
"title": "LFM2.5-230M — evaluare ca preprocesor de rezumat / fallback conversațional (2026-08-21)",
@@ -1444,19 +1457,6 @@
"video": "",
"tldr": "*Instrumentale clasice și cinematic — Einaudi, Vangelis, Hisaishi, Tiersen, Secret Garden.*"
},
{
"file": "notes-data/tools/infrastructure.md",
"title": "Infrastructură (Proxmox + Docker)",
"date": "2026-04-26",
"tags": [],
"domains": [],
"types": [],
"category": "tools",
"project": null,
"subdir": null,
"video": "",
"tldr": "- Orice operație distructivă"
},
{
"file": "notes-data/coaching/2026-04-25-negativity-bias-reframing.md",
"title": "Negativity Bias & Positive Reframing",

146
tools/epub_to_audio.py Normal file
View File

@@ -0,0 +1,146 @@
#!/usr/bin/env python3
"""EPUB -> audio cu una din vocile din tools/tts.py.
CLI:
python3 tools/epub_to_audio.py <epub_path> --voice "Marius 4" --out-dir DIR
[--chapters N] [--start-chapter N]
Per capitol: extrage text curat (ebooklib + BeautifulSoup), îl împarte în bucăți
sigure pentru TTS (pe propoziții, sub _CHUNK_CHARS), sintetizează secvențial cu
tools.tts.synthesize, apoi concatenează cu ffmpeg într-un singur WAV per capitol.
"""
import argparse
import json
import re
import subprocess
import sys
from pathlib import Path
from bs4 import BeautifulSoup
from ebooklib import epub, ITEM_DOCUMENT
REPO_ROOT = Path(__file__).resolve().parent.parent
if str(REPO_ROOT) not in sys.path:
sys.path.insert(0, str(REPO_ROOT))
from tools.tts import synthesize # noqa: E402
# pocket-tts n-are un hard limit documentat ca Supertonic, dar bucăți mari cresc
# riscul de timeout (60s) și degradarea calității pe propoziții foarte lungi.
_CHUNK_CHARS = 600
def extract_chapters(epub_path: Path) -> list[dict]:
book = epub.read_epub(str(epub_path))
chapters = []
for item in book.get_items_of_type(ITEM_DOCUMENT):
soup = BeautifulSoup(item.get_content(), "html.parser")
for tag in soup(["script", "style"]):
tag.decompose()
text = soup.get_text(separator=" ", strip=True)
text = re.sub(r"\s+", " ", text).strip()
if len(text) < 200:
continue # sar peste pagini de copertă/titlu/ToC, prea puțin conținut
title_tag = soup.find(["h1", "h2", "title"])
title = title_tag.get_text(strip=True) if title_tag else item.get_name()
chapters.append({"name": item.get_name(), "title": title, "text": text})
return chapters
def chunk_text(text: str, max_chars: int = _CHUNK_CHARS) -> list[str]:
sentences = re.split(r"(?<=[.!?])\s+", text)
chunks = []
current = ""
for sentence in sentences:
if current and len(current) + 1 + len(sentence) > max_chars:
chunks.append(current)
current = sentence
else:
current = f"{current} {sentence}".strip()
if current:
chunks.append(current)
return chunks
def synthesize_chapter(text: str, voice: str, out_path: Path, mp3: bool = True) -> dict:
chunks = chunk_text(text)
wav_paths = []
for i, chunk in enumerate(chunks):
result = synthesize(chunk, voice=voice, lang="en")
if not result.get("ok"):
return {"ok": False, "error": f"chunk {i}/{len(chunks)}: {result.get('error')}"}
wav_paths.append(result["path"])
print(f" chunk {i + 1}/{len(chunks)} ok ({result['size_bytes']} bytes)", file=sys.stderr)
if not wav_paths:
return {"ok": False, "error": "niciun chunk de sintetizat"}
wav_path = out_path.with_suffix(".wav")
if len(wav_paths) == 1:
Path(wav_paths[0]).replace(wav_path)
else:
list_file = out_path.with_suffix(".concat.txt")
list_file.write_text("\n".join(f"file '{p}'" for p in wav_paths))
subprocess.run(
["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", str(list_file), "-c", "copy", str(wav_path)],
check=True,
capture_output=True,
)
list_file.unlink()
for p in wav_paths:
Path(p).unlink(missing_ok=True)
if not mp3:
return {"ok": True, "path": str(wav_path), "chunks": len(chunks)}
subprocess.run(
["ffmpeg", "-y", "-i", str(wav_path), "-ac", "1", "-b:a", "96k", str(out_path)],
check=True,
capture_output=True,
)
wav_path.unlink()
return {"ok": True, "path": str(out_path), "chunks": len(chunks)}
_SKIP_TITLES = {"contents"}
_MIN_CHAPTER_CHARS = 500 # sub asta e de regulă pagină de titlu, nu conținut de citit
def main():
parser = argparse.ArgumentParser()
parser.add_argument("epub_path")
parser.add_argument("--voice", default="Marius 4")
parser.add_argument("--out-dir", default="/home/moltbot/workspace/epub-audio")
parser.add_argument("--chapters", type=int, default=None, help="limitează la primele N capitole")
parser.add_argument("--start-chapter", type=int, default=0)
parser.add_argument("--wav", action="store_true", help="păstrează WAV în loc de MP3")
args = parser.parse_args()
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
ext = "wav" if args.wav else "mp3"
chapters = extract_chapters(Path(args.epub_path))
chapters = [
ch for ch in chapters
if len(ch["text"]) >= _MIN_CHAPTER_CHARS and ch["title"].strip().lower() not in _SKIP_TITLES
]
print(f"{len(chapters)} capitole de citit (după filtrare titlu/ToC)", file=sys.stderr)
end = args.start_chapter + args.chapters if args.chapters else len(chapters)
selected = chapters[args.start_chapter:end]
results = []
for n, ch in enumerate(selected, start=args.start_chapter + 1):
out_path = out_dir / f"{n:02d}_{re.sub(r'[^a-zA-Z0-9]+', '_', ch['title'])[:40]}.{ext}"
print(f"[{n}] {ch['title']} ({len(ch['text'])} chars) -> {out_path.name}", file=sys.stderr)
result = synthesize_chapter(ch["text"], args.voice, out_path, mp3=not args.wav)
results.append({"chapter": ch["title"], **result})
print(json.dumps(results[-1]), file=sys.stderr)
print(json.dumps({"ok": True, "results": results}))
if __name__ == "__main__":
main()