feat(dashboard): endpoint de download fișiere binare + tool epub-to-audio
- dashboard/handlers/files.py: handle_files_download() servește mp3/wav/zip din WORKSPACE_DIR cu Content-Disposition attachment, resolve dedicat (nu _resolve_sandboxed, care nu ajunge niciodată la WORKSPACE_DIR) - dashboard/api.py: rutează /api/files/download - tools/epub_to_audio.py: tool nou de conversie EPUB → audio - cron/jobs.json, memory/kb/index.json: stare auto-generată (job runs, regenerare index KB) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -11,9 +11,9 @@
|
||||
"report_on": "changes",
|
||||
"timeout": 120,
|
||||
"enabled": true,
|
||||
"last_run": "2026-08-21T16:00:00.000757+00:00",
|
||||
"last_run": "2026-08-27T16:00:00.002423+00:00",
|
||||
"last_status": "ok",
|
||||
"next_run": "2026-08-22T10:00:00+00:00"
|
||||
"next_run": "2026-08-28T10:00:00+00:00"
|
||||
},
|
||||
{
|
||||
"name": "security-audit-daily",
|
||||
@@ -43,9 +43,9 @@
|
||||
"report_on": "never",
|
||||
"timeout": 120,
|
||||
"enabled": true,
|
||||
"last_run": "2026-08-22T03:30:00.002627+00:00",
|
||||
"last_run": "2026-08-27T03:30:00.002353+00:00",
|
||||
"last_status": "ok",
|
||||
"next_run": "2026-08-23T03:30:00+00:00"
|
||||
"next_run": "2026-08-28T03:30:00+00:00"
|
||||
},
|
||||
{
|
||||
"name": "archive-tasks-daily",
|
||||
@@ -59,9 +59,9 @@
|
||||
"report_on": "changes",
|
||||
"timeout": 60,
|
||||
"enabled": true,
|
||||
"last_run": "2026-08-22T03:00:00.001220+00:00",
|
||||
"last_run": "2026-08-27T03:00:00.001621+00:00",
|
||||
"last_status": "ok",
|
||||
"next_run": "2026-08-23T03:00:00+00:00"
|
||||
"next_run": "2026-08-28T03:00:00+00:00"
|
||||
},
|
||||
{
|
||||
"name": "backup-config",
|
||||
@@ -75,9 +75,9 @@
|
||||
"report_on": "never",
|
||||
"timeout": 120,
|
||||
"enabled": true,
|
||||
"last_run": "2026-08-22T02:00:00.001599+00:00",
|
||||
"last_run": "2026-08-27T02:00:00.000911+00:00",
|
||||
"last_status": "ok",
|
||||
"next_run": "2026-08-23T02:00:00+00:00"
|
||||
"next_run": "2026-08-28T02:00:00+00:00"
|
||||
},
|
||||
{
|
||||
"name": "insights-extract",
|
||||
@@ -243,9 +243,9 @@
|
||||
"prompt": "Heartbeat check. Rulează src/heartbeat.py printr-un scurt raport de status.\nDacă nu e nimic de raportat (email=0, calendar nu are evenimente <2h, kb ok), răspunde doar cu HEARTBEAT_OK și oprește-te — nu trimite mesaj.\nDacă e ceva: raport scurt pe Discord #echo-work.",
|
||||
"allowed_tools": [],
|
||||
"enabled": true,
|
||||
"last_run": "2026-08-22T06:00:00.001316+00:00",
|
||||
"last_run": "2026-08-27T14:00:00.002158+00:00",
|
||||
"last_status": "ok",
|
||||
"next_run": "2026-08-22T10:00:00+00:00"
|
||||
"next_run": "2026-08-27T18:00:00+00:00"
|
||||
},
|
||||
{
|
||||
"name": "night-execute",
|
||||
|
||||
@@ -173,6 +173,8 @@ class TaskBoardHandler(
|
||||
self.handle_cron_status()
|
||||
elif self.path == '/api/habits':
|
||||
self.handle_habits_get()
|
||||
elif self.path.startswith('/api/files/download'):
|
||||
self.handle_files_download()
|
||||
elif self.path.startswith('/api/files'):
|
||||
self.handle_files_get()
|
||||
elif self.path.startswith('/api/diff'):
|
||||
|
||||
@@ -72,6 +72,42 @@ class FilesHandlers:
|
||||
except Exception as e:
|
||||
self.send_json({'error': str(e)}, 500)
|
||||
|
||||
def handle_files_download(self):
|
||||
"""Stream a binary file (audio, zip, etc.) from WORKSPACE_DIR as a download.
|
||||
|
||||
Dedicated resolution (not `_resolve_sandboxed`, which always resolves
|
||||
against ALLOWED_WORKSPACES[0]/BASE_DIR and never reaches WORKSPACE_DIR)
|
||||
so this stays isolated from the shared file-browser sandbox logic.
|
||||
"""
|
||||
params = parse_qs(urlparse(self.path).query)
|
||||
path = params.get('path', [''])[0]
|
||||
|
||||
try:
|
||||
target = (constants.WORKSPACE_DIR / path).resolve()
|
||||
target.relative_to(constants.WORKSPACE_DIR.resolve())
|
||||
except (ValueError, OSError):
|
||||
self.send_json({'error': 'Access denied'}, 403)
|
||||
return
|
||||
if not target.is_file():
|
||||
self.send_json({'error': 'Not found'}, 404)
|
||||
return
|
||||
|
||||
ext = target.suffix.lstrip('.').lower()
|
||||
ctype = {
|
||||
'mp3': 'audio/mpeg',
|
||||
'wav': 'audio/wav',
|
||||
'zip': 'application/zip',
|
||||
}.get(ext, 'application/octet-stream')
|
||||
|
||||
data = target.read_bytes()
|
||||
self.send_response(200)
|
||||
self.send_header('Content-Type', ctype)
|
||||
self.send_header('Content-Length', str(len(data)))
|
||||
self.send_header('Content-Disposition', f'attachment; filename="{target.name}"')
|
||||
self.send_header('Cache-Control', 'public, max-age=3600')
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
def handle_files_post(self):
|
||||
"""Save file content."""
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,18 @@
|
||||
{
|
||||
"notes": [
|
||||
{
|
||||
"file": "notes-data/tools/infrastructure.md",
|
||||
"title": "Infrastructură (Proxmox + Docker)",
|
||||
"date": "2026-08-23",
|
||||
"tags": [],
|
||||
"domains": [],
|
||||
"types": [],
|
||||
"category": "tools",
|
||||
"project": null,
|
||||
"subdir": null,
|
||||
"video": "",
|
||||
"tldr": "- Orice operație distructivă"
|
||||
},
|
||||
{
|
||||
"file": "notes-data/projects/lfm2.5-230m-summarization-eval.md",
|
||||
"title": "LFM2.5-230M — evaluare ca preprocesor de rezumat / fallback conversațional (2026-08-21)",
|
||||
@@ -1444,19 +1457,6 @@
|
||||
"video": "",
|
||||
"tldr": "*Instrumentale clasice și cinematic — Einaudi, Vangelis, Hisaishi, Tiersen, Secret Garden.*"
|
||||
},
|
||||
{
|
||||
"file": "notes-data/tools/infrastructure.md",
|
||||
"title": "Infrastructură (Proxmox + Docker)",
|
||||
"date": "2026-04-26",
|
||||
"tags": [],
|
||||
"domains": [],
|
||||
"types": [],
|
||||
"category": "tools",
|
||||
"project": null,
|
||||
"subdir": null,
|
||||
"video": "",
|
||||
"tldr": "- Orice operație distructivă"
|
||||
},
|
||||
{
|
||||
"file": "notes-data/coaching/2026-04-25-negativity-bias-reframing.md",
|
||||
"title": "Negativity Bias & Positive Reframing",
|
||||
|
||||
146
tools/epub_to_audio.py
Normal file
146
tools/epub_to_audio.py
Normal file
@@ -0,0 +1,146 @@
|
||||
#!/usr/bin/env python3
|
||||
"""EPUB -> audio cu una din vocile din tools/tts.py.
|
||||
|
||||
CLI:
|
||||
python3 tools/epub_to_audio.py <epub_path> --voice "Marius 4" --out-dir DIR
|
||||
[--chapters N] [--start-chapter N]
|
||||
|
||||
Per capitol: extrage text curat (ebooklib + BeautifulSoup), îl împarte în bucăți
|
||||
sigure pentru TTS (pe propoziții, sub _CHUNK_CHARS), sintetizează secvențial cu
|
||||
tools.tts.synthesize, apoi concatenează cu ffmpeg într-un singur WAV per capitol.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from ebooklib import epub, ITEM_DOCUMENT
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from tools.tts import synthesize # noqa: E402
|
||||
|
||||
# pocket-tts n-are un hard limit documentat ca Supertonic, dar bucăți mari cresc
|
||||
# riscul de timeout (60s) și degradarea calității pe propoziții foarte lungi.
|
||||
_CHUNK_CHARS = 600
|
||||
|
||||
|
||||
def extract_chapters(epub_path: Path) -> list[dict]:
|
||||
book = epub.read_epub(str(epub_path))
|
||||
chapters = []
|
||||
for item in book.get_items_of_type(ITEM_DOCUMENT):
|
||||
soup = BeautifulSoup(item.get_content(), "html.parser")
|
||||
for tag in soup(["script", "style"]):
|
||||
tag.decompose()
|
||||
text = soup.get_text(separator=" ", strip=True)
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
if len(text) < 200:
|
||||
continue # sar peste pagini de copertă/titlu/ToC, prea puțin conținut
|
||||
title_tag = soup.find(["h1", "h2", "title"])
|
||||
title = title_tag.get_text(strip=True) if title_tag else item.get_name()
|
||||
chapters.append({"name": item.get_name(), "title": title, "text": text})
|
||||
return chapters
|
||||
|
||||
|
||||
def chunk_text(text: str, max_chars: int = _CHUNK_CHARS) -> list[str]:
|
||||
sentences = re.split(r"(?<=[.!?])\s+", text)
|
||||
chunks = []
|
||||
current = ""
|
||||
for sentence in sentences:
|
||||
if current and len(current) + 1 + len(sentence) > max_chars:
|
||||
chunks.append(current)
|
||||
current = sentence
|
||||
else:
|
||||
current = f"{current} {sentence}".strip()
|
||||
if current:
|
||||
chunks.append(current)
|
||||
return chunks
|
||||
|
||||
|
||||
def synthesize_chapter(text: str, voice: str, out_path: Path, mp3: bool = True) -> dict:
|
||||
chunks = chunk_text(text)
|
||||
wav_paths = []
|
||||
for i, chunk in enumerate(chunks):
|
||||
result = synthesize(chunk, voice=voice, lang="en")
|
||||
if not result.get("ok"):
|
||||
return {"ok": False, "error": f"chunk {i}/{len(chunks)}: {result.get('error')}"}
|
||||
wav_paths.append(result["path"])
|
||||
print(f" chunk {i + 1}/{len(chunks)} ok ({result['size_bytes']} bytes)", file=sys.stderr)
|
||||
|
||||
if not wav_paths:
|
||||
return {"ok": False, "error": "niciun chunk de sintetizat"}
|
||||
|
||||
wav_path = out_path.with_suffix(".wav")
|
||||
if len(wav_paths) == 1:
|
||||
Path(wav_paths[0]).replace(wav_path)
|
||||
else:
|
||||
list_file = out_path.with_suffix(".concat.txt")
|
||||
list_file.write_text("\n".join(f"file '{p}'" for p in wav_paths))
|
||||
subprocess.run(
|
||||
["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", str(list_file), "-c", "copy", str(wav_path)],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
list_file.unlink()
|
||||
for p in wav_paths:
|
||||
Path(p).unlink(missing_ok=True)
|
||||
|
||||
if not mp3:
|
||||
return {"ok": True, "path": str(wav_path), "chunks": len(chunks)}
|
||||
|
||||
subprocess.run(
|
||||
["ffmpeg", "-y", "-i", str(wav_path), "-ac", "1", "-b:a", "96k", str(out_path)],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
wav_path.unlink()
|
||||
return {"ok": True, "path": str(out_path), "chunks": len(chunks)}
|
||||
|
||||
|
||||
_SKIP_TITLES = {"contents"}
|
||||
_MIN_CHAPTER_CHARS = 500 # sub asta e de regulă pagină de titlu, nu conținut de citit
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("epub_path")
|
||||
parser.add_argument("--voice", default="Marius 4")
|
||||
parser.add_argument("--out-dir", default="/home/moltbot/workspace/epub-audio")
|
||||
parser.add_argument("--chapters", type=int, default=None, help="limitează la primele N capitole")
|
||||
parser.add_argument("--start-chapter", type=int, default=0)
|
||||
parser.add_argument("--wav", action="store_true", help="păstrează WAV în loc de MP3")
|
||||
args = parser.parse_args()
|
||||
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
ext = "wav" if args.wav else "mp3"
|
||||
|
||||
chapters = extract_chapters(Path(args.epub_path))
|
||||
chapters = [
|
||||
ch for ch in chapters
|
||||
if len(ch["text"]) >= _MIN_CHAPTER_CHARS and ch["title"].strip().lower() not in _SKIP_TITLES
|
||||
]
|
||||
print(f"{len(chapters)} capitole de citit (după filtrare titlu/ToC)", file=sys.stderr)
|
||||
|
||||
end = args.start_chapter + args.chapters if args.chapters else len(chapters)
|
||||
selected = chapters[args.start_chapter:end]
|
||||
|
||||
results = []
|
||||
for n, ch in enumerate(selected, start=args.start_chapter + 1):
|
||||
out_path = out_dir / f"{n:02d}_{re.sub(r'[^a-zA-Z0-9]+', '_', ch['title'])[:40]}.{ext}"
|
||||
print(f"[{n}] {ch['title']} ({len(ch['text'])} chars) -> {out_path.name}", file=sys.stderr)
|
||||
result = synthesize_chapter(ch["text"], args.voice, out_path, mp3=not args.wav)
|
||||
results.append({"chapter": ch["title"], **result})
|
||||
print(json.dumps(results[-1]), file=sys.stderr)
|
||||
|
||||
print(json.dumps({"ok": True, "results": results}))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user