feat(dashboard): endpoint de download fișiere binare + tool epub-to-audio
- dashboard/handlers/files.py: handle_files_download() servește mp3/wav/zip din WORKSPACE_DIR cu Content-Disposition attachment, resolve dedicat (nu _resolve_sandboxed, care nu ajunge niciodată la WORKSPACE_DIR) - dashboard/api.py: rutează /api/files/download - tools/epub_to_audio.py: tool nou de conversie EPUB → audio - cron/jobs.json, memory/kb/index.json: stare auto-generată (job runs, regenerare index KB) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -11,9 +11,9 @@
|
|||||||
"report_on": "changes",
|
"report_on": "changes",
|
||||||
"timeout": 120,
|
"timeout": 120,
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"last_run": "2026-08-21T16:00:00.000757+00:00",
|
"last_run": "2026-08-27T16:00:00.002423+00:00",
|
||||||
"last_status": "ok",
|
"last_status": "ok",
|
||||||
"next_run": "2026-08-22T10:00:00+00:00"
|
"next_run": "2026-08-28T10:00:00+00:00"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "security-audit-daily",
|
"name": "security-audit-daily",
|
||||||
@@ -43,9 +43,9 @@
|
|||||||
"report_on": "never",
|
"report_on": "never",
|
||||||
"timeout": 120,
|
"timeout": 120,
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"last_run": "2026-08-22T03:30:00.002627+00:00",
|
"last_run": "2026-08-27T03:30:00.002353+00:00",
|
||||||
"last_status": "ok",
|
"last_status": "ok",
|
||||||
"next_run": "2026-08-23T03:30:00+00:00"
|
"next_run": "2026-08-28T03:30:00+00:00"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "archive-tasks-daily",
|
"name": "archive-tasks-daily",
|
||||||
@@ -59,9 +59,9 @@
|
|||||||
"report_on": "changes",
|
"report_on": "changes",
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"last_run": "2026-08-22T03:00:00.001220+00:00",
|
"last_run": "2026-08-27T03:00:00.001621+00:00",
|
||||||
"last_status": "ok",
|
"last_status": "ok",
|
||||||
"next_run": "2026-08-23T03:00:00+00:00"
|
"next_run": "2026-08-28T03:00:00+00:00"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "backup-config",
|
"name": "backup-config",
|
||||||
@@ -75,9 +75,9 @@
|
|||||||
"report_on": "never",
|
"report_on": "never",
|
||||||
"timeout": 120,
|
"timeout": 120,
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"last_run": "2026-08-22T02:00:00.001599+00:00",
|
"last_run": "2026-08-27T02:00:00.000911+00:00",
|
||||||
"last_status": "ok",
|
"last_status": "ok",
|
||||||
"next_run": "2026-08-23T02:00:00+00:00"
|
"next_run": "2026-08-28T02:00:00+00:00"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "insights-extract",
|
"name": "insights-extract",
|
||||||
@@ -243,9 +243,9 @@
|
|||||||
"prompt": "Heartbeat check. Rulează src/heartbeat.py printr-un scurt raport de status.\nDacă nu e nimic de raportat (email=0, calendar nu are evenimente <2h, kb ok), răspunde doar cu HEARTBEAT_OK și oprește-te — nu trimite mesaj.\nDacă e ceva: raport scurt pe Discord #echo-work.",
|
"prompt": "Heartbeat check. Rulează src/heartbeat.py printr-un scurt raport de status.\nDacă nu e nimic de raportat (email=0, calendar nu are evenimente <2h, kb ok), răspunde doar cu HEARTBEAT_OK și oprește-te — nu trimite mesaj.\nDacă e ceva: raport scurt pe Discord #echo-work.",
|
||||||
"allowed_tools": [],
|
"allowed_tools": [],
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"last_run": "2026-08-22T06:00:00.001316+00:00",
|
"last_run": "2026-08-27T14:00:00.002158+00:00",
|
||||||
"last_status": "ok",
|
"last_status": "ok",
|
||||||
"next_run": "2026-08-22T10:00:00+00:00"
|
"next_run": "2026-08-27T18:00:00+00:00"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "night-execute",
|
"name": "night-execute",
|
||||||
|
|||||||
@@ -173,6 +173,8 @@ class TaskBoardHandler(
|
|||||||
self.handle_cron_status()
|
self.handle_cron_status()
|
||||||
elif self.path == '/api/habits':
|
elif self.path == '/api/habits':
|
||||||
self.handle_habits_get()
|
self.handle_habits_get()
|
||||||
|
elif self.path.startswith('/api/files/download'):
|
||||||
|
self.handle_files_download()
|
||||||
elif self.path.startswith('/api/files'):
|
elif self.path.startswith('/api/files'):
|
||||||
self.handle_files_get()
|
self.handle_files_get()
|
||||||
elif self.path.startswith('/api/diff'):
|
elif self.path.startswith('/api/diff'):
|
||||||
|
|||||||
@@ -72,6 +72,42 @@ class FilesHandlers:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.send_json({'error': str(e)}, 500)
|
self.send_json({'error': str(e)}, 500)
|
||||||
|
|
||||||
|
def handle_files_download(self):
|
||||||
|
"""Stream a binary file (audio, zip, etc.) from WORKSPACE_DIR as a download.
|
||||||
|
|
||||||
|
Dedicated resolution (not `_resolve_sandboxed`, which always resolves
|
||||||
|
against ALLOWED_WORKSPACES[0]/BASE_DIR and never reaches WORKSPACE_DIR)
|
||||||
|
so this stays isolated from the shared file-browser sandbox logic.
|
||||||
|
"""
|
||||||
|
params = parse_qs(urlparse(self.path).query)
|
||||||
|
path = params.get('path', [''])[0]
|
||||||
|
|
||||||
|
try:
|
||||||
|
target = (constants.WORKSPACE_DIR / path).resolve()
|
||||||
|
target.relative_to(constants.WORKSPACE_DIR.resolve())
|
||||||
|
except (ValueError, OSError):
|
||||||
|
self.send_json({'error': 'Access denied'}, 403)
|
||||||
|
return
|
||||||
|
if not target.is_file():
|
||||||
|
self.send_json({'error': 'Not found'}, 404)
|
||||||
|
return
|
||||||
|
|
||||||
|
ext = target.suffix.lstrip('.').lower()
|
||||||
|
ctype = {
|
||||||
|
'mp3': 'audio/mpeg',
|
||||||
|
'wav': 'audio/wav',
|
||||||
|
'zip': 'application/zip',
|
||||||
|
}.get(ext, 'application/octet-stream')
|
||||||
|
|
||||||
|
data = target.read_bytes()
|
||||||
|
self.send_response(200)
|
||||||
|
self.send_header('Content-Type', ctype)
|
||||||
|
self.send_header('Content-Length', str(len(data)))
|
||||||
|
self.send_header('Content-Disposition', f'attachment; filename="{target.name}"')
|
||||||
|
self.send_header('Cache-Control', 'public, max-age=3600')
|
||||||
|
self.end_headers()
|
||||||
|
self.wfile.write(data)
|
||||||
|
|
||||||
def handle_files_post(self):
|
def handle_files_post(self):
|
||||||
"""Save file content."""
|
"""Save file content."""
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -1,5 +1,18 @@
|
|||||||
{
|
{
|
||||||
"notes": [
|
"notes": [
|
||||||
|
{
|
||||||
|
"file": "notes-data/tools/infrastructure.md",
|
||||||
|
"title": "Infrastructură (Proxmox + Docker)",
|
||||||
|
"date": "2026-08-23",
|
||||||
|
"tags": [],
|
||||||
|
"domains": [],
|
||||||
|
"types": [],
|
||||||
|
"category": "tools",
|
||||||
|
"project": null,
|
||||||
|
"subdir": null,
|
||||||
|
"video": "",
|
||||||
|
"tldr": "- Orice operație distructivă"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"file": "notes-data/projects/lfm2.5-230m-summarization-eval.md",
|
"file": "notes-data/projects/lfm2.5-230m-summarization-eval.md",
|
||||||
"title": "LFM2.5-230M — evaluare ca preprocesor de rezumat / fallback conversațional (2026-08-21)",
|
"title": "LFM2.5-230M — evaluare ca preprocesor de rezumat / fallback conversațional (2026-08-21)",
|
||||||
@@ -1444,19 +1457,6 @@
|
|||||||
"video": "",
|
"video": "",
|
||||||
"tldr": "*Instrumentale clasice și cinematic — Einaudi, Vangelis, Hisaishi, Tiersen, Secret Garden.*"
|
"tldr": "*Instrumentale clasice și cinematic — Einaudi, Vangelis, Hisaishi, Tiersen, Secret Garden.*"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"file": "notes-data/tools/infrastructure.md",
|
|
||||||
"title": "Infrastructură (Proxmox + Docker)",
|
|
||||||
"date": "2026-04-26",
|
|
||||||
"tags": [],
|
|
||||||
"domains": [],
|
|
||||||
"types": [],
|
|
||||||
"category": "tools",
|
|
||||||
"project": null,
|
|
||||||
"subdir": null,
|
|
||||||
"video": "",
|
|
||||||
"tldr": "- Orice operație distructivă"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"file": "notes-data/coaching/2026-04-25-negativity-bias-reframing.md",
|
"file": "notes-data/coaching/2026-04-25-negativity-bias-reframing.md",
|
||||||
"title": "Negativity Bias & Positive Reframing",
|
"title": "Negativity Bias & Positive Reframing",
|
||||||
|
|||||||
146
tools/epub_to_audio.py
Normal file
146
tools/epub_to_audio.py
Normal file
@@ -0,0 +1,146 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""EPUB -> audio cu una din vocile din tools/tts.py.
|
||||||
|
|
||||||
|
CLI:
|
||||||
|
python3 tools/epub_to_audio.py <epub_path> --voice "Marius 4" --out-dir DIR
|
||||||
|
[--chapters N] [--start-chapter N]
|
||||||
|
|
||||||
|
Per capitol: extrage text curat (ebooklib + BeautifulSoup), îl împarte în bucăți
|
||||||
|
sigure pentru TTS (pe propoziții, sub _CHUNK_CHARS), sintetizează secvențial cu
|
||||||
|
tools.tts.synthesize, apoi concatenează cu ffmpeg într-un singur WAV per capitol.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
from ebooklib import epub, ITEM_DOCUMENT
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
if str(REPO_ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(REPO_ROOT))
|
||||||
|
|
||||||
|
from tools.tts import synthesize # noqa: E402
|
||||||
|
|
||||||
|
# pocket-tts n-are un hard limit documentat ca Supertonic, dar bucăți mari cresc
|
||||||
|
# riscul de timeout (60s) și degradarea calității pe propoziții foarte lungi.
|
||||||
|
_CHUNK_CHARS = 600
|
||||||
|
|
||||||
|
|
||||||
|
def extract_chapters(epub_path: Path) -> list[dict]:
|
||||||
|
book = epub.read_epub(str(epub_path))
|
||||||
|
chapters = []
|
||||||
|
for item in book.get_items_of_type(ITEM_DOCUMENT):
|
||||||
|
soup = BeautifulSoup(item.get_content(), "html.parser")
|
||||||
|
for tag in soup(["script", "style"]):
|
||||||
|
tag.decompose()
|
||||||
|
text = soup.get_text(separator=" ", strip=True)
|
||||||
|
text = re.sub(r"\s+", " ", text).strip()
|
||||||
|
if len(text) < 200:
|
||||||
|
continue # sar peste pagini de copertă/titlu/ToC, prea puțin conținut
|
||||||
|
title_tag = soup.find(["h1", "h2", "title"])
|
||||||
|
title = title_tag.get_text(strip=True) if title_tag else item.get_name()
|
||||||
|
chapters.append({"name": item.get_name(), "title": title, "text": text})
|
||||||
|
return chapters
|
||||||
|
|
||||||
|
|
||||||
|
def chunk_text(text: str, max_chars: int = _CHUNK_CHARS) -> list[str]:
|
||||||
|
sentences = re.split(r"(?<=[.!?])\s+", text)
|
||||||
|
chunks = []
|
||||||
|
current = ""
|
||||||
|
for sentence in sentences:
|
||||||
|
if current and len(current) + 1 + len(sentence) > max_chars:
|
||||||
|
chunks.append(current)
|
||||||
|
current = sentence
|
||||||
|
else:
|
||||||
|
current = f"{current} {sentence}".strip()
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
|
||||||
|
def synthesize_chapter(text: str, voice: str, out_path: Path, mp3: bool = True) -> dict:
|
||||||
|
chunks = chunk_text(text)
|
||||||
|
wav_paths = []
|
||||||
|
for i, chunk in enumerate(chunks):
|
||||||
|
result = synthesize(chunk, voice=voice, lang="en")
|
||||||
|
if not result.get("ok"):
|
||||||
|
return {"ok": False, "error": f"chunk {i}/{len(chunks)}: {result.get('error')}"}
|
||||||
|
wav_paths.append(result["path"])
|
||||||
|
print(f" chunk {i + 1}/{len(chunks)} ok ({result['size_bytes']} bytes)", file=sys.stderr)
|
||||||
|
|
||||||
|
if not wav_paths:
|
||||||
|
return {"ok": False, "error": "niciun chunk de sintetizat"}
|
||||||
|
|
||||||
|
wav_path = out_path.with_suffix(".wav")
|
||||||
|
if len(wav_paths) == 1:
|
||||||
|
Path(wav_paths[0]).replace(wav_path)
|
||||||
|
else:
|
||||||
|
list_file = out_path.with_suffix(".concat.txt")
|
||||||
|
list_file.write_text("\n".join(f"file '{p}'" for p in wav_paths))
|
||||||
|
subprocess.run(
|
||||||
|
["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", str(list_file), "-c", "copy", str(wav_path)],
|
||||||
|
check=True,
|
||||||
|
capture_output=True,
|
||||||
|
)
|
||||||
|
list_file.unlink()
|
||||||
|
for p in wav_paths:
|
||||||
|
Path(p).unlink(missing_ok=True)
|
||||||
|
|
||||||
|
if not mp3:
|
||||||
|
return {"ok": True, "path": str(wav_path), "chunks": len(chunks)}
|
||||||
|
|
||||||
|
subprocess.run(
|
||||||
|
["ffmpeg", "-y", "-i", str(wav_path), "-ac", "1", "-b:a", "96k", str(out_path)],
|
||||||
|
check=True,
|
||||||
|
capture_output=True,
|
||||||
|
)
|
||||||
|
wav_path.unlink()
|
||||||
|
return {"ok": True, "path": str(out_path), "chunks": len(chunks)}
|
||||||
|
|
||||||
|
|
||||||
|
_SKIP_TITLES = {"contents"}
|
||||||
|
_MIN_CHAPTER_CHARS = 500 # sub asta e de regulă pagină de titlu, nu conținut de citit
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("epub_path")
|
||||||
|
parser.add_argument("--voice", default="Marius 4")
|
||||||
|
parser.add_argument("--out-dir", default="/home/moltbot/workspace/epub-audio")
|
||||||
|
parser.add_argument("--chapters", type=int, default=None, help="limitează la primele N capitole")
|
||||||
|
parser.add_argument("--start-chapter", type=int, default=0)
|
||||||
|
parser.add_argument("--wav", action="store_true", help="păstrează WAV în loc de MP3")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
out_dir = Path(args.out_dir)
|
||||||
|
out_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
ext = "wav" if args.wav else "mp3"
|
||||||
|
|
||||||
|
chapters = extract_chapters(Path(args.epub_path))
|
||||||
|
chapters = [
|
||||||
|
ch for ch in chapters
|
||||||
|
if len(ch["text"]) >= _MIN_CHAPTER_CHARS and ch["title"].strip().lower() not in _SKIP_TITLES
|
||||||
|
]
|
||||||
|
print(f"{len(chapters)} capitole de citit (după filtrare titlu/ToC)", file=sys.stderr)
|
||||||
|
|
||||||
|
end = args.start_chapter + args.chapters if args.chapters else len(chapters)
|
||||||
|
selected = chapters[args.start_chapter:end]
|
||||||
|
|
||||||
|
results = []
|
||||||
|
for n, ch in enumerate(selected, start=args.start_chapter + 1):
|
||||||
|
out_path = out_dir / f"{n:02d}_{re.sub(r'[^a-zA-Z0-9]+', '_', ch['title'])[:40]}.{ext}"
|
||||||
|
print(f"[{n}] {ch['title']} ({len(ch['text'])} chars) -> {out_path.name}", file=sys.stderr)
|
||||||
|
result = synthesize_chapter(ch["text"], args.voice, out_path, mp3=not args.wav)
|
||||||
|
results.append({"chapter": ch["title"], **result})
|
||||||
|
print(json.dumps(results[-1]), file=sys.stderr)
|
||||||
|
|
||||||
|
print(json.dumps({"ok": True, "results": results}))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user