Add Maria WhatsApp+RAG bridge as a service (LXC 171)
Move the /tmp prototype (Baileys bridge + RAG consumer) into git as a proper sibling project to discord-bridge/: own systemd --user units (whatsapp bridge, rag consumer, dashboard, periodic Drive sync timer), a filesystem document store with a stdlib control dashboard (start/ stop/restart, document CRUD, reindex, Google Drive sync via rclone), and an idempotent ops/install.sh following the same conventions. Co-Authored-By: Claude Agent <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env python3
|
||||
"""(Re)construieste rag_index.json din toate documentele din depozit (store.py),
|
||||
cu embeddings Ollama. Rulat manual sau declansat din dashboard (`/api/reindex`)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
|
||||
import requests
|
||||
|
||||
import config
|
||||
import store
|
||||
|
||||
|
||||
def chunk_text(text: str) -> list[str]:
|
||||
# imparte pe linii goale in paragrafe, uneste bucatile mici cu urmatoarea
|
||||
raw_parts = re.split(r"\n\s*\n", text.strip())
|
||||
chunks: list[str] = []
|
||||
buffer = ""
|
||||
for part in raw_parts:
|
||||
part = part.strip()
|
||||
if not part:
|
||||
continue
|
||||
buffer = f"{buffer}\n\n{part}" if buffer else part
|
||||
if len(buffer) >= 200:
|
||||
chunks.append(buffer)
|
||||
buffer = ""
|
||||
if buffer:
|
||||
chunks.append(buffer)
|
||||
return chunks
|
||||
|
||||
|
||||
def embed(text: str) -> list[float]:
|
||||
resp = requests.post(
|
||||
f"{config.get('OLLAMA_URL')}/api/embeddings",
|
||||
json={"model": config.get("EMBED_MODEL"), "prompt": text},
|
||||
timeout=60,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp.json()["embedding"]
|
||||
|
||||
|
||||
def build() -> dict:
|
||||
entries = []
|
||||
docs = store.list_documents()
|
||||
for doc in docs:
|
||||
text = store.read_document(doc["name"])
|
||||
for i, chunk in enumerate(chunk_text(text)):
|
||||
vec = embed(chunk)
|
||||
entries.append({"source": doc["name"], "chunk": i, "text": chunk, "embedding": vec})
|
||||
config.STATE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
config.INDEX_FILE.write_text(json.dumps(entries, ensure_ascii=False), encoding="utf-8")
|
||||
return {"documents": len(docs), "chunks": len(entries)}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
result = build()
|
||||
print(
|
||||
f"[indexer] {result['documents']} documente, {result['chunks']} chunk-uri -> {config.INDEX_FILE}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
Reference in New Issue
Block a user