Files
ROMFASTSQL/proxmox/lxc171-claude-agent/maria-whatsapp-bridge/rag/indexer.py
Claude Agent dd5553e327 Add Maria WhatsApp+RAG bridge as a service (LXC 171)
Move the /tmp prototype (Baileys bridge + RAG consumer) into git as a
proper sibling project to discord-bridge/: own systemd --user units
(whatsapp bridge, rag consumer, dashboard, periodic Drive sync timer),
a filesystem document store with a stdlib control dashboard (start/
stop/restart, document CRUD, reindex, Google Drive sync via rclone),
and an idempotent ops/install.sh following the same conventions.

Co-Authored-By: Claude Agent <noreply@anthropic.com>
2026-08-31 17:38:51 +00:00

64 lines
1.8 KiB
Python

#!/usr/bin/env python3
"""(Re)construieste rag_index.json din toate documentele din depozit (store.py),
cu embeddings Ollama. Rulat manual sau declansat din dashboard (`/api/reindex`)."""
from __future__ import annotations
import json
import re
import sys
import requests
import config
import store
def chunk_text(text: str) -> list[str]:
# imparte pe linii goale in paragrafe, uneste bucatile mici cu urmatoarea
raw_parts = re.split(r"\n\s*\n", text.strip())
chunks: list[str] = []
buffer = ""
for part in raw_parts:
part = part.strip()
if not part:
continue
buffer = f"{buffer}\n\n{part}" if buffer else part
if len(buffer) >= 200:
chunks.append(buffer)
buffer = ""
if buffer:
chunks.append(buffer)
return chunks
def embed(text: str) -> list[float]:
resp = requests.post(
f"{config.get('OLLAMA_URL')}/api/embeddings",
json={"model": config.get("EMBED_MODEL"), "prompt": text},
timeout=60,
)
resp.raise_for_status()
return resp.json()["embedding"]
def build() -> dict:
entries = []
docs = store.list_documents()
for doc in docs:
text = store.read_document(doc["name"])
for i, chunk in enumerate(chunk_text(text)):
vec = embed(chunk)
entries.append({"source": doc["name"], "chunk": i, "text": chunk, "embedding": vec})
config.STATE_DIR.mkdir(parents=True, exist_ok=True)
config.INDEX_FILE.write_text(json.dumps(entries, ensure_ascii=False), encoding="utf-8")
return {"documents": len(docs), "chunks": len(entries)}
if __name__ == "__main__":
result = build()
print(
f"[indexer] {result['documents']} documente, {result['chunks']} chunk-uri -> {config.INDEX_FILE}",
file=sys.stderr,
)