Move the /tmp prototype (Baileys bridge + RAG consumer) into git as a proper sibling project to discord-bridge/: own systemd --user units (whatsapp bridge, rag consumer, dashboard, periodic Drive sync timer), a filesystem document store with a stdlib control dashboard (start/ stop/restart, document CRUD, reindex, Google Drive sync via rclone), and an idempotent ops/install.sh following the same conventions. Co-Authored-By: Claude Agent <noreply@anthropic.com>
64 lines
1.8 KiB
Python
64 lines
1.8 KiB
Python
#!/usr/bin/env python3
|
|
"""(Re)construieste rag_index.json din toate documentele din depozit (store.py),
|
|
cu embeddings Ollama. Rulat manual sau declansat din dashboard (`/api/reindex`)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import sys
|
|
|
|
import requests
|
|
|
|
import config
|
|
import store
|
|
|
|
|
|
def chunk_text(text: str) -> list[str]:
|
|
# imparte pe linii goale in paragrafe, uneste bucatile mici cu urmatoarea
|
|
raw_parts = re.split(r"\n\s*\n", text.strip())
|
|
chunks: list[str] = []
|
|
buffer = ""
|
|
for part in raw_parts:
|
|
part = part.strip()
|
|
if not part:
|
|
continue
|
|
buffer = f"{buffer}\n\n{part}" if buffer else part
|
|
if len(buffer) >= 200:
|
|
chunks.append(buffer)
|
|
buffer = ""
|
|
if buffer:
|
|
chunks.append(buffer)
|
|
return chunks
|
|
|
|
|
|
def embed(text: str) -> list[float]:
|
|
resp = requests.post(
|
|
f"{config.get('OLLAMA_URL')}/api/embeddings",
|
|
json={"model": config.get("EMBED_MODEL"), "prompt": text},
|
|
timeout=60,
|
|
)
|
|
resp.raise_for_status()
|
|
return resp.json()["embedding"]
|
|
|
|
|
|
def build() -> dict:
|
|
entries = []
|
|
docs = store.list_documents()
|
|
for doc in docs:
|
|
text = store.read_document(doc["name"])
|
|
for i, chunk in enumerate(chunk_text(text)):
|
|
vec = embed(chunk)
|
|
entries.append({"source": doc["name"], "chunk": i, "text": chunk, "embedding": vec})
|
|
config.STATE_DIR.mkdir(parents=True, exist_ok=True)
|
|
config.INDEX_FILE.write_text(json.dumps(entries, ensure_ascii=False), encoding="utf-8")
|
|
return {"documents": len(docs), "chunks": len(entries)}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
result = build()
|
|
print(
|
|
f"[indexer] {result['documents']} documente, {result['chunks']} chunk-uri -> {config.INDEX_FILE}",
|
|
file=sys.stderr,
|
|
)
|