Dosarul document_store din Drive are 3 surse .xml pe care depozitul le ignora
complet, fiindca store.py accepta doar .txt/.md. La d406_saft_knowledge exista
ambele formate, iar .xml e cu trei luni mai nou (2026-01-28 vs 2025-10-15) si cu
50% mai mare (64 KB vs 41 KB) — deci indexam varianta mai saraca.
- store.py devine sursa unica pentru extensii (DOC_EXTENSIONS = .txt/.md/.xml).
Cand acelasi nume de baza exista in mai multe formate, la indexare intra unul
singur, cel mai bogat (.xml > .md > .txt); celalalt ramane pe disc, marcat
`shadowed_by`. Fara asta, acelasi raspuns ar aparea de doua ori in rezultate.
`list_documents()` arata tot (dashboard), `documents_for_index()` doar
castigatorii (indexer).
- indexer.py taie XML-ul altfel: un chunk per element de nivel 1, adica o
problema = un chunk, cu <mesaj_eroare> si <rezolvare> impreuna. Taierea pe
linii goale le-ar separa si cautarea ar returna eroarea fara raspuns.
Etichetele raman prefixe lizibile ("mesaj eroare: ..."), fara paranteze
unghiulare care doar dilueaza embedding-ul. XML invalid nu opreste indexarea:
cade pe taierea obisnuita, cu o linie in log. Elementele peste 4000 de
caractere se taie mai departe pe granite de cuvant — `chunk_text` imparte doar
pe linii goale, deci un element scris ca un paragraf lung ar fi ramas intreg
(prins de test).
- sync.py: amprenta si `rclone --include` derivate din DOC_EXTENSIONS.
- dashboard: acelasi filtru si aceeasi preferinta (copie, fiindca nu poate
importa `store` — coliziune de nume pe `config`), plus marcajul "umbrit de X"
in tabelul de documente si numarul de documente chiar indexate.
- README: sectiunea Drive rescrisa pe `rclone authorize` (autorizezi pe o masina
cu browser, muti tokenul) in loc de cont de serviciu — mai putini pasi, fara
consola Google Cloud. Documentat si ca `sync` sterge local ce nu mai e in Drive.
tests/ nou (20 de teste, fara retea si fara Ollama): preferinta de format,
vizibilitatea in dashboard, taierea XML, entitati, comentarii, XML invalid,
elemente uriase. Suita puntii Discord: 426 pass, neafectata.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Q4uzvgm7AyJch5WH8QHRhY
105 lines
3.5 KiB
Python
105 lines
3.5 KiB
Python
"""Configuratie comuna a puntii WhatsApp+RAG pentru Maria (LXC 171 claude-agent).
|
|
|
|
Citeste ~/.maria-bridge/env (KEY=value, tolerant la comentarii si ghilimele).
|
|
Mirrors deliberat conventiile din discord-bridge/config.py, ca sa fie un singur
|
|
model de citit pentru cine intretine ambele punti pe acest container.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import pathlib
|
|
|
|
# Directorul de baza poate fi mutat in teste prin MARIA_BRIDGE_DIR.
|
|
_DEFAULT_DIR = pathlib.Path.home() / ".maria-bridge"
|
|
|
|
STATE_DIR: pathlib.Path = pathlib.Path(os.environ.get("MARIA_BRIDGE_DIR") or _DEFAULT_DIR)
|
|
DOCS_DIR: pathlib.Path = STATE_DIR / "documents"
|
|
INDEX_FILE: pathlib.Path = STATE_DIR / "rag_index.json"
|
|
LOG_DIR: pathlib.Path = STATE_DIR / "logs"
|
|
ENV_FILE: pathlib.Path = STATE_DIR / "env"
|
|
AUTH_DIR: pathlib.Path = STATE_DIR / "whatsapp-auth"
|
|
|
|
_env: dict[str, str] = {}
|
|
|
|
# Valori implicite. LLM_URL/OLLAMA_URL presupun tunel/proxy local catre backend-ul
|
|
# real (vezi docs/maria-whatsapp-rag-prototype.md) — de completat in env dupa caz.
|
|
DEFAULTS: dict[str, str] = {
|
|
"BRIDGE_HOST": "127.0.0.1",
|
|
"BRIDGE_PORT": "8099",
|
|
"LLM_URL": "http://127.0.0.1:8091",
|
|
"OLLAMA_URL": "http://127.0.0.1:11434",
|
|
"EMBED_MODEL": "nomic-embed-text",
|
|
"TOP_K": "3",
|
|
"MAX_TOKENS": "250",
|
|
"POLL_INTERVAL_S": "2",
|
|
"TEST_MODE_SELF_CHAT_ONLY": "true",
|
|
# Tinta rclone pentru sincronizarea depozitului de documente, ex:
|
|
# "gdrive,root_folder_id=1C4e75zgH1_7ZK-_oBP5ZZBvUPh3iEo1O:" (dosarul
|
|
# document_store din Drive, vazut pe Windows ca D:\GoogleDrive\romfast\document_store).
|
|
# Gol = sincronizare dezactivata, doar upload manual din dashboard.
|
|
"DRIVE_REMOTE": "",
|
|
}
|
|
|
|
|
|
def parse_env(text: str) -> dict[str, str]:
|
|
"""Parseaza un fisier de tip KEY=value. Nu arunca niciodata."""
|
|
out: dict[str, str] = {}
|
|
for raw in text.splitlines():
|
|
line = raw.strip()
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
if line.startswith("export "):
|
|
line = line[len("export "):].strip()
|
|
if "=" not in line:
|
|
continue
|
|
key, _, val = line.partition("=")
|
|
key = key.strip()
|
|
if not key:
|
|
continue
|
|
val = val.strip()
|
|
if val[:1] not in ("'", '"'):
|
|
cut = val.find(" #")
|
|
if cut >= 0:
|
|
val = val[:cut].rstrip()
|
|
if len(val) >= 2 and val[0] == val[-1] and val[0] in ("'", '"'):
|
|
val = val[1:-1]
|
|
out[key] = val
|
|
return out
|
|
|
|
|
|
def reload(base_dir: str | os.PathLike | None = None) -> dict[str, str]:
|
|
"""Recalculeaza caile si reciteste env-ul. Returneaza dictionarul incarcat."""
|
|
global STATE_DIR, DOCS_DIR, INDEX_FILE, LOG_DIR, ENV_FILE, AUTH_DIR, _env
|
|
if base_dir is None:
|
|
base_dir = os.environ.get("MARIA_BRIDGE_DIR") or _DEFAULT_DIR
|
|
STATE_DIR = pathlib.Path(base_dir)
|
|
DOCS_DIR = STATE_DIR / "documents"
|
|
INDEX_FILE = STATE_DIR / "rag_index.json"
|
|
LOG_DIR = STATE_DIR / "logs"
|
|
ENV_FILE = STATE_DIR / "env"
|
|
AUTH_DIR = STATE_DIR / "whatsapp-auth"
|
|
try:
|
|
_env = parse_env(ENV_FILE.read_text(encoding="utf-8"))
|
|
except OSError:
|
|
_env = {}
|
|
return _env
|
|
|
|
|
|
def get(key: str, default: str | None = None) -> str | None:
|
|
if key in _env:
|
|
return _env[key]
|
|
if key in DEFAULTS:
|
|
return DEFAULTS[key]
|
|
return default
|
|
|
|
|
|
def get_int(key: str, default: int) -> int:
|
|
try:
|
|
return int(get(key, str(default)) or default)
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
|
|
reload()
|