Dictionarul de erori Oracle pus in `documents/` ar fi disparut in 10 minute: `rclone sync` sterge de acolo tot ce nu exista in Drive. Aceeasi problema o avea si butonul de adaugare document din dashboard — documentul traia pana la urmatorul tur de sincronizare, ceea ce README-ul mentiona ca pe o ciudatenie, nu ca pe un bug. Acum sunt doua directoare cu un singur spatiu de nume: `documents/` (oglinda Drive) si `documents-local/` (ce nu vine din Drive). La acelasi nume castiga Drive-ul — e sursa comuna a echipei, iar copia locala poate fi o versiune veche uitata acolo. Dashboard-ul scrie in cel local si marcheaza documentele „local". Recalibrat cu dictionarul indexat (169 chunk-uri): 19/19, cele patru intrebari Oracle noi ies la 0,73-0,75. Reindexarea a durat cat cele 29 de chunk-uri noi, nu cat toate 169 — refolosirea vectorilor isi face treaba (log: "29 embeddings noi, 140 refolosite"). Verificat pe canalul viu: o escaladare cu captura chiar pleaca pe WhatsApp, imagine + rezumat + referinta, si apare in dashboard. 73 pass. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Q4uzvgm7AyJch5WH8QHRhY
294 lines
9.1 KiB
Python
294 lines
9.1 KiB
Python
"""Ce intra in index (preferinta de format) si cum se taie XML-ul in chunk-uri."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
import indexer
|
|
import store
|
|
|
|
XML = """<?xml version="1.0" encoding="UTF-8"?>
|
|
<erori_rag versiune="1.0">
|
|
<!-- comentariu -->
|
|
<eroare_stocuri_negative>
|
|
<keywords>stocuri negative NIR</keywords>
|
|
<mesaj_eroare>Documentul nu poate fi sters: stocuri negative pentru BANDA HARTIE.</mesaj_eroare>
|
|
<rezolvare>Optiuni firma > VERIFICARESTOCNEGATIV = 0.</rezolvare>
|
|
</eroare_stocuri_negative>
|
|
<eroare_a_doua>
|
|
<mesaj_eroare>Alta eroare.</mesaj_eroare>
|
|
<rezolvare>Alta rezolvare.</rezolvare>
|
|
</eroare_a_doua>
|
|
</erori_rag>"""
|
|
|
|
|
|
# ------------------------------------------------------- preferinta de format
|
|
def test_xml_umbreste_md_cu_acelasi_nume(write):
|
|
write("d406.xml", "<r><a>x</a></r>")
|
|
write("d406.md", "versiune veche")
|
|
umbrite = {d["name"]: d["shadowed_by"] for d in store.list_documents()}
|
|
assert umbrite["d406.md"] == "d406.xml"
|
|
assert umbrite["d406.xml"] is None
|
|
assert [d["name"] for d in store.documents_for_index()] == ["d406.xml"]
|
|
|
|
|
|
def test_md_umbreste_txt(write):
|
|
write("x.md", "md")
|
|
write("x.txt", "txt")
|
|
assert [d["name"] for d in store.documents_for_index()] == ["x.md"]
|
|
|
|
|
|
def test_fisierele_fara_pereche_raman_toate(write):
|
|
write("bilant.md", "a")
|
|
write("roagest.xml", "<r><a>b</a></r>")
|
|
write("note.txt", "c")
|
|
assert len(store.documents_for_index()) == 3
|
|
|
|
|
|
def test_depozitul_e_vizibil_intreg_chiar_daca_nu_tot_se_indexeaza(write):
|
|
write("d406.xml", "<r><a>x</a></r>")
|
|
write("d406.md", "vechi")
|
|
assert len(store.list_documents()) == 2 # dashboard le arata pe ambele
|
|
assert len(store.documents_for_index()) == 1 # indexul ia doar una
|
|
|
|
|
|
def test_extensiile_necunoscute_sunt_ignorate(write):
|
|
write("chatflow.json", "{}")
|
|
write("script.ps1", "echo")
|
|
assert store.list_documents() == []
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["a.xml", "a.md", "a.txt"])
|
|
def test_numele_acceptate(name):
|
|
assert store.validate_name(name) == name
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["a.json", "a.pdf", "../a.md", "a b.md"])
|
|
def test_numele_respinse(name):
|
|
with pytest.raises(ValueError):
|
|
store.validate_name(name)
|
|
|
|
|
|
# ----------------------------------------------------------- chunking XML
|
|
def test_o_problema_ramane_un_singur_chunk():
|
|
"""Miezul: mesajul de eroare si rezolvarea NU trebuie separate."""
|
|
chunks = indexer.chunk_document("x.xml", XML)
|
|
assert len(chunks) == 2
|
|
assert "BANDA HARTIE" in chunks[0] and "VERIFICARESTOCNEGATIV" in chunks[0]
|
|
assert "Alta eroare" in chunks[1] and "Alta rezolvare" in chunks[1]
|
|
|
|
|
|
def test_etichetele_raman_ca_etichete_fara_paranteze():
|
|
chunk = indexer.chunk_document("x.xml", XML)[0]
|
|
assert "mesaj eroare:" in chunk and "rezolvare:" in chunk
|
|
assert "<mesaj_eroare>" not in chunk
|
|
|
|
|
|
def test_entitatile_sunt_decodate():
|
|
assert ">" not in indexer.chunk_document("x.xml", XML)[0]
|
|
|
|
|
|
def test_comentariile_nu_devin_chunk():
|
|
assert not any("comentariu" in c for c in indexer.chunk_document("x.xml", XML))
|
|
|
|
|
|
def test_xml_invalid_cade_pe_taierea_obisnuita():
|
|
chunks = indexer.chunk_document("stricat.xml", "<root><neinchis>")
|
|
assert chunks == ["<root><neinchis>"]
|
|
|
|
|
|
def test_xml_fara_copii_e_tratat_ca_text():
|
|
assert indexer.chunk_document("x.xml", "<root>doar text</root>") == ["doar text"]
|
|
|
|
|
|
def test_element_urias_e_taiat_mai_departe():
|
|
mare = "<r><a>" + ("propozitie lunga. " * 500) + "</a></r>"
|
|
chunks = indexer.chunk_document("x.xml", mare)
|
|
assert len(chunks) > 1
|
|
assert all(len(c) < indexer.MAX_XML_CHUNK * 2 for c in chunks)
|
|
|
|
|
|
def test_md_foloseste_taierea_pe_paragrafe():
|
|
text = "primul paragraf\n\n" + "al doilea paragraf " * 20
|
|
assert len(indexer.chunk_document("x.md", text)) >= 1
|
|
assert indexer.chunk_document("x.md", text) == indexer.chunk_text(text)
|
|
|
|
|
|
# --------------------------------------------- concurenta intre reindexari
|
|
def test_lacatul_refuza_a_doua_reindexare():
|
|
"""Timer-ul nu trebuie sa porneasca peste o sincronizare manuala."""
|
|
import config
|
|
|
|
with config.exclusive():
|
|
with pytest.raises(config.Busy):
|
|
with config.exclusive():
|
|
pass
|
|
|
|
|
|
def test_lacatul_se_elibereaza_dupa_iesire():
|
|
import config
|
|
|
|
with config.exclusive():
|
|
pass
|
|
with config.exclusive(): # trebuie sa mearga din nou
|
|
pass
|
|
|
|
|
|
def test_lacatul_se_elibereaza_si_la_exceptie():
|
|
import config
|
|
|
|
with pytest.raises(ValueError):
|
|
with config.exclusive():
|
|
raise ValueError("ceva")
|
|
with config.exclusive():
|
|
pass
|
|
|
|
|
|
def test_indexul_se_scrie_atomic(write, monkeypatch):
|
|
"""Consumer-ul reciteste indexul la 30s; nu are voie sa prinda JSON pe jumatate."""
|
|
import config
|
|
import indexer
|
|
|
|
write("x.md", "un paragraf oarecare")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.1, 0.2])
|
|
vazute = []
|
|
real_replace = indexer.os.replace
|
|
|
|
def spion(src, dst):
|
|
vazute.append((str(src), str(dst)))
|
|
return real_replace(src, dst)
|
|
|
|
monkeypatch.setattr(indexer.os, "replace", spion)
|
|
indexer.build()
|
|
assert vazute and vazute[0][1] == str(config.INDEX_FILE)
|
|
assert not config.INDEX_FILE.with_suffix(".json.tmp").exists()
|
|
|
|
|
|
def test_xml_invalid_e_raportat_ca_avertisment(write, monkeypatch):
|
|
import indexer
|
|
|
|
write("stricat.xml", "<root><a>x</a></root><in-plus/>")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.0])
|
|
out = indexer.build()
|
|
assert any("stricat.xml" in w for w in out.get("warnings", []))
|
|
|
|
|
|
def test_xml_valid_nu_produce_avertismente(write, monkeypatch):
|
|
import indexer
|
|
|
|
write("bun.xml", "<root><a>x</a></root>")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.0])
|
|
assert "warnings" not in indexer.build()
|
|
|
|
|
|
# --- refolosirea embeddings-urilor la reindexare -----------------------------
|
|
|
|
def test_vectorii_din_indexul_vechi_se_refolosesc(monkeypatch, write, tmp_path):
|
|
"""Un document nou nu are voie sa reforteze embeddings-urile celorlalte.
|
|
|
|
Pe CPU un embedding costa ~7 secunde; fara refolosire, adaugarea unui
|
|
document la un index de 170 de chunk-uri inseamna 20 de minute de asteptare.
|
|
"""
|
|
import json
|
|
import config
|
|
import indexer
|
|
|
|
write("primul.md", "text vechi, deja indexat")
|
|
calculate = []
|
|
|
|
def fals_embed(chunk):
|
|
calculate.append(chunk)
|
|
return [0.5, 0.5]
|
|
|
|
monkeypatch.setattr(indexer, "embed", fals_embed)
|
|
|
|
indexer.build()
|
|
assert calculate == ["text vechi, deja indexat"]
|
|
|
|
calculate.clear()
|
|
write("al-doilea.md", "text nou")
|
|
rezultat = indexer.build()
|
|
|
|
# doar chunk-ul nou s-a calculat
|
|
assert calculate == ["text nou"]
|
|
assert rezultat["embeddings_noi"] == 1
|
|
assert rezultat["embeddings_refolosite"] == 1
|
|
|
|
index = json.loads(config.INDEX_FILE.read_text(encoding="utf-8"))
|
|
assert {e["source"] for e in index} == {"primul.md", "al-doilea.md"}
|
|
|
|
|
|
def test_textul_modificat_se_reembedeaza(monkeypatch, write):
|
|
import indexer
|
|
|
|
write("doc.md", "prima varianta")
|
|
monkeypatch.setattr(indexer, "embed", lambda chunk: [1.0])
|
|
indexer.build()
|
|
|
|
calculate = []
|
|
|
|
def fals_embed(chunk):
|
|
calculate.append(chunk)
|
|
return [2.0]
|
|
|
|
monkeypatch.setattr(indexer, "embed", fals_embed)
|
|
write("doc.md", "varianta modificata")
|
|
rezultat = indexer.build()
|
|
|
|
assert calculate == ["varianta modificata"]
|
|
assert rezultat["embeddings_refolosite"] == 0
|
|
|
|
|
|
# --- documente locale, in afara oglinzii Drive -------------------------------
|
|
|
|
def _scrie_local(name: str, text: str):
|
|
import config
|
|
config.DOCS_LOCAL_DIR.mkdir(parents=True, exist_ok=True)
|
|
p = config.DOCS_LOCAL_DIR / name
|
|
p.write_text(text, encoding="utf-8")
|
|
return p
|
|
|
|
|
|
def test_documentele_locale_intra_in_index(write):
|
|
import store
|
|
write("din-drive.md", "vine din Drive")
|
|
_scrie_local("al-meu.md", "adaugat local")
|
|
|
|
nume = {d["name"] for d in store.documents_for_index()}
|
|
assert nume == {"din-drive.md", "al-meu.md"}
|
|
|
|
|
|
def test_se_vede_care_e_local(write):
|
|
import store
|
|
write("din-drive.md", "x")
|
|
_scrie_local("al-meu.md", "y")
|
|
dupa_nume = {d["name"]: d for d in store.list_documents()}
|
|
assert dupa_nume["al-meu.md"]["local"] is True
|
|
assert dupa_nume["din-drive.md"]["local"] is False
|
|
|
|
|
|
def test_la_acelasi_nume_castiga_versiunea_din_drive(write):
|
|
import store
|
|
_scrie_local("acelasi.md", "copie locala veche")
|
|
write("acelasi.md", "versiunea din Drive")
|
|
assert store.read_document("acelasi.md") == "versiunea din Drive"
|
|
assert len(store.list_documents()) == 1
|
|
|
|
|
|
def test_scrierea_din_dashboard_merge_in_directorul_local():
|
|
# Pana acum ajungea in oglinda Drive si disparea la urmatorul rclone sync.
|
|
import config
|
|
import store
|
|
store.write_document("nota.md", "ceva de tinut minte")
|
|
assert (config.DOCS_LOCAL_DIR / "nota.md").exists()
|
|
assert not (config.DOCS_DIR / "nota.md").exists()
|
|
|
|
|
|
def test_stergerea_gaseste_documentul_in_ambele_directoare(write):
|
|
import store
|
|
_scrie_local("local.md", "x")
|
|
write("drive.md", "y")
|
|
assert store.delete_document("local.md") is True
|
|
assert store.delete_document("drive.md") is True
|
|
assert store.delete_document("inexistent.md") is False
|