Cheia de refolosire a embeddings-urilor era doar textul chunk-ului. La schimbarea
lui EMBED_MODEL indexul ar fi ramas un amestec de vectori din doua modele, iar
cautarea ar fi dat rezultate aiurea fara nici un mesaj de eroare. Acum fiecare
intrare poarta modelul cu care a fost calculata si se refolosesc doar cele cu
modelul curent; intrarile vechi, fara camp, se recalculeaza o singura data.
Indexul de pe container a fost stampilat manual cu `nomic-embed-text` (singurul
folosit pana acum), deci nu s-au recalculat cele 169 de chunk-uri.
`GET /groups` listeaza grupurile contului cu JID, nume si numar de participanti.
JID-ul unui grup ("120363...@g.us") nu se vede nicaieri in WhatsApp, iar el e
singurul mod de a scrie SUPPORT_JID pentru un grup de suport.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Q4uzvgm7AyJch5WH8QHRhY
335 lines
11 KiB
Python
335 lines
11 KiB
Python
"""Ce intra in index (preferinta de format) si cum se taie XML-ul in chunk-uri."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
import indexer
|
|
import store
|
|
|
|
XML = """<?xml version="1.0" encoding="UTF-8"?>
|
|
<erori_rag versiune="1.0">
|
|
<!-- comentariu -->
|
|
<eroare_stocuri_negative>
|
|
<keywords>stocuri negative NIR</keywords>
|
|
<mesaj_eroare>Documentul nu poate fi sters: stocuri negative pentru BANDA HARTIE.</mesaj_eroare>
|
|
<rezolvare>Optiuni firma > VERIFICARESTOCNEGATIV = 0.</rezolvare>
|
|
</eroare_stocuri_negative>
|
|
<eroare_a_doua>
|
|
<mesaj_eroare>Alta eroare.</mesaj_eroare>
|
|
<rezolvare>Alta rezolvare.</rezolvare>
|
|
</eroare_a_doua>
|
|
</erori_rag>"""
|
|
|
|
|
|
# ------------------------------------------------------- preferinta de format
|
|
def test_xml_umbreste_md_cu_acelasi_nume(write):
|
|
write("d406.xml", "<r><a>x</a></r>")
|
|
write("d406.md", "versiune veche")
|
|
umbrite = {d["name"]: d["shadowed_by"] for d in store.list_documents()}
|
|
assert umbrite["d406.md"] == "d406.xml"
|
|
assert umbrite["d406.xml"] is None
|
|
assert [d["name"] for d in store.documents_for_index()] == ["d406.xml"]
|
|
|
|
|
|
def test_md_umbreste_txt(write):
|
|
write("x.md", "md")
|
|
write("x.txt", "txt")
|
|
assert [d["name"] for d in store.documents_for_index()] == ["x.md"]
|
|
|
|
|
|
def test_fisierele_fara_pereche_raman_toate(write):
|
|
write("bilant.md", "a")
|
|
write("roagest.xml", "<r><a>b</a></r>")
|
|
write("note.txt", "c")
|
|
assert len(store.documents_for_index()) == 3
|
|
|
|
|
|
def test_depozitul_e_vizibil_intreg_chiar_daca_nu_tot_se_indexeaza(write):
|
|
write("d406.xml", "<r><a>x</a></r>")
|
|
write("d406.md", "vechi")
|
|
assert len(store.list_documents()) == 2 # dashboard le arata pe ambele
|
|
assert len(store.documents_for_index()) == 1 # indexul ia doar una
|
|
|
|
|
|
def test_extensiile_necunoscute_sunt_ignorate(write):
|
|
write("chatflow.json", "{}")
|
|
write("script.ps1", "echo")
|
|
assert store.list_documents() == []
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["a.xml", "a.md", "a.txt"])
|
|
def test_numele_acceptate(name):
|
|
assert store.validate_name(name) == name
|
|
|
|
|
|
@pytest.mark.parametrize("name", ["a.json", "a.pdf", "../a.md", "a b.md"])
|
|
def test_numele_respinse(name):
|
|
with pytest.raises(ValueError):
|
|
store.validate_name(name)
|
|
|
|
|
|
# ----------------------------------------------------------- chunking XML
|
|
def test_o_problema_ramane_un_singur_chunk():
|
|
"""Miezul: mesajul de eroare si rezolvarea NU trebuie separate."""
|
|
chunks = indexer.chunk_document("x.xml", XML)
|
|
assert len(chunks) == 2
|
|
assert "BANDA HARTIE" in chunks[0] and "VERIFICARESTOCNEGATIV" in chunks[0]
|
|
assert "Alta eroare" in chunks[1] and "Alta rezolvare" in chunks[1]
|
|
|
|
|
|
def test_etichetele_raman_ca_etichete_fara_paranteze():
|
|
chunk = indexer.chunk_document("x.xml", XML)[0]
|
|
assert "mesaj eroare:" in chunk and "rezolvare:" in chunk
|
|
assert "<mesaj_eroare>" not in chunk
|
|
|
|
|
|
def test_entitatile_sunt_decodate():
|
|
assert ">" not in indexer.chunk_document("x.xml", XML)[0]
|
|
|
|
|
|
def test_comentariile_nu_devin_chunk():
|
|
assert not any("comentariu" in c for c in indexer.chunk_document("x.xml", XML))
|
|
|
|
|
|
def test_xml_invalid_cade_pe_taierea_obisnuita():
|
|
chunks = indexer.chunk_document("stricat.xml", "<root><neinchis>")
|
|
assert chunks == ["<root><neinchis>"]
|
|
|
|
|
|
def test_xml_fara_copii_e_tratat_ca_text():
|
|
assert indexer.chunk_document("x.xml", "<root>doar text</root>") == ["doar text"]
|
|
|
|
|
|
def test_element_urias_e_taiat_mai_departe():
|
|
mare = "<r><a>" + ("propozitie lunga. " * 500) + "</a></r>"
|
|
chunks = indexer.chunk_document("x.xml", mare)
|
|
assert len(chunks) > 1
|
|
assert all(len(c) < indexer.MAX_XML_CHUNK * 2 for c in chunks)
|
|
|
|
|
|
def test_md_foloseste_taierea_pe_paragrafe():
|
|
text = "primul paragraf\n\n" + "al doilea paragraf " * 20
|
|
assert len(indexer.chunk_document("x.md", text)) >= 1
|
|
assert indexer.chunk_document("x.md", text) == indexer.chunk_text(text)
|
|
|
|
|
|
# --------------------------------------------- concurenta intre reindexari
|
|
def test_lacatul_refuza_a_doua_reindexare():
|
|
"""Timer-ul nu trebuie sa porneasca peste o sincronizare manuala."""
|
|
import config
|
|
|
|
with config.exclusive():
|
|
with pytest.raises(config.Busy):
|
|
with config.exclusive():
|
|
pass
|
|
|
|
|
|
def test_lacatul_se_elibereaza_dupa_iesire():
|
|
import config
|
|
|
|
with config.exclusive():
|
|
pass
|
|
with config.exclusive(): # trebuie sa mearga din nou
|
|
pass
|
|
|
|
|
|
def test_lacatul_se_elibereaza_si_la_exceptie():
|
|
import config
|
|
|
|
with pytest.raises(ValueError):
|
|
with config.exclusive():
|
|
raise ValueError("ceva")
|
|
with config.exclusive():
|
|
pass
|
|
|
|
|
|
def test_indexul_se_scrie_atomic(write, monkeypatch):
|
|
"""Consumer-ul reciteste indexul la 30s; nu are voie sa prinda JSON pe jumatate."""
|
|
import config
|
|
import indexer
|
|
|
|
write("x.md", "un paragraf oarecare")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.1, 0.2])
|
|
vazute = []
|
|
real_replace = indexer.os.replace
|
|
|
|
def spion(src, dst):
|
|
vazute.append((str(src), str(dst)))
|
|
return real_replace(src, dst)
|
|
|
|
monkeypatch.setattr(indexer.os, "replace", spion)
|
|
indexer.build()
|
|
assert vazute and vazute[0][1] == str(config.INDEX_FILE)
|
|
assert not config.INDEX_FILE.with_suffix(".json.tmp").exists()
|
|
|
|
|
|
def test_xml_invalid_e_raportat_ca_avertisment(write, monkeypatch):
|
|
import indexer
|
|
|
|
write("stricat.xml", "<root><a>x</a></root><in-plus/>")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.0])
|
|
out = indexer.build()
|
|
assert any("stricat.xml" in w for w in out.get("warnings", []))
|
|
|
|
|
|
def test_xml_valid_nu_produce_avertismente(write, monkeypatch):
|
|
import indexer
|
|
|
|
write("bun.xml", "<root><a>x</a></root>")
|
|
monkeypatch.setattr(indexer, "embed", lambda text: [0.0])
|
|
assert "warnings" not in indexer.build()
|
|
|
|
|
|
# --- refolosirea embeddings-urilor la reindexare -----------------------------
|
|
|
|
def test_vectorii_din_indexul_vechi_se_refolosesc(monkeypatch, write, tmp_path):
|
|
"""Un document nou nu are voie sa reforteze embeddings-urile celorlalte.
|
|
|
|
Pe CPU un embedding costa ~7 secunde; fara refolosire, adaugarea unui
|
|
document la un index de 170 de chunk-uri inseamna 20 de minute de asteptare.
|
|
"""
|
|
import json
|
|
import config
|
|
import indexer
|
|
|
|
write("primul.md", "text vechi, deja indexat")
|
|
calculate = []
|
|
|
|
def fals_embed(chunk):
|
|
calculate.append(chunk)
|
|
return [0.5, 0.5]
|
|
|
|
monkeypatch.setattr(indexer, "embed", fals_embed)
|
|
|
|
indexer.build()
|
|
assert calculate == ["text vechi, deja indexat"]
|
|
|
|
calculate.clear()
|
|
write("al-doilea.md", "text nou")
|
|
rezultat = indexer.build()
|
|
|
|
# doar chunk-ul nou s-a calculat
|
|
assert calculate == ["text nou"]
|
|
assert rezultat["embeddings_noi"] == 1
|
|
assert rezultat["embeddings_refolosite"] == 1
|
|
|
|
index = json.loads(config.INDEX_FILE.read_text(encoding="utf-8"))
|
|
assert {e["source"] for e in index} == {"primul.md", "al-doilea.md"}
|
|
|
|
|
|
def test_textul_modificat_se_reembedeaza(monkeypatch, write):
|
|
import indexer
|
|
|
|
write("doc.md", "prima varianta")
|
|
monkeypatch.setattr(indexer, "embed", lambda chunk: [1.0])
|
|
indexer.build()
|
|
|
|
calculate = []
|
|
|
|
def fals_embed(chunk):
|
|
calculate.append(chunk)
|
|
return [2.0]
|
|
|
|
monkeypatch.setattr(indexer, "embed", fals_embed)
|
|
write("doc.md", "varianta modificata")
|
|
rezultat = indexer.build()
|
|
|
|
assert calculate == ["varianta modificata"]
|
|
assert rezultat["embeddings_refolosite"] == 0
|
|
|
|
|
|
# --- documente locale, in afara oglinzii Drive -------------------------------
|
|
|
|
def _scrie_local(name: str, text: str):
|
|
import config
|
|
config.DOCS_LOCAL_DIR.mkdir(parents=True, exist_ok=True)
|
|
p = config.DOCS_LOCAL_DIR / name
|
|
p.write_text(text, encoding="utf-8")
|
|
return p
|
|
|
|
|
|
def test_documentele_locale_intra_in_index(write):
|
|
import store
|
|
write("din-drive.md", "vine din Drive")
|
|
_scrie_local("al-meu.md", "adaugat local")
|
|
|
|
nume = {d["name"] for d in store.documents_for_index()}
|
|
assert nume == {"din-drive.md", "al-meu.md"}
|
|
|
|
|
|
def test_se_vede_care_e_local(write):
|
|
import store
|
|
write("din-drive.md", "x")
|
|
_scrie_local("al-meu.md", "y")
|
|
dupa_nume = {d["name"]: d for d in store.list_documents()}
|
|
assert dupa_nume["al-meu.md"]["local"] is True
|
|
assert dupa_nume["din-drive.md"]["local"] is False
|
|
|
|
|
|
def test_la_acelasi_nume_castiga_versiunea_din_drive(write):
|
|
import store
|
|
_scrie_local("acelasi.md", "copie locala veche")
|
|
write("acelasi.md", "versiunea din Drive")
|
|
assert store.read_document("acelasi.md") == "versiunea din Drive"
|
|
assert len(store.list_documents()) == 1
|
|
|
|
|
|
def test_scrierea_din_dashboard_merge_in_directorul_local():
|
|
# Pana acum ajungea in oglinda Drive si disparea la urmatorul rclone sync.
|
|
import config
|
|
import store
|
|
store.write_document("nota.md", "ceva de tinut minte")
|
|
assert (config.DOCS_LOCAL_DIR / "nota.md").exists()
|
|
assert not (config.DOCS_DIR / "nota.md").exists()
|
|
|
|
|
|
def test_stergerea_gaseste_documentul_in_ambele_directoare(write):
|
|
import store
|
|
_scrie_local("local.md", "x")
|
|
write("drive.md", "y")
|
|
assert store.delete_document("local.md") is True
|
|
assert store.delete_document("drive.md") is True
|
|
assert store.delete_document("inexistent.md") is False
|
|
|
|
|
|
def test_schimbarea_modelului_reface_embeddings(monkeypatch, write):
|
|
"""Alt EMBED_MODEL = alti vectori; refolosirea lor ar amesteca tacut doua modele."""
|
|
import config
|
|
import indexer
|
|
|
|
write("doc.md", "acelasi text, alt model")
|
|
monkeypatch.setattr(indexer, "embed", lambda chunk: [1.0])
|
|
indexer.build()
|
|
|
|
calculate = []
|
|
monkeypatch.setattr(indexer, "embed", lambda chunk: calculate.append(chunk) or [2.0])
|
|
monkeypatch.setattr(config, "get", lambda k, d=None: "alt-model"
|
|
if k == "EMBED_MODEL" else config.DEFAULTS.get(k, d))
|
|
|
|
rezultat = indexer.build()
|
|
assert calculate == ["acelasi text, alt model"]
|
|
assert rezultat["embeddings_refolosite"] == 0
|
|
|
|
|
|
def test_indexul_fara_camp_model_se_recalculeaza_o_data(monkeypatch, write):
|
|
"""Intrarile scrise inainte de campul `model` nu se pot atribui unui model."""
|
|
import json
|
|
import config
|
|
import indexer
|
|
|
|
write("doc.md", "text vechi")
|
|
config.STATE_DIR.mkdir(parents=True, exist_ok=True)
|
|
config.INDEX_FILE.write_text(json.dumps(
|
|
[{"source": "doc.md", "chunk": 0, "text": "text vechi", "embedding": [9.0]}]),
|
|
encoding="utf-8")
|
|
|
|
calculate = []
|
|
monkeypatch.setattr(indexer, "embed", lambda chunk: calculate.append(chunk) or [1.0])
|
|
assert indexer.build()["embeddings_refolosite"] == 0
|
|
assert calculate == ["text vechi"]
|
|
|
|
calculate.clear()
|
|
assert indexer.build()["embeddings_refolosite"] == 1
|
|
assert calculate == []
|