Lane regresie: poarta finala trece, etapa 2 e gata
tests/test_regresie_capat_la_capat.py ruleaza extrage.py si genereaza.py ca subprocese, pornind de la PDF-urile originale, si compara rezultatul anonimizat cu tests/golden/ pe valori de celula: 0 diferente pe ambele firme. Fluxul nou reproduce exact xlsx-urile deja importate in ROACONT. 33 de teste trec, fara niciun skip. CLAUDE.md: dependintele noi (xlrd, firebird-driver) si punctele de intrare. Handoff-ul rescris pe starea de final, cu ce NU e dovedit: cititorul FDB e verificat doar pe cazul sintetic al firmei de joaca si nu se duce la un client fara rerularea portii pe datele lui. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EYeAtVxeS8m4oXekjX8Am2
This commit is contained in:
210
tests/test_regresie_capat_la_capat.py
Normal file
210
tests/test_regresie_capat_la_capat.py
Normal file
@@ -0,0 +1,210 @@
|
||||
"""Poarta finala: de la PDF-urile originale la xlsx-urile de import ROA.
|
||||
|
||||
Complementul lui tests/test_regresie.py. Acela verifica biblioteca pornind de la
|
||||
balantele CSV de pe disc; acesta ruleaza EXACT ce ruleaza un om in teren, ca
|
||||
subprocese (deci si parsarea argumentelor e sub test):
|
||||
|
||||
py extrage.py --sursa pdf --fisier "<pdf>" --firma <FIRMA> --an 2025 --luna 12 --dir <temp>
|
||||
py genereaza.py --firma <FIRMA> --an 2025 --luna 12 --dir <temp>
|
||||
|
||||
Iesirile se scriu in temporar (nimic in radacina), se anonimizeaza cu ACELEASI
|
||||
reguli si ACEIASI tabela ca tests/anonimizeaza_etalon.py si se compara cu
|
||||
tests/golden/ pe VALORI DE CELULA, nu pe hash de fisier.
|
||||
|
||||
Asocierea PDF -> firma nu se presupune: fiecare balanta extrasa se compara cu
|
||||
balanta de referinta din radacina (balanta_FUNDATIA.csv / balanta_MASTERJOB.csv),
|
||||
care e cea folosita la constructia etalonului.
|
||||
|
||||
Daca iesirea difera de etalon, etalonul are dreptate: se raporteaza, nu se ajusteaza.
|
||||
|
||||
Rulare: py -m unittest discover -s tests -v
|
||||
"""
|
||||
import csv
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
import openpyxl
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
RADACINA = pathlib.Path(__file__).resolve().parent.parent
|
||||
TESTS = pathlib.Path(__file__).resolve().parent
|
||||
for cale in (str(RADACINA), str(TESTS)):
|
||||
if cale not in sys.path:
|
||||
sys.path.insert(0, cale)
|
||||
|
||||
import anonimizeaza_etalon as anonim
|
||||
|
||||
CORESP = TESTS / "_corespondenta_anonimizare.json"
|
||||
GOLDEN = TESTS / "golden"
|
||||
EXTRAGE = RADACINA / "extrage.py"
|
||||
GENEREAZA = RADACINA / "genereaza.py"
|
||||
|
||||
AN, LUNA = 2025, 12
|
||||
MAX_DIFERENTE_RAPORTATE = 20
|
||||
|
||||
# PDF din radacina -> firma -> balanta de referinta din radacina -> etalon.
|
||||
# Asocierea e verificata de test_pdf_corespunde_referintei, nu doar afirmata.
|
||||
FIRME = [
|
||||
("FUNDATIA", "IBB 31.12.2025.pdf", "balanta_FUNDATIA.csv",
|
||||
"golden_FUNDATIA_2025_12.xlsx"),
|
||||
("MASTER", "MJC 31.12.2025.pdf", "balanta_MASTERJOB.csv",
|
||||
"golden_MASTER_2025_12.xlsx"),
|
||||
]
|
||||
|
||||
COLOANE_REF = ["cont", "denumire", "prec_d", "prec_c", "rulaj_d", "rulaj_c",
|
||||
"total_d", "total_c", "sold_d", "sold_c", "este_total"]
|
||||
COLOANE_SUMA = {"prec_d", "prec_c", "rulaj_d", "rulaj_c",
|
||||
"total_d", "total_c", "sold_d", "sold_c"}
|
||||
TOL = 0.001
|
||||
|
||||
|
||||
def _ruleaza(script, argumente, cwd):
|
||||
"""Ruleaza un punct de intrare ca subproces; orice cod de iesire != 0 e esec."""
|
||||
comanda = [sys.executable, str(script)] + [str(a) for a in argumente]
|
||||
rez = subprocess.run(comanda, cwd=str(cwd), capture_output=True, text=True)
|
||||
if rez.returncode != 0:
|
||||
raise AssertionError(
|
||||
"comanda a esuat (cod %d): %s\n--- stdout ---\n%s\n--- stderr ---\n%s"
|
||||
% (rez.returncode, " ".join(comanda), rez.stdout, rez.stderr))
|
||||
return rez
|
||||
|
||||
|
||||
def _citeste_csv(cale):
|
||||
with open(cale, encoding="utf-8-sig", newline="") as f:
|
||||
return list(csv.DictReader(f))
|
||||
|
||||
|
||||
def _diferente_referinta(gasit, referinta):
|
||||
"""Diferente pe coloanele comune, ca lista (rand, coloana, asteptat, gasit)."""
|
||||
if len(gasit) != len(referinta):
|
||||
return [(0, "nr_randuri", len(referinta), len(gasit))]
|
||||
dif = []
|
||||
for i, (a, b) in enumerate(zip(gasit, referinta)):
|
||||
for c in COLOANE_REF:
|
||||
va, vb = a.get(c), b.get(c)
|
||||
if c in COLOANE_SUMA:
|
||||
if va in (None, "") or vb in (None, ""):
|
||||
if (va or "").strip() != (vb or "").strip():
|
||||
dif.append((i + 1, c, vb, va))
|
||||
elif abs(float(va) - float(vb)) > TOL:
|
||||
dif.append((i + 1, c, vb, va))
|
||||
elif (va or "").strip() != (vb or "").strip():
|
||||
dif.append((i + 1, c, vb, va))
|
||||
return dif
|
||||
|
||||
|
||||
class TestRegresieCapatLaCapat(unittest.TestCase):
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
lipsa = [pdf for _, pdf, _, _ in FIRME if not (RADACINA / pdf).exists()]
|
||||
if lipsa:
|
||||
raise unittest.SkipTest("lipsesc PDF-urile din radacina: %s"
|
||||
% ", ".join(lipsa))
|
||||
if not CORESP.exists():
|
||||
raise unittest.SkipTest(
|
||||
"lipseste %s (e ignorata de git). Etalonul nu se poate reproduce "
|
||||
"fara tabela de corespondenta." % CORESP)
|
||||
cls.coresp = anonim.citeste_corespondenta(CORESP)
|
||||
|
||||
cls._tmp = tempfile.TemporaryDirectory(prefix="regresie_cap_")
|
||||
tmpdir = pathlib.Path(cls._tmp.name)
|
||||
cls.balante = {}
|
||||
cls.iesiri = {}
|
||||
try:
|
||||
for firma, pdf, _, _ in FIRME:
|
||||
_ruleaza(EXTRAGE, ["--sursa", "pdf", "--fisier", str(RADACINA / pdf),
|
||||
"--firma", firma, "--an", AN, "--luna", LUNA,
|
||||
"--dir", str(tmpdir)], tmpdir)
|
||||
cls.balante[firma] = tmpdir / ("balanta_%s.csv" % firma)
|
||||
|
||||
for firma, _, _, _ in FIRME:
|
||||
_ruleaza(GENEREAZA, ["--firma", firma, "--an", AN, "--luna", LUNA,
|
||||
"--dir", str(tmpdir)], tmpdir)
|
||||
|
||||
for firma, _, _, golden_nume in FIRME:
|
||||
brut = tmpdir / ("init_%s_%d_%d.xlsx" % (firma, AN, LUNA))
|
||||
anonimizat = tmpdir / ("anon_%s.xlsx" % firma)
|
||||
anonim.anonimizeaza_fisier(str(brut), str(anonimizat), cls.coresp)
|
||||
cls.iesiri[firma] = (anonimizat, GOLDEN / golden_nume)
|
||||
except Exception:
|
||||
cls._tmp.cleanup()
|
||||
raise
|
||||
|
||||
@classmethod
|
||||
def tearDownClass(cls):
|
||||
cls._tmp.cleanup()
|
||||
|
||||
def _verifica_referinta(self, firma):
|
||||
"""Balanta extrasa din PDF = balanta de referinta (verifica asocierea PDF->firma)."""
|
||||
referinta = RADACINA / dict(
|
||||
(f, r) for f, _, r, _ in FIRME)[firma]
|
||||
if not referinta.exists():
|
||||
self.skipTest("lipseste referinta %s" % referinta)
|
||||
gasit = _citeste_csv(self.balante[firma])
|
||||
asteptat = _citeste_csv(referinta)
|
||||
diferente = _diferente_referinta(gasit, asteptat)
|
||||
if diferente:
|
||||
linii = ["%s: %d diferente fata de %s (PDF-ul nu e al firmei %s?)"
|
||||
% (self.balante[firma].name, len(diferente), referinta.name, firma)]
|
||||
for rand, col, asteptat_v, gasit_v in diferente[:MAX_DIFERENTE_RAPORTATE]:
|
||||
linii.append(" rand %s, coloana %s: asteptat %r, gasit %r"
|
||||
% (rand, col, asteptat_v, gasit_v))
|
||||
self.fail("\n".join(linii))
|
||||
|
||||
def _verifica_etalon(self, firma):
|
||||
anonimizat, golden = self.iesiri[firma]
|
||||
self.assertTrue(golden.exists(), "lipseste etalonul %s" % golden)
|
||||
wb_gasit = openpyxl.load_workbook(anonimizat)
|
||||
wb_asteptat = openpyxl.load_workbook(golden)
|
||||
try:
|
||||
ws_gasit, ws_asteptat = wb_gasit.active, wb_asteptat.active
|
||||
nr_g, nc_g = ws_gasit.max_row, ws_gasit.max_column
|
||||
nr_a, nc_a = ws_asteptat.max_row, ws_asteptat.max_column
|
||||
self.assertEqual(
|
||||
(nr_g, nc_g), (nr_a, nc_a),
|
||||
"Dimensiuni diferite pentru %s: asteptat %d randuri x %d coloane, "
|
||||
"gasit %d randuri x %d coloane" % (golden.name, nr_a, nc_a, nr_g, nc_g))
|
||||
|
||||
diferente = []
|
||||
for rand in range(1, max(nr_g, nr_a) + 1):
|
||||
for col in range(1, max(nc_g, nc_a) + 1):
|
||||
gasit = ws_gasit.cell(rand, col).value
|
||||
asteptat = ws_asteptat.cell(rand, col).value
|
||||
if gasit != asteptat:
|
||||
diferente.append((rand, col, asteptat, gasit))
|
||||
|
||||
if diferente:
|
||||
linii = ["%s: %d celule difera (fluxul din PDF nu reproduce etalonul):"
|
||||
% (golden.name, len(diferente))]
|
||||
for rand, col, asteptat, gasit in diferente[:MAX_DIFERENTE_RAPORTATE]:
|
||||
linii.append(
|
||||
" %s: rand %d, coloana %d (%s): asteptat %r, gasit %r"
|
||||
% (golden.name, rand, col, get_column_letter(col),
|
||||
asteptat, gasit))
|
||||
if len(diferente) > MAX_DIFERENTE_RAPORTATE:
|
||||
linii.append(" ... si alte %d celule"
|
||||
% (len(diferente) - MAX_DIFERENTE_RAPORTATE))
|
||||
self.fail("\n".join(linii))
|
||||
finally:
|
||||
wb_gasit.close()
|
||||
wb_asteptat.close()
|
||||
|
||||
def test_pdf_fundatia_corespunde_referintei(self):
|
||||
self._verifica_referinta("FUNDATIA")
|
||||
|
||||
def test_pdf_master_corespunde_referintei(self):
|
||||
self._verifica_referinta("MASTER")
|
||||
|
||||
def test_etalon_fundatia(self):
|
||||
self._verifica_etalon("FUNDATIA")
|
||||
|
||||
def test_etalon_master(self):
|
||||
self._verifica_etalon("MASTER")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user