date/balanta_MASTERJOB.csv -> date/balanta_MASTER.csv: numele intermediarului urmeaza intotdeauna --firma, deci `py genereaza.py --firma MASTER` merge acum fara --balanta explicit. Firma se cheama in acte MASTERJOB, dar in unealta si in ROA e MASTER; documentele care vorbesc despre firma pastreaza numele real. Handoff: sectiune noua "Ce urmeaza", cu proba pe un client real ca prim pas, si sectiunea de riscuri asumate in locul listei de lucruri ramase. plan_solduri2roa.md: pasul 2 marcat terminat, cu trimitere la handoff. 33 de teste trec, zero skip-uri. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EYeAtVxeS8m4oXekjX8Am2
213 lines
8.6 KiB
Python
213 lines
8.6 KiB
Python
"""Poarta finala: de la PDF-urile originale la xlsx-urile de import ROA.
|
|
|
|
Complementul lui tests/test_regresie.py. Acela verifica biblioteca pornind de la
|
|
balantele CSV de pe disc; acesta ruleaza EXACT ce ruleaza un om in teren, ca
|
|
subprocese (deci si parsarea argumentelor e sub test):
|
|
|
|
py extrage.py --sursa pdf --fisier "<pdf>" --firma <FIRMA> --an 2025 --luna 12 --dir <temp>
|
|
py genereaza.py --firma <FIRMA> --an 2025 --luna 12 --dir <temp>
|
|
|
|
Iesirile se scriu in temporar (nimic in radacina), se anonimizeaza cu ACELEASI
|
|
reguli si ACEIASI tabela ca tests/anonimizeaza_etalon.py si se compara cu
|
|
tests/golden/ pe VALORI DE CELULA, nu pe hash de fisier.
|
|
|
|
Asocierea PDF -> firma nu se presupune: fiecare balanta extrasa se compara cu
|
|
balanta de referinta din date/ (balanta_FUNDATIA.csv / balanta_MASTER.csv),
|
|
care e cea folosita la constructia etalonului.
|
|
|
|
Daca iesirea difera de etalon, etalonul are dreptate: se raporteaza, nu se ajusteaza.
|
|
|
|
Rulare: py -m unittest discover -s tests -v
|
|
"""
|
|
import csv
|
|
import pathlib
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
import unittest
|
|
|
|
import openpyxl
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
RADACINA = pathlib.Path(__file__).resolve().parent.parent
|
|
TESTS = pathlib.Path(__file__).resolve().parent
|
|
for cale in (str(RADACINA), str(TESTS)):
|
|
if cale not in sys.path:
|
|
sys.path.insert(0, cale)
|
|
|
|
import anonimizeaza_etalon as anonim
|
|
|
|
CORESP = TESTS / "_corespondenta_anonimizare.json"
|
|
GOLDEN = TESTS / "golden"
|
|
DATE = RADACINA / "date"
|
|
EXTRAGE = RADACINA / "extrage.py"
|
|
GENEREAZA = RADACINA / "genereaza.py"
|
|
|
|
AN, LUNA = 2025, 12
|
|
MAX_DIFERENTE_RAPORTATE = 20
|
|
|
|
# PDF din date/ -> firma -> balanta de referinta din date/ -> etalon.
|
|
# Asocierea e verificata de test_pdf_corespunde_referintei, nu doar afirmata.
|
|
FIRME = [
|
|
("FUNDATIA", "IBB 31.12.2025.pdf", "balanta_FUNDATIA.csv",
|
|
"golden_FUNDATIA_2025_12.xlsx"),
|
|
("MASTER", "MJC 31.12.2025.pdf", "balanta_MASTER.csv",
|
|
"golden_MASTER_2025_12.xlsx"),
|
|
]
|
|
|
|
COLOANE_REF = ["cont", "denumire", "prec_d", "prec_c", "rulaj_d", "rulaj_c",
|
|
"total_d", "total_c", "sold_d", "sold_c", "este_total"]
|
|
COLOANE_SUMA = {"prec_d", "prec_c", "rulaj_d", "rulaj_c",
|
|
"total_d", "total_c", "sold_d", "sold_c"}
|
|
TOL = 0.001
|
|
|
|
|
|
def _ruleaza(script, argumente, cwd):
|
|
"""Ruleaza un punct de intrare ca subproces; orice cod de iesire != 0 e esec."""
|
|
comanda = [sys.executable, str(script)] + [str(a) for a in argumente]
|
|
rez = subprocess.run(comanda, cwd=str(cwd), capture_output=True, text=True)
|
|
if rez.returncode != 0:
|
|
raise AssertionError(
|
|
"comanda a esuat (cod %d): %s\n--- stdout ---\n%s\n--- stderr ---\n%s"
|
|
% (rez.returncode, " ".join(comanda), rez.stdout, rez.stderr))
|
|
return rez
|
|
|
|
|
|
def _citeste_csv(cale):
|
|
with open(cale, encoding="utf-8-sig", newline="") as f:
|
|
return list(csv.DictReader(f))
|
|
|
|
|
|
def _diferente_referinta(gasit, referinta):
|
|
"""Diferente pe coloanele comune, ca lista (rand, coloana, asteptat, gasit)."""
|
|
if len(gasit) != len(referinta):
|
|
return [(0, "nr_randuri", len(referinta), len(gasit))]
|
|
dif = []
|
|
for i, (a, b) in enumerate(zip(gasit, referinta)):
|
|
for c in COLOANE_REF:
|
|
va, vb = a.get(c), b.get(c)
|
|
if c in COLOANE_SUMA:
|
|
if va in (None, "") or vb in (None, ""):
|
|
if (va or "").strip() != (vb or "").strip():
|
|
dif.append((i + 1, c, vb, va))
|
|
elif abs(float(va) - float(vb)) > TOL:
|
|
dif.append((i + 1, c, vb, va))
|
|
elif (va or "").strip() != (vb or "").strip():
|
|
dif.append((i + 1, c, vb, va))
|
|
return dif
|
|
|
|
|
|
class TestRegresieCapatLaCapat(unittest.TestCase):
|
|
|
|
@classmethod
|
|
def setUpClass(cls):
|
|
lipsa = [pdf for _, pdf, _, _ in FIRME if not (DATE / pdf).exists()]
|
|
if lipsa:
|
|
raise unittest.SkipTest("lipsesc PDF-urile din date/: %s"
|
|
% ", ".join(lipsa))
|
|
if not CORESP.exists():
|
|
raise unittest.SkipTest(
|
|
"lipseste %s (e ignorata de git). Etalonul nu se poate reproduce "
|
|
"fara tabela de corespondenta." % CORESP)
|
|
cls.coresp = anonim.citeste_corespondenta(CORESP)
|
|
|
|
cls._tmp = tempfile.TemporaryDirectory(prefix="regresie_cap_")
|
|
tmpdir = pathlib.Path(cls._tmp.name)
|
|
cls.balante = {}
|
|
cls.iesiri = {}
|
|
try:
|
|
for firma, pdf, _, _ in FIRME:
|
|
_ruleaza(EXTRAGE, ["--sursa", "pdf", "--fisier", str(DATE / pdf),
|
|
"--firma", firma, "--an", AN, "--luna", LUNA,
|
|
"--dir", str(tmpdir)], tmpdir)
|
|
cls.balante[firma] = tmpdir / ("balanta_%s.csv" % firma)
|
|
|
|
for firma, _, _, _ in FIRME:
|
|
_ruleaza(GENEREAZA, ["--firma", firma, "--an", AN, "--luna", LUNA,
|
|
"--dir", str(tmpdir),
|
|
"--dir-iesire", str(tmpdir)], tmpdir)
|
|
|
|
for firma, _, _, golden_nume in FIRME:
|
|
brut = tmpdir / ("init_%s_%d_%d.xlsx" % (firma, AN, LUNA))
|
|
anonimizat = tmpdir / ("anon_%s.xlsx" % firma)
|
|
anonim.anonimizeaza_fisier(str(brut), str(anonimizat), cls.coresp)
|
|
cls.iesiri[firma] = (anonimizat, GOLDEN / golden_nume)
|
|
except Exception:
|
|
cls._tmp.cleanup()
|
|
raise
|
|
|
|
@classmethod
|
|
def tearDownClass(cls):
|
|
cls._tmp.cleanup()
|
|
|
|
def _verifica_referinta(self, firma):
|
|
"""Balanta extrasa din PDF = balanta de referinta (verifica asocierea PDF->firma)."""
|
|
referinta = DATE / dict(
|
|
(f, r) for f, _, r, _ in FIRME)[firma]
|
|
if not referinta.exists():
|
|
self.skipTest("lipseste referinta %s" % referinta)
|
|
gasit = _citeste_csv(self.balante[firma])
|
|
asteptat = _citeste_csv(referinta)
|
|
diferente = _diferente_referinta(gasit, asteptat)
|
|
if diferente:
|
|
linii = ["%s: %d diferente fata de %s (PDF-ul nu e al firmei %s?)"
|
|
% (self.balante[firma].name, len(diferente), referinta.name, firma)]
|
|
for rand, col, asteptat_v, gasit_v in diferente[:MAX_DIFERENTE_RAPORTATE]:
|
|
linii.append(" rand %s, coloana %s: asteptat %r, gasit %r"
|
|
% (rand, col, asteptat_v, gasit_v))
|
|
self.fail("\n".join(linii))
|
|
|
|
def _verifica_etalon(self, firma):
|
|
anonimizat, golden = self.iesiri[firma]
|
|
self.assertTrue(golden.exists(), "lipseste etalonul %s" % golden)
|
|
wb_gasit = openpyxl.load_workbook(anonimizat)
|
|
wb_asteptat = openpyxl.load_workbook(golden)
|
|
try:
|
|
ws_gasit, ws_asteptat = wb_gasit.active, wb_asteptat.active
|
|
nr_g, nc_g = ws_gasit.max_row, ws_gasit.max_column
|
|
nr_a, nc_a = ws_asteptat.max_row, ws_asteptat.max_column
|
|
self.assertEqual(
|
|
(nr_g, nc_g), (nr_a, nc_a),
|
|
"Dimensiuni diferite pentru %s: asteptat %d randuri x %d coloane, "
|
|
"gasit %d randuri x %d coloane" % (golden.name, nr_a, nc_a, nr_g, nc_g))
|
|
|
|
diferente = []
|
|
for rand in range(1, max(nr_g, nr_a) + 1):
|
|
for col in range(1, max(nc_g, nc_a) + 1):
|
|
gasit = ws_gasit.cell(rand, col).value
|
|
asteptat = ws_asteptat.cell(rand, col).value
|
|
if gasit != asteptat:
|
|
diferente.append((rand, col, asteptat, gasit))
|
|
|
|
if diferente:
|
|
linii = ["%s: %d celule difera (fluxul din PDF nu reproduce etalonul):"
|
|
% (golden.name, len(diferente))]
|
|
for rand, col, asteptat, gasit in diferente[:MAX_DIFERENTE_RAPORTATE]:
|
|
linii.append(
|
|
" %s: rand %d, coloana %d (%s): asteptat %r, gasit %r"
|
|
% (golden.name, rand, col, get_column_letter(col),
|
|
asteptat, gasit))
|
|
if len(diferente) > MAX_DIFERENTE_RAPORTATE:
|
|
linii.append(" ... si alte %d celule"
|
|
% (len(diferente) - MAX_DIFERENTE_RAPORTATE))
|
|
self.fail("\n".join(linii))
|
|
finally:
|
|
wb_gasit.close()
|
|
wb_asteptat.close()
|
|
|
|
def test_pdf_fundatia_corespunde_referintei(self):
|
|
self._verifica_referinta("FUNDATIA")
|
|
|
|
def test_pdf_master_corespunde_referintei(self):
|
|
self._verifica_referinta("MASTER")
|
|
|
|
def test_etalon_fundatia(self):
|
|
self._verifica_etalon("FUNDATIA")
|
|
|
|
def test_etalon_master(self):
|
|
self._verifica_etalon("MASTER")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|