- test_fixture_sweep: extract+validate every fixture PDF on every run; known extractor/validator gaps are catalogued explicitly and the sweep fails on any deviation from that catalogue - tests/fixtures/real/: anonymized XML from real production invoices that revealed bugs (verified to contain no real-world data); first entry locks in the FC-tax-number/basis-quantity/spaced-VAT-ID regressions - CI workflow: uv sync --frozen from lockfile + pytest + docker build
89 lines
2.6 KiB
Python
89 lines
2.6 KiB
Python
"""Regression sweep over all invoice fixtures.
|
|
|
|
Every fixture PDF is extracted and validated end-to-end on every test run.
|
|
Known limitations of extractor and validator are catalogued explicitly in
|
|
KNOWN_GAPS and VALIDATION_EXCLUDED below. The sweep fails when:
|
|
|
|
- any fixture stops extracting (exception),
|
|
- a fixture that used to be clean produces new critical errors,
|
|
- a fixture with known gaps produces more/different errors than catalogued,
|
|
- a gap was fixed but its catalogue entry was not removed.
|
|
|
|
So: fix a gap, then remove its entry — the sweep enforces honest bookkeeping.
|
|
"""
|
|
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from src.extractor import extract_zugferd
|
|
from src.models import ValidateRequest
|
|
from src.validator import validate_invoice
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures"
|
|
ALL_CHECKS = ["pflichtfelder", "betraege", "ustid", "pdf_abgleich"]
|
|
|
|
EXPECTED_NON_ZUGFERD = {"EmptyPDFA1.pdf"}
|
|
|
|
VALIDATION_EXCLUDED = {
|
|
"ORDER-X_EX01_ORDER_FULL_DATA-COMFORT.pdf": (
|
|
"ORDER-X order document, not an invoice"
|
|
),
|
|
"zugferd_invoice.pdf": (
|
|
"ZUGFeRD 1.0 uses a different XML namespace (not supported)"
|
|
),
|
|
}
|
|
|
|
KNOWN_GAPS: dict[str, set[str]] = {
|
|
"EN16931_1_Teilrechnung.pdf": {
|
|
"line_items[3].line_total",
|
|
"totals.net",
|
|
},
|
|
"MustangBeispiel20221026.pdf": {
|
|
"totals.net",
|
|
},
|
|
"validAvoir_FR_type380_BASICWL.pdf": {
|
|
"line_items",
|
|
"totals.net",
|
|
"vat_id",
|
|
},
|
|
"zugferd_2p1_EXTENDED_PDFA-3A.pdf": {
|
|
"totals.net",
|
|
},
|
|
}
|
|
|
|
|
|
def _pdf_files() -> list[str]:
|
|
return sorted(path.name for path in FIXTURES.glob("*.pdf"))
|
|
|
|
|
|
@pytest.mark.parametrize("filename", _pdf_files())
|
|
def test_fixture_sweep(filename: str) -> None:
|
|
"""Extract and validate one fixture; deviations from the catalogue fail."""
|
|
pdf_bytes = (FIXTURES / filename).read_bytes()
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
if filename in EXPECTED_NON_ZUGFERD:
|
|
assert result.is_zugferd is False
|
|
return
|
|
|
|
assert result.is_zugferd is True, f"{filename}: expected ZUGFeRD XML"
|
|
assert result.xml_data is not None
|
|
|
|
if filename in VALIDATION_EXCLUDED:
|
|
return
|
|
|
|
request = ValidateRequest(
|
|
xml_data=result.xml_data.model_dump(),
|
|
pdf_text=result.pdf_text,
|
|
checks=ALL_CHECKS,
|
|
)
|
|
validation = validate_invoice(request)
|
|
critical_fields = {error.field for error in validation.errors}
|
|
expected_fields = KNOWN_GAPS.get(filename, set())
|
|
|
|
assert critical_fields == expected_fields, (
|
|
f"{filename}: critical fields changed — update KNOWN_GAPS "
|
|
f"(new state: {sorted(critical_fields)})"
|
|
)
|