- supplier/buyer: extract FC tax registration (BT-32/BT-49) into new tax_number field; previously only VA (USt-IdNr.) was read - pflichtfelder: critical error only when neither vat_id nor tax_number is present (EN 16931 BR-CO-26) - ustid: tolerate formatting whitespace inside VAT IDs (e.g. 'DE 140 978 617') - line items: normalize unit price by basis quantity (BT-149), so a price per 100 units no longer inflates line totals by factor 100 - supplier email: also look in DefinedTradeContact (BT-42) - rewrite xrechnung fallback test to assert only our own behavior, independent of factur-x version specifics
521 lines
18 KiB
Python
521 lines
18 KiB
Python
"""Tests for ZUGFeRD extractor.
|
|
|
|
Tests are written following TDD: FAILING TESTS FIRST (RED phase),
|
|
then implementation makes them pass (GREEN phase).
|
|
"""
|
|
|
|
import base64
|
|
import binascii
|
|
|
|
import pytest
|
|
|
|
from pypdf import PdfReader, PdfWriter
|
|
|
|
|
|
class TestExtractionError:
|
|
"""Test ExtractionError exception class."""
|
|
|
|
def test_extraction_error_initialization(self):
|
|
"""Test ExtractionError can be created with all fields."""
|
|
from src.extractor import ExtractionError
|
|
|
|
error = ExtractionError(
|
|
error_code="corrupt_pdf",
|
|
message="PDF is corrupted",
|
|
details="Trailer not found",
|
|
)
|
|
assert error.error_code == "corrupt_pdf"
|
|
assert error.message == "PDF is corrupted"
|
|
assert error.details == "Trailer not found"
|
|
|
|
def test_extraction_error_without_details(self):
|
|
"""Test ExtractionError can be created without details."""
|
|
from src.extractor import ExtractionError
|
|
|
|
error = ExtractionError(error_code="invalid_pdf", message="Not a PDF file")
|
|
assert error.error_code == "invalid_pdf"
|
|
assert error.message == "Not a PDF file"
|
|
assert error.details == ""
|
|
|
|
def test_extraction_error_is_exception(self):
|
|
"""Test ExtractionError is a proper exception."""
|
|
from src.extractor import ExtractionError
|
|
|
|
error = ExtractionError(error_code="file_too_large", message="File too large")
|
|
assert isinstance(error, Exception)
|
|
assert str(error) == "File too large"
|
|
|
|
|
|
class TestFileSizeValidation:
|
|
"""Test file size validation in extract_zugferd()."""
|
|
|
|
def test_file_size_limit_exactly_10mb(self):
|
|
"""Test PDF exactly at 10MB limit passes size check but fails PDF parsing."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
# 10MB = 10 * 1024 * 1024 bytes
|
|
large_pdf = b"X" * (10 * 1024 * 1024)
|
|
|
|
# 10MB exactly is allowed (not > 10MB), but invalid PDF data causes parse error
|
|
with pytest.raises(ExtractionError) as exc_info:
|
|
extract_zugferd(large_pdf)
|
|
|
|
# Should fail PDF parsing, not file size check
|
|
assert exc_info.value.error_code in ["corrupt_pdf", "invalid_pdf"]
|
|
|
|
def test_file_size_limit_10mb_plus_one_byte(self):
|
|
"""Test PDF one byte over 10MB limit is rejected."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
# 10MB + 1 byte
|
|
too_large = b"X" * (10 * 1024 * 1024 + 1)
|
|
|
|
with pytest.raises(ExtractionError) as exc_info:
|
|
extract_zugferd(too_large)
|
|
|
|
assert exc_info.value.error_code == "file_too_large"
|
|
|
|
def test_file_size_under_10mb_accepted(self):
|
|
"""Test PDF under 10MB is accepted for processing."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
# Small PDF (9MB)
|
|
small_pdf = b"X" * (9 * 1024 * 1024)
|
|
|
|
# Should process (even if invalid PDF, different error)
|
|
try:
|
|
extract_zugferd(small_pdf)
|
|
except ExtractionError as e:
|
|
# Different error is expected (not file_too_large)
|
|
assert e.error_code != "file_too_large"
|
|
|
|
|
|
class TestNonZUGFeRDPDF:
|
|
"""Test extraction from PDF without ZUGFeRD XML."""
|
|
|
|
def test_non_zugferd_pdf(self):
|
|
"""Test PDF without ZUGFeRD XML returns is_zugferd=False."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
# Load non-ZUGFeRD sample PDF
|
|
with open("tests/fixtures/EmptyPDFA1.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.is_zugferd is False
|
|
assert result.zugferd_profil is None
|
|
assert result.xml_raw is None
|
|
assert result.xml_data is None
|
|
assert result.pdf_text is not None
|
|
assert len(result.pdf_text) > 0
|
|
assert result.extraction_meta.pages >= 1
|
|
assert result.extraction_meta.extraction_time_ms >= 0
|
|
|
|
|
|
class TestEN16931Extraction:
|
|
"""Test extraction from EN16931 profile PDF."""
|
|
|
|
def test_extract_en16931_profile(self):
|
|
"""Test EN16931 PDF extraction detects correct profile."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/EN16931_Einfach.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.is_zugferd is True
|
|
assert result.zugferd_profil == "EN16931"
|
|
assert result.xml_raw is not None
|
|
assert len(result.xml_raw) > 0
|
|
assert result.xml_data is not None
|
|
assert result.pdf_text is not None
|
|
assert result.extraction_meta.xml_attachment_name is not None
|
|
assert result.extraction_meta.pages >= 1
|
|
assert result.extraction_meta.extraction_time_ms >= 0
|
|
|
|
def test_extract_all_required_fields(self):
|
|
"""Test all XmlData fields are populated from EN16931."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/EN16931_Einfach.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.xml_data is not None
|
|
xml_data = result.xml_data
|
|
|
|
# Required fields
|
|
assert xml_data.invoice_number is not None and len(xml_data.invoice_number) > 0
|
|
assert xml_data.invoice_date is not None and len(xml_data.invoice_date) > 0
|
|
assert xml_data.supplier is not None
|
|
assert xml_data.buyer is not None
|
|
assert xml_data.line_items is not None
|
|
assert xml_data.totals is not None
|
|
|
|
# Supplier fields
|
|
assert xml_data.supplier.name is not None and len(xml_data.supplier.name) > 0
|
|
|
|
# Buyer fields
|
|
assert xml_data.buyer.name is not None and len(xml_data.buyer.name) > 0
|
|
|
|
# Line items
|
|
assert len(xml_data.line_items) > 0
|
|
first_item = xml_data.line_items[0]
|
|
assert first_item.position >= 1
|
|
assert first_item.description is not None and len(first_item.description) > 0
|
|
assert first_item.quantity > 0
|
|
assert first_item.unit is not None and len(first_item.unit) > 0
|
|
assert first_item.unit_price is not None and first_item.unit_price > 0
|
|
assert first_item.line_total is not None and first_item.line_total > 0
|
|
|
|
# Totals
|
|
assert xml_data.totals.line_total_sum > 0
|
|
assert xml_data.totals.net > 0
|
|
assert xml_data.totals.vat_total >= 0
|
|
assert xml_data.totals.gross > 0
|
|
|
|
|
|
class TestErrorHandling:
|
|
"""Test error handling for various PDF issues."""
|
|
|
|
def test_corrupt_pdf_raises_error(self):
|
|
"""Test corrupt PDF raises ExtractionError with correct code."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
# Invalid PDF data
|
|
corrupt_pdf = b"NOT A PDF FILE AT ALL"
|
|
|
|
with pytest.raises(ExtractionError) as exc_info:
|
|
extract_zugferd(corrupt_pdf)
|
|
|
|
# Should raise either corrupt_pdf or invalid_pdf
|
|
assert exc_info.value.error_code in ["corrupt_pdf", "invalid_pdf"]
|
|
|
|
def test_empty_pdf_raises_error(self):
|
|
"""Test empty PDF raises ExtractionError."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
with pytest.raises(ExtractionError):
|
|
extract_zugferd(b"")
|
|
|
|
def test_invalid_base64(self):
|
|
"""Test invalid base64 raises ExtractionError."""
|
|
from src.extractor import ExtractionError, extract_zugferd
|
|
|
|
# This would be called by API layer, but we can test the concept
|
|
# Invalid PDF that's not valid base64-encoded
|
|
try:
|
|
invalid_base64 = b"$$$INVALID$$$"
|
|
# If API layer decodes invalid base64, it gets error
|
|
decoded = base64.b64decode(invalid_base64, validate=True)
|
|
extract_zugferd(decoded)
|
|
except (binascii.Error, ValueError):
|
|
# base64 error is expected
|
|
pass
|
|
except ExtractionError as e:
|
|
# Or extraction error from invalid PDF
|
|
assert e.error_code in ["invalid_pdf", "corrupt_pdf"]
|
|
|
|
|
|
class TestPDFTextExtraction:
|
|
"""Test PDF text extraction."""
|
|
|
|
def test_pdf_text_extraction(self):
|
|
"""Test PDF text is extracted correctly."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/EN16931_Einfach.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.pdf_text is not None
|
|
assert len(result.pdf_text) > 0
|
|
# PDF text may contain invoice-related terms in German or English
|
|
|
|
|
|
class TestExtractionMeta:
|
|
"""Test extraction metadata."""
|
|
|
|
def test_extraction_meta_populated(self):
|
|
"""Test extraction metadata is populated correctly."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/EN16931_Einfach.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.extraction_meta is not None
|
|
assert result.extraction_meta.pages >= 1
|
|
assert result.extraction_meta.extraction_time_ms >= 0
|
|
|
|
def test_extraction_meta_non_zugferd(self):
|
|
"""Test extraction metadata for non-ZUGFeRD PDF."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/EmptyPDFA1.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.extraction_meta is not None
|
|
assert result.extraction_meta.pages >= 1
|
|
assert result.extraction_meta.extraction_time_ms >= 0
|
|
assert result.extraction_meta.xml_attachment_name is None
|
|
|
|
|
|
class TestExtendedProfile:
|
|
"""Test extraction from EXTENDED profile PDF (if available)."""
|
|
|
|
def test_extract_extended_profile(self):
|
|
"""Test EXTENDED PDF extraction detects correct profile."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/zugferd_2p1_EXTENDED_PDFA-3A.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.is_zugferd is True
|
|
assert result.zugferd_profil == "EXTENDED"
|
|
assert result.xml_data is not None
|
|
|
|
|
|
class TestZUGFeRDProfileVariations:
|
|
"""Test various ZUGFeRD profile detection."""
|
|
|
|
def test_detect_basicwl_profile(self):
|
|
"""Test BASIC WL profile detection."""
|
|
from src.extractor import extract_zugferd
|
|
|
|
with open("tests/fixtures/validAvoir_FR_type380_BASICWL.pdf", "rb") as f:
|
|
pdf_bytes = f.read()
|
|
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.is_zugferd is True
|
|
# Profile should be detected (BASIC, BASICWL, etc.)
|
|
assert result.zugferd_profil is not None
|
|
assert result.xml_data is not None
|
|
|
|
|
|
class TestXRechnungExtraction:
|
|
"""Test extraction from XRechnung PDFs with non-standard filenames."""
|
|
|
|
def test_xrechnung_by_xml_extension_fallback(self):
|
|
"""Test XRechnung with 'xrechnung.xml' filename is extracted correctly.
|
|
|
|
Depending on the factur-x version, the library recognizes non-standard
|
|
attachment names itself or our fallback kicks in — both paths must
|
|
yield the same result, so we only assert our own behavior here.
|
|
"""
|
|
from io import BytesIO
|
|
|
|
from src.extractor import extract_zugferd, _find_xml_attachment_fallback
|
|
|
|
# Load valid XML from existing test fixture
|
|
with open("tests/fixtures/validXRechnung.pdf", "rb") as f:
|
|
orig_reader = PdfReader(f)
|
|
xml_content = None
|
|
for att in orig_reader.attachment_list:
|
|
xml_content = att.content
|
|
break
|
|
if xml_content is None:
|
|
pytest.fail("Test fixture has no XML attachment")
|
|
|
|
# Create a new PDF with a non-standard attachment name
|
|
pdf_writer = PdfWriter()
|
|
pdf_writer.add_blank_page(width=72, height=72)
|
|
pdf_writer.add_attachment("xrechnung.xml", xml_content)
|
|
|
|
output = BytesIO()
|
|
pdf_writer.write(output)
|
|
pdf_bytes = output.getvalue()
|
|
|
|
# Verify our fallback function finds it on its own
|
|
fallback_result = _find_xml_attachment_fallback(pdf_bytes)
|
|
assert fallback_result[0] == "xrechnung.xml", (
|
|
"Fallback should find xrechnung.xml"
|
|
)
|
|
assert fallback_result[1] is not None, "Fallback should return XML content"
|
|
|
|
# Verify full extraction works, whichever path finds the XML
|
|
result = extract_zugferd(pdf_bytes)
|
|
|
|
assert result.is_zugferd is True
|
|
assert result.xml_raw is not None
|
|
assert result.xml_data is not None
|
|
assert result.xml_data.invoice_number is not None
|
|
assert result.extraction_meta.xml_attachment_name == "xrechnung.xml"
|
|
|
|
|
|
TAX_REGISTRATION_XML_TEMPLATE = """<rsm:CrossIndustryInvoice
|
|
xmlns:rsm="urn:un:unece:uncefact:data:standard:CrossIndustryInvoice:100"
|
|
xmlns:ram="urn:un:unece:uncefact:data:standard:ReusableAggregateBusinessInformationEntity:100">
|
|
<rsm:SupplyChainTradeTransaction>
|
|
<ram:ApplicableHeaderTradeAgreement>
|
|
<ram:SellerTradeParty>
|
|
<ram:Name>Test Seller GmbH</ram:Name>
|
|
{seller_tax_registrations}
|
|
{seller_contact}
|
|
</ram:SellerTradeParty>
|
|
<ram:BuyerTradeParty>
|
|
<ram:Name>Test Buyer AG</ram:Name>
|
|
{buyer_tax_registrations}
|
|
</ram:BuyerTradeParty>
|
|
</ram:ApplicableHeaderTradeAgreement>
|
|
</rsm:SupplyChainTradeTransaction>
|
|
</rsm:CrossIndustryInvoice>"""
|
|
|
|
VA_REGISTRATION = (
|
|
"<ram:SpecifiedTaxRegistration>"
|
|
'<ram:ID schemeID="VA">DE123456789</ram:ID>'
|
|
"</ram:SpecifiedTaxRegistration>"
|
|
)
|
|
FC_REGISTRATION = (
|
|
"<ram:SpecifiedTaxRegistration>"
|
|
'<ram:ID schemeID="FC">2281410000697</ram:ID>'
|
|
"</ram:SpecifiedTaxRegistration>"
|
|
)
|
|
BUYER_FC_REGISTRATION = (
|
|
"<ram:SpecifiedTaxRegistration>"
|
|
'<ram:ID schemeID="FC">9876543210</ram:ID>'
|
|
"</ram:SpecifiedTaxRegistration>"
|
|
)
|
|
CONTACT_WITH_EMAIL = (
|
|
"<ram:DefinedTradeContact>"
|
|
"<ram:EmailURIUniversalCommunication>"
|
|
"<ram:URIID>seller@example.de</ram:URIID>"
|
|
"</ram:EmailURIUniversalCommunication>"
|
|
"</ram:DefinedTradeContact>"
|
|
)
|
|
|
|
|
|
class TestTaxRegistrationExtraction:
|
|
"""Test extraction of tax registrations (VAT ID vs. tax number)."""
|
|
|
|
def _parse(self, seller_regs: str, buyer_regs: str, seller_contact: str = ""):
|
|
from lxml import etree
|
|
|
|
from src.extractor import parse_buyer, parse_supplier
|
|
|
|
xml = TAX_REGISTRATION_XML_TEMPLATE.format(
|
|
seller_tax_registrations=seller_regs,
|
|
buyer_tax_registrations=buyer_regs,
|
|
seller_contact=seller_contact,
|
|
)
|
|
root = etree.fromstring(xml.encode("utf-8"))
|
|
return parse_supplier(root), parse_buyer(root)
|
|
|
|
def test_seller_vat_id_and_tax_number(self):
|
|
"""Seller with VA and FC registrations extracts both fields."""
|
|
supplier, _ = self._parse(VA_REGISTRATION + FC_REGISTRATION, "")
|
|
|
|
assert supplier.vat_id == "DE123456789"
|
|
assert supplier.tax_number == "2281410000697"
|
|
|
|
def test_seller_tax_number_only(self):
|
|
"""Seller with only FC registration (e.g. Kleinunternehmer/Werkstatt)."""
|
|
supplier, _ = self._parse(FC_REGISTRATION, "")
|
|
|
|
assert supplier.vat_id is None
|
|
assert supplier.tax_number == "2281410000697"
|
|
|
|
def test_seller_vat_id_only(self):
|
|
"""Seller with only VA registration leaves tax_number empty."""
|
|
supplier, _ = self._parse(VA_REGISTRATION, "")
|
|
|
|
assert supplier.vat_id == "DE123456789"
|
|
assert supplier.tax_number is None
|
|
|
|
def test_buyer_tax_number(self):
|
|
"""Buyer with FC registration extracts tax_number."""
|
|
_, buyer = self._parse(VA_REGISTRATION, BUYER_FC_REGISTRATION)
|
|
|
|
assert buyer.vat_id is None
|
|
assert buyer.tax_number == "9876543210"
|
|
|
|
def test_seller_email_from_trade_contact(self):
|
|
"""Seller email is found in DefinedTradeContact (BT-42)."""
|
|
supplier, _ = self._parse(
|
|
VA_REGISTRATION, "", seller_contact=CONTACT_WITH_EMAIL
|
|
)
|
|
|
|
assert supplier.email == "seller@example.de"
|
|
|
|
def test_seller_email_absent(self):
|
|
"""No email elements leaves email as None."""
|
|
supplier, _ = self._parse(VA_REGISTRATION, "")
|
|
|
|
assert supplier.email is None
|
|
|
|
|
|
LINE_ITEM_XML_TEMPLATE = """<rsm:CrossIndustryInvoice
|
|
xmlns:rsm="urn:un:unece:uncefact:data:standard:CrossIndustryInvoice:100"
|
|
xmlns:ram="urn:un:unece:uncefact:data:standard:ReusableAggregateBusinessInformationEntity:100">
|
|
<rsm:SupplyChainTradeTransaction>
|
|
<ram:IncludedSupplyChainTradeLineItem>
|
|
<ram:SpecifiedTradeProduct>
|
|
<ram:Name>Flanschsicherungsset</ram:Name>
|
|
</ram:SpecifiedTradeProduct>
|
|
<ram:SpecifiedLineTradeAgreement>
|
|
<ram:NetPriceProductTradePrice>
|
|
<ram:ChargeAmount>{charge_amount}</ram:ChargeAmount>
|
|
{basis_quantity}
|
|
</ram:NetPriceProductTradePrice>
|
|
</ram:SpecifiedLineTradeAgreement>
|
|
<ram:SpecifiedLineTradeDelivery>
|
|
<ram:BilledQuantity unitCode="H87">{billed_quantity}</ram:BilledQuantity>
|
|
</ram:SpecifiedLineTradeDelivery>
|
|
<ram:SpecifiedLineTradeSettlement>
|
|
<ram:SpecifiedTradeSettlementLineMonetarySummation>
|
|
<ram:LineTotalAmount>{line_total}</ram:LineTotalAmount>
|
|
</ram:SpecifiedTradeSettlementLineMonetarySummation>
|
|
</ram:SpecifiedLineTradeSettlement>
|
|
</ram:IncludedSupplyChainTradeLineItem>
|
|
</rsm:SupplyChainTradeTransaction>
|
|
</rsm:CrossIndustryInvoice>"""
|
|
|
|
|
|
class TestBasisQuantityExtraction:
|
|
"""Test unit price normalization via BasisQuantity (BT-149)."""
|
|
|
|
def _parse_items(self, basis_quantity: str):
|
|
from lxml import etree
|
|
|
|
from src.extractor import parse_line_items
|
|
|
|
xml = LINE_ITEM_XML_TEMPLATE.format(
|
|
charge_amount="75.85",
|
|
basis_quantity=basis_quantity,
|
|
billed_quantity="140",
|
|
line_total="106.19",
|
|
)
|
|
root = etree.fromstring(xml.encode("utf-8"))
|
|
return parse_line_items(root)
|
|
|
|
def test_unit_price_divided_by_basis_quantity(self):
|
|
"""Charge amount per 100 units is normalized to per-unit price."""
|
|
items = self._parse_items("<ram:BasisQuantity>100</ram:BasisQuantity>")
|
|
|
|
assert len(items) == 1
|
|
assert items[0].unit_price == pytest.approx(0.7585)
|
|
|
|
def test_unit_price_without_basis_quantity(self):
|
|
"""Without BasisQuantity the charge amount stays as-is."""
|
|
items = self._parse_items("")
|
|
|
|
assert len(items) == 1
|
|
assert items[0].unit_price == pytest.approx(75.85)
|
|
|
|
def test_unit_price_with_zero_basis_quantity(self):
|
|
"""Zero BasisQuantity is ignored (division guard)."""
|
|
items = self._parse_items("<ram:BasisQuantity>0</ram:BasisQuantity>")
|
|
|
|
assert len(items) == 1
|
|
assert items[0].unit_price == pytest.approx(75.85)
|