Neuer Parser extract_contracts.py liest die ~1,4k Abgabevertrags-DOCX (\truenas\…\Verträge): er extrahiert aus dem Dokument-Body (zuverlässiger als die Dateinamen) Käufer, Tier(e), Farbschlag, Abgabedatum und Preis — robust gegen Word-Run-Splits (z. B. „F r au"/„3 0,00"); überspringt Vorlage, Abstammungsnachweise und als .docx getarnte .doc. enrich_from_contracts() in merge_and_resolve.py: Käufer werden als Kontakte (IsReceiver) angelegt/zusammengeführt; Tiere werden KONSERVATIV per Rufname (+ DOB-Jahr bei Mehrdeutigkeit) auf eigene Bestandstiere gematcht und erhalten ReceiverContactId, GoHomeDate und Status „abgegeben" — nur wo nicht bereits gesetzt; Konflikte werden geloggt, nicht überschrieben. Jede Übernahme bekommt eine Herkunfts-Zeile („Abgabe an … aus Vertrag … übernommen."). Ergebnis: 1095 Verträge → 783 Tier-Treffer (400 mehrdeutige übersprungen), 274 neue Abnehmer-Kontakte, 153 Tiere mit Abnehmer, 49 mit Abgabedatum, 23 neu „abgegeben". Keine Backend-/Frontend-Änderung nötig (Akte zeigt Abnehmer/ Abgabedatum/Herkunft bereits). SaleContract-Records bewusst nicht erzeugt (bräuchte Migration + ingest-sichere Id — späterer Schritt). Tests: test_extract_contracts.py (Dateiname/Body/Run-Split/Skip-Regeln) + alle bestehenden grün; dotnet 212. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
190 lines
7.1 KiB
Python
190 lines
7.1 KiB
Python
"""Tests for extract_contracts.py — run: python test_extract_contracts.py
|
|
|
|
Zero third-party deps (mirrors test_extract_docx.py). Builds tiny in-memory
|
|
.docx files (a zip with word/document.xml) so the tests run without the
|
|
network share. Covers: filename parsing, run-splitting de-mangling, both body
|
|
layouts (side-by-side / stacked), price/date normalisation, and skip rules.
|
|
"""
|
|
import io
|
|
import os
|
|
import sys
|
|
import zipfile
|
|
|
|
try:
|
|
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
except Exception:
|
|
pass
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import extract_contracts as ec
|
|
|
|
failed = 0
|
|
|
|
|
|
def check(name, cond):
|
|
global failed
|
|
print(("ok: " if cond else "FAIL: ") + name)
|
|
if not cond:
|
|
failed += 1
|
|
|
|
|
|
def make_docx(paragraphs):
|
|
"""paragraphs: list of lists of run-strings → bytes of a .docx zip."""
|
|
body = []
|
|
for runs in paragraphs:
|
|
rs = "".join(f"<w:r><w:t>{r}</w:t></w:r>" for r in runs)
|
|
body.append(f"<w:p>{rs}</w:p>")
|
|
xml = ('<?xml version="1.0"?><w:document xmlns:w="x"><w:body>'
|
|
+ "".join(body) + "</w:body></w:document>")
|
|
buf = io.BytesIO()
|
|
with zipfile.ZipFile(buf, "w") as z:
|
|
z.writestr("word/document.xml", xml)
|
|
return buf.getvalue()
|
|
|
|
|
|
def write_tmp(name, data):
|
|
d = os.path.join(os.path.dirname(os.path.abspath(__file__)), "_test_tmp")
|
|
os.makedirs(d, exist_ok=True)
|
|
p = os.path.join(d, name)
|
|
with open(p, "wb") as f:
|
|
f.write(data)
|
|
return p
|
|
|
|
|
|
# --- _norm_date -----------------------------------------------------------
|
|
check("date 4-digit year", ec._norm_date("18.02.2014") == "2014-02-18")
|
|
check("date 2-digit year", ec._norm_date("09.12.14") == "2014-12-09")
|
|
check("date embedded", ec._norm_date("Bonn, 20.09.2014") == "2014-09-20")
|
|
check("date invalid", ec._norm_date("99.99.9999") == "")
|
|
check("date empty", ec._norm_date("") == "")
|
|
|
|
# --- _clean_money ---------------------------------------------------------
|
|
check("money comma", ec._clean_money("27,50 Euro (Überweisung)") == "27,50")
|
|
check("money dot→comma", ec._clean_money("25.00 €") == "25,00")
|
|
check("money plain", ec._clean_money("28 Euro") == "28")
|
|
check("money none", ec._clean_money("kostenlos") == "")
|
|
|
|
# --- parse_filename -------------------------------------------------------
|
|
fn = ec.parse_filename(
|
|
"Zucht der kleinen Chaoten _ Agouti (Kathlin.Baxter) - Stephan Füchsle_.docx")
|
|
check("fn color", fn["color"] == "Agouti")
|
|
check("fn animals", fn["animals"] == ["Kathlin", "Baxter"])
|
|
check("fn buyer", fn["buyer"] == "Stephan Füchsle")
|
|
|
|
fn2 = ec.parse_filename(
|
|
"Zucht der kleinen Chaoten _ dd Polar Sp (Velvet.Vance Jr.)- Alexandra Wendler_.docx")
|
|
check("fn multiword color", fn2["color"] == "dd Polar Sp")
|
|
check("fn multiword animal", fn2["animals"] == ["Velvet", "Vance Jr"])
|
|
check("fn buyer 2", fn2["buyer"] == "Alexandra Wendler")
|
|
|
|
fn3 = ec.parse_filename("Zucht der kleinen Chaoten _ Ethan - Stefanie Stoica_.docx")
|
|
check("fn no-paren animal", fn3["animals"] == ["Ethan"])
|
|
check("fn no-paren buyer", fn3["buyer"] == "Stefanie Stoica")
|
|
|
|
fn4 = ec.parse_filename("Zucht der kleinen Chaoten _Azrael_.docx")
|
|
check("fn name-only animal", fn4["animals"] == ["Azrael"])
|
|
check("fn name-only no buyer", fn4["buyer"] == "")
|
|
|
|
# --- run-splitting de-mangling (the core text-extraction fix) -------------
|
|
data = make_docx([["Name:", "F", "r", "au Josephin Kiefer"]])
|
|
p = write_tmp("split.docx", data)
|
|
check("run-split joins to 'Frau Josephin Kiefer'",
|
|
ec._full_text(p) == "Name:Frau Josephin Kiefer")
|
|
|
|
money = make_docx([["Schutzgebühr: ", "3", "0,00 €"]])
|
|
p = write_tmp("money.docx", money)
|
|
check("run-split price intact",
|
|
ec._clean_money(ec._label_value(ec._full_text(p), r"Schutzgeb\w+")) == "30,00")
|
|
|
|
# --- full contract: stacked layout (B) ------------------------------------
|
|
stacked = make_docx([
|
|
["Vermittlungsvertrag"],
|
|
["Abgebender/Züchter:"],
|
|
["Name:", "Frau Drazena Rimac"],
|
|
["Straße:", "New-York-Str. 30"],
|
|
["Empfänger/Abnehmer:"],
|
|
["Name:", "Familie Tanja und Thorsten Kurz"],
|
|
["Straße:", "Ebelstraße 2"],
|
|
["Tierdaten:"],
|
|
["Name:", "Einstein"],
|
|
["Geburtsdatum:", "10.08.2019"],
|
|
["Geschlecht:", "männlich"],
|
|
["Farbschlag: ", "Schwarz"],
|
|
["Abgabedatum:", "31.12.2019"],
|
|
["Schutzgebühr: ", "25,00 €"],
|
|
["Butzbach, den 31.12.2019"],
|
|
])
|
|
p = write_tmp("stacked.docx",
|
|
make_docx_b := stacked)
|
|
rec = ec.parse_contract(p)
|
|
check("stacked buyer", rec["buyer"] == "Familie Tanja und Thorsten Kurz")
|
|
check("stacked not seller", rec["buyer"] != "Frau Drazena Rimac")
|
|
check("stacked animal name", rec["animalNameBody"] == "Einstein")
|
|
check("stacked dob", rec["dob"] == "2019-08-10")
|
|
check("stacked gender", rec["gender"] == "Male")
|
|
check("stacked color", rec["color"] == "Schwarz")
|
|
check("stacked handover", rec["handoverDate"] == "2019-12-31")
|
|
check("stacked price", rec["price"] == "25,00")
|
|
check("stacked contract date", rec["contractDate"] == "2019-12-31")
|
|
|
|
# --- full contract: side-by-side columns (A) ------------------------------
|
|
# Two columns render as separate paragraphs (table cells).
|
|
sidebyside = make_docx([
|
|
["Vermittlungsvertrag"],
|
|
["Züchter:"], ["Abnehmer:"],
|
|
["Name:", "Drazena Rimac"], ["Name:", "Andrea Thesing"],
|
|
["Straße:", "Alten-Busecker-Str. 57"], ["Straße:", "Veilchenweg 32"],
|
|
["Tierdaten:"],
|
|
["Name:", "Sally"],
|
|
["Geburtsdatum:", "03.02.2014"],
|
|
["Geschlecht:", "weiblich"],
|
|
["Farbschlag:", "Kohlfuchs-hell"],
|
|
["Abgabedatum:", "20.09.2014"],
|
|
["Gesamtpreis: ", "27,00 Euro (Überweisung)"],
|
|
["Bonn, 20.09.2014"],
|
|
])
|
|
p = write_tmp("sidebyside.docx", sidebyside)
|
|
rec = ec.parse_contract(p)
|
|
check("sbs buyer is Andrea Thesing", rec["buyer"] == "Andrea Thesing")
|
|
check("sbs buyer not seller", rec["buyer"] != "Drazena Rimac")
|
|
check("sbs animal", rec["animalNameBody"] == "Sally")
|
|
check("sbs gender female", rec["gender"] == "Female")
|
|
check("sbs price", rec["price"] == "27,00")
|
|
|
|
# --- skip rules -----------------------------------------------------------
|
|
abstammung = make_docx([
|
|
["Abstammungsnachweis"],
|
|
["Name:"], ["Grace"],
|
|
["Geburtsdatum:", "04.10.2015"],
|
|
])
|
|
p = write_tmp("abst.docx", abstammung)
|
|
check("Abstammungsnachweis skipped", ec.parse_contract(p) is None)
|
|
|
|
# bad zip
|
|
badp = write_tmp("bad.docx", b"not a zip")
|
|
check("bad zip → None", ec.parse_contract(badp) is None)
|
|
|
|
# --- filename color fallback when body has no Farbschlag ------------------
|
|
nocolor = make_docx([
|
|
["Vermittlungsvertrag"], ["Abnehmer:"], ["Name:", "Max Mustermann"],
|
|
["Tierdaten:"], ["Name:", "Bello"],
|
|
["Geburtsdatum:", "01.01.2020"], ["Geschlecht:", "männlich"],
|
|
["Gesamtpreis: ", "20,00 €"],
|
|
])
|
|
p = write_tmp("Zucht der kleinen Chaoten _ Agouti (Bello)- Max Mustermann_.docx",
|
|
nocolor)
|
|
rec = ec.parse_contract(p)
|
|
check("color falls back to filename", rec["color"] == "Agouti")
|
|
check("animals from filename", rec["animals"] == ["Bello"])
|
|
|
|
# cleanup tmp
|
|
import shutil
|
|
shutil.rmtree(os.path.join(os.path.dirname(os.path.abspath(__file__)), "_test_tmp"),
|
|
ignore_errors=True)
|
|
|
|
print()
|
|
if failed:
|
|
print(f"{failed} test(s) FAILED")
|
|
sys.exit(1)
|
|
print("All extract_contracts tests passed.")
|