feat(import): Abgabeverträge (DOCX) auswerten und Tiere/Kontakte anreichern
Neuer Parser extract_contracts.py liest die ~1,4k Abgabevertrags-DOCX (\truenas\…\Verträge): er extrahiert aus dem Dokument-Body (zuverlässiger als die Dateinamen) Käufer, Tier(e), Farbschlag, Abgabedatum und Preis — robust gegen Word-Run-Splits (z. B. „F r au"/„3 0,00"); überspringt Vorlage, Abstammungsnachweise und als .docx getarnte .doc. enrich_from_contracts() in merge_and_resolve.py: Käufer werden als Kontakte (IsReceiver) angelegt/zusammengeführt; Tiere werden KONSERVATIV per Rufname (+ DOB-Jahr bei Mehrdeutigkeit) auf eigene Bestandstiere gematcht und erhalten ReceiverContactId, GoHomeDate und Status „abgegeben" — nur wo nicht bereits gesetzt; Konflikte werden geloggt, nicht überschrieben. Jede Übernahme bekommt eine Herkunfts-Zeile („Abgabe an … aus Vertrag … übernommen."). Ergebnis: 1095 Verträge → 783 Tier-Treffer (400 mehrdeutige übersprungen), 274 neue Abnehmer-Kontakte, 153 Tiere mit Abnehmer, 49 mit Abgabedatum, 23 neu „abgegeben". Keine Backend-/Frontend-Änderung nötig (Akte zeigt Abnehmer/ Abgabedatum/Herkunft bereits). SaleContract-Records bewusst nicht erzeugt (bräuchte Migration + ingest-sichere Id — späterer Schritt). Tests: test_extract_contracts.py (Dateiname/Body/Run-Split/Skip-Regeln) + alle bestehenden grün; dotnet 212. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
189
tools/import/test_extract_contracts.py
Normal file
189
tools/import/test_extract_contracts.py
Normal file
@@ -0,0 +1,189 @@
|
||||
"""Tests for extract_contracts.py — run: python test_extract_contracts.py
|
||||
|
||||
Zero third-party deps (mirrors test_extract_docx.py). Builds tiny in-memory
|
||||
.docx files (a zip with word/document.xml) so the tests run without the
|
||||
network share. Covers: filename parsing, run-splitting de-mangling, both body
|
||||
layouts (side-by-side / stacked), price/date normalisation, and skip rules.
|
||||
"""
|
||||
import io
|
||||
import os
|
||||
import sys
|
||||
import zipfile
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import extract_contracts as ec
|
||||
|
||||
failed = 0
|
||||
|
||||
|
||||
def check(name, cond):
|
||||
global failed
|
||||
print(("ok: " if cond else "FAIL: ") + name)
|
||||
if not cond:
|
||||
failed += 1
|
||||
|
||||
|
||||
def make_docx(paragraphs):
|
||||
"""paragraphs: list of lists of run-strings → bytes of a .docx zip."""
|
||||
body = []
|
||||
for runs in paragraphs:
|
||||
rs = "".join(f"<w:r><w:t>{r}</w:t></w:r>" for r in runs)
|
||||
body.append(f"<w:p>{rs}</w:p>")
|
||||
xml = ('<?xml version="1.0"?><w:document xmlns:w="x"><w:body>'
|
||||
+ "".join(body) + "</w:body></w:document>")
|
||||
buf = io.BytesIO()
|
||||
with zipfile.ZipFile(buf, "w") as z:
|
||||
z.writestr("word/document.xml", xml)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def write_tmp(name, data):
|
||||
d = os.path.join(os.path.dirname(os.path.abspath(__file__)), "_test_tmp")
|
||||
os.makedirs(d, exist_ok=True)
|
||||
p = os.path.join(d, name)
|
||||
with open(p, "wb") as f:
|
||||
f.write(data)
|
||||
return p
|
||||
|
||||
|
||||
# --- _norm_date -----------------------------------------------------------
|
||||
check("date 4-digit year", ec._norm_date("18.02.2014") == "2014-02-18")
|
||||
check("date 2-digit year", ec._norm_date("09.12.14") == "2014-12-09")
|
||||
check("date embedded", ec._norm_date("Bonn, 20.09.2014") == "2014-09-20")
|
||||
check("date invalid", ec._norm_date("99.99.9999") == "")
|
||||
check("date empty", ec._norm_date("") == "")
|
||||
|
||||
# --- _clean_money ---------------------------------------------------------
|
||||
check("money comma", ec._clean_money("27,50 Euro (Überweisung)") == "27,50")
|
||||
check("money dot→comma", ec._clean_money("25.00 €") == "25,00")
|
||||
check("money plain", ec._clean_money("28 Euro") == "28")
|
||||
check("money none", ec._clean_money("kostenlos") == "")
|
||||
|
||||
# --- parse_filename -------------------------------------------------------
|
||||
fn = ec.parse_filename(
|
||||
"Zucht der kleinen Chaoten _ Agouti (Kathlin.Baxter) - Stephan Füchsle_.docx")
|
||||
check("fn color", fn["color"] == "Agouti")
|
||||
check("fn animals", fn["animals"] == ["Kathlin", "Baxter"])
|
||||
check("fn buyer", fn["buyer"] == "Stephan Füchsle")
|
||||
|
||||
fn2 = ec.parse_filename(
|
||||
"Zucht der kleinen Chaoten _ dd Polar Sp (Velvet.Vance Jr.)- Alexandra Wendler_.docx")
|
||||
check("fn multiword color", fn2["color"] == "dd Polar Sp")
|
||||
check("fn multiword animal", fn2["animals"] == ["Velvet", "Vance Jr"])
|
||||
check("fn buyer 2", fn2["buyer"] == "Alexandra Wendler")
|
||||
|
||||
fn3 = ec.parse_filename("Zucht der kleinen Chaoten _ Ethan - Stefanie Stoica_.docx")
|
||||
check("fn no-paren animal", fn3["animals"] == ["Ethan"])
|
||||
check("fn no-paren buyer", fn3["buyer"] == "Stefanie Stoica")
|
||||
|
||||
fn4 = ec.parse_filename("Zucht der kleinen Chaoten _Azrael_.docx")
|
||||
check("fn name-only animal", fn4["animals"] == ["Azrael"])
|
||||
check("fn name-only no buyer", fn4["buyer"] == "")
|
||||
|
||||
# --- run-splitting de-mangling (the core text-extraction fix) -------------
|
||||
data = make_docx([["Name:", "F", "r", "au Josephin Kiefer"]])
|
||||
p = write_tmp("split.docx", data)
|
||||
check("run-split joins to 'Frau Josephin Kiefer'",
|
||||
ec._full_text(p) == "Name:Frau Josephin Kiefer")
|
||||
|
||||
money = make_docx([["Schutzgebühr: ", "3", "0,00 €"]])
|
||||
p = write_tmp("money.docx", money)
|
||||
check("run-split price intact",
|
||||
ec._clean_money(ec._label_value(ec._full_text(p), r"Schutzgeb\w+")) == "30,00")
|
||||
|
||||
# --- full contract: stacked layout (B) ------------------------------------
|
||||
stacked = make_docx([
|
||||
["Vermittlungsvertrag"],
|
||||
["Abgebender/Züchter:"],
|
||||
["Name:", "Frau Drazena Rimac"],
|
||||
["Straße:", "New-York-Str. 30"],
|
||||
["Empfänger/Abnehmer:"],
|
||||
["Name:", "Familie Tanja und Thorsten Kurz"],
|
||||
["Straße:", "Ebelstraße 2"],
|
||||
["Tierdaten:"],
|
||||
["Name:", "Einstein"],
|
||||
["Geburtsdatum:", "10.08.2019"],
|
||||
["Geschlecht:", "männlich"],
|
||||
["Farbschlag: ", "Schwarz"],
|
||||
["Abgabedatum:", "31.12.2019"],
|
||||
["Schutzgebühr: ", "25,00 €"],
|
||||
["Butzbach, den 31.12.2019"],
|
||||
])
|
||||
p = write_tmp("stacked.docx",
|
||||
make_docx_b := stacked)
|
||||
rec = ec.parse_contract(p)
|
||||
check("stacked buyer", rec["buyer"] == "Familie Tanja und Thorsten Kurz")
|
||||
check("stacked not seller", rec["buyer"] != "Frau Drazena Rimac")
|
||||
check("stacked animal name", rec["animalNameBody"] == "Einstein")
|
||||
check("stacked dob", rec["dob"] == "2019-08-10")
|
||||
check("stacked gender", rec["gender"] == "Male")
|
||||
check("stacked color", rec["color"] == "Schwarz")
|
||||
check("stacked handover", rec["handoverDate"] == "2019-12-31")
|
||||
check("stacked price", rec["price"] == "25,00")
|
||||
check("stacked contract date", rec["contractDate"] == "2019-12-31")
|
||||
|
||||
# --- full contract: side-by-side columns (A) ------------------------------
|
||||
# Two columns render as separate paragraphs (table cells).
|
||||
sidebyside = make_docx([
|
||||
["Vermittlungsvertrag"],
|
||||
["Züchter:"], ["Abnehmer:"],
|
||||
["Name:", "Drazena Rimac"], ["Name:", "Andrea Thesing"],
|
||||
["Straße:", "Alten-Busecker-Str. 57"], ["Straße:", "Veilchenweg 32"],
|
||||
["Tierdaten:"],
|
||||
["Name:", "Sally"],
|
||||
["Geburtsdatum:", "03.02.2014"],
|
||||
["Geschlecht:", "weiblich"],
|
||||
["Farbschlag:", "Kohlfuchs-hell"],
|
||||
["Abgabedatum:", "20.09.2014"],
|
||||
["Gesamtpreis: ", "27,00 Euro (Überweisung)"],
|
||||
["Bonn, 20.09.2014"],
|
||||
])
|
||||
p = write_tmp("sidebyside.docx", sidebyside)
|
||||
rec = ec.parse_contract(p)
|
||||
check("sbs buyer is Andrea Thesing", rec["buyer"] == "Andrea Thesing")
|
||||
check("sbs buyer not seller", rec["buyer"] != "Drazena Rimac")
|
||||
check("sbs animal", rec["animalNameBody"] == "Sally")
|
||||
check("sbs gender female", rec["gender"] == "Female")
|
||||
check("sbs price", rec["price"] == "27,00")
|
||||
|
||||
# --- skip rules -----------------------------------------------------------
|
||||
abstammung = make_docx([
|
||||
["Abstammungsnachweis"],
|
||||
["Name:"], ["Grace"],
|
||||
["Geburtsdatum:", "04.10.2015"],
|
||||
])
|
||||
p = write_tmp("abst.docx", abstammung)
|
||||
check("Abstammungsnachweis skipped", ec.parse_contract(p) is None)
|
||||
|
||||
# bad zip
|
||||
badp = write_tmp("bad.docx", b"not a zip")
|
||||
check("bad zip → None", ec.parse_contract(badp) is None)
|
||||
|
||||
# --- filename color fallback when body has no Farbschlag ------------------
|
||||
nocolor = make_docx([
|
||||
["Vermittlungsvertrag"], ["Abnehmer:"], ["Name:", "Max Mustermann"],
|
||||
["Tierdaten:"], ["Name:", "Bello"],
|
||||
["Geburtsdatum:", "01.01.2020"], ["Geschlecht:", "männlich"],
|
||||
["Gesamtpreis: ", "20,00 €"],
|
||||
])
|
||||
p = write_tmp("Zucht der kleinen Chaoten _ Agouti (Bello)- Max Mustermann_.docx",
|
||||
nocolor)
|
||||
rec = ec.parse_contract(p)
|
||||
check("color falls back to filename", rec["color"] == "Agouti")
|
||||
check("animals from filename", rec["animals"] == ["Bello"])
|
||||
|
||||
# cleanup tmp
|
||||
import shutil
|
||||
shutil.rmtree(os.path.join(os.path.dirname(os.path.abspath(__file__)), "_test_tmp"),
|
||||
ignore_errors=True)
|
||||
|
||||
print()
|
||||
if failed:
|
||||
print(f"{failed} test(s) FAILED")
|
||||
sys.exit(1)
|
||||
print("All extract_contracts tests passed.")
|
||||
Reference in New Issue
Block a user