Merge feature/feat-8d-docx (FEAT-8d Phase 2): Wurfchronik-docx-Importer (Abnehmer/GoHomeDate/Tod)
Some checks failed
CI / Backend Tests (.NET) (push) Successful in 58s
CI / Frontend Tests (Node/Vite) (push) Successful in 9m32s
CI / Docker Build & Push (push) Failing after 9s

extract_docx.py (stdlib, 93 Wuerfe/227 Tiere) + ImportDocxService (fill-NULL-only: WS-Code→LitterId,
Abnehmer→ReceiverContactId, ABD→GoHomeDate, Tod→DateOfDeath+Ursache) + /import/docx/dry-run|execute.
157/157 C# + 28/28 python, has-pending=No. Dormant bis execute (NACHZUG nach Re-Import).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-06-06 22:05:35 +02:00
5 changed files with 677 additions and 0 deletions

View File

@@ -0,0 +1,312 @@
#!/usr/bin/env python3
"""FEAT-8d Stage 1 — Wurfchronik-Detail-Dokument (.docx) extrahieren.
Liest 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML, stdlib-Python,
kein pip) und erzeugt:
output/docx_litters.json — Wurf-Kopfdaten (WS-Code, DOB, Eltern, Notiz)
output/docx_animals.json — Tier-Zeilen (Name, Farbe, Abnehmer, ABD, Tod)
Format der Ausgabe ist so gestaltet, dass ImportDocxService.cs in C# direkt
darüber laden kann. Idempotent: mehrfaches Ausführen überschreibt denselben Output.
Bekannte Sonderwerte im Dokument:
ZT = Zucht-Tier (bleibt in Zucht, kein externer Abnehmer)
BLEIBT = vorläufig beim Züchter
FREI = noch verfügbar
VG: = Verpaarungs-Geschichte (bisherige Partner; nicht als Abnehmer werten)
RG: = Rückgabe
BEW = Bewerbung (Adoptionsinteressent in Prüfung)
-- ??? = Platzhalter, kein echter Name
Feld 'gender': '' = weiblich (kein Marker), '*' auf Farbschlag oder 'G'-Spalte = männlich.
Ausführung: python extract_docx.py [--docx PFAD]
"""
import os
import re
import sys
import json
import zipfile
import argparse
HERE = os.path.dirname(os.path.abspath(__file__))
DEFAULT_DOCX = os.path.join(
r"C:\Users\gulum\dev",
"Wurfchronik der Kleinen Chaoten im Detail.docx",
)
OUT = os.path.join(HERE, "output")
# --- Regex patterns -------------------------------------------------------
# Litter header paragraph (after whitespace-collapsing).
# Edge cases handled:
# - Dual birth date: "*16./17.03.2021"
# - WS without numerator: "WS: /5"
# - WS with trailing text: "WS: 4/4, davon 1 später..."
# - No space before WS: "...ChaotenWS: 2/4"
# Date part allows simple DD.MM.YYYY, dual-day (16./17.03.2021), or dual-month (31.05/*01.06.2023).
# We capture the LAST complete DD.MM.YYYY in the date token as the birth date.
_DATE_TOKEN = r"[\d./\*]+"
# Full litter header regex
LITTER_RE = re.compile(
r"([A-Za-z\d\-]*Wurf)\s*\*\s*(" + _DATE_TOKEN + r")"
r"\s*Von:\s*(.+?)\s*&\s*(.+?)\s*WS:\s*(\d*\s*/\s*\d+)"
r"(?:[,\s].*?)?(?:Notiz:\s*(.*?))?$",
re.IGNORECASE,
)
# Used to extract the canonical date from a date token like "31.05/*01.06.2023"
_LAST_DATE_RE = re.compile(r"(\d{1,2}\.\d{2}\.\d{4})(?![\d.])")
# Death/adoption date at start of combined T.D column: "16.09.23Tumor am After"
DATE_START_RE = re.compile(r"^(\d{1,2}\.\d{1,2}\.\d{2,4})\s*(.*)")
# Partner birth date: "Crow (*25.12.20)" or "Tom (*05.01.21)"
PARTNER_DOB_RE = re.compile(r"\(\s*\*\s*(\d{2}\.\d{2}\.\d{2,4})\s*\)")
# Special-value sentinel names to skip
PLACEHOLDER_NAMES = {"--", "???", ""}
INTERNAL_TOKENS = {"ZT", "BLEIBT", "FREI", "VG:", "VG*:", "RG:", "BEW"}
def _norm_dob(d: str) -> str:
"""Normalise German date to DD.MM.YYYY."""
if not d:
return ""
p = d.strip().split(".")
if len(p) == 3:
y = p[2].strip()
if len(y) == 2:
y = "20" + y
return f"{p[0].zfill(2)}.{p[1].zfill(2)}.{y}"
return d.strip()
def _cell_text(cell_xml: str) -> str:
"""Strip XML from a <w:tc> cell and return clean text."""
t = re.sub(r"<[^>]+>", "", cell_xml)
t = t.replace("&amp;", "&").replace("&lt;", "<").replace("&gt;", ">")
t = t.replace("&apos;", "'").replace("&quot;", '"')
return re.sub(r"\s+", " ", t).strip()
def _is_internal(value: str) -> bool:
"""True if the owner/name field holds an internal sentinel, not a real person."""
v = value.strip()
return v in INTERNAL_TOKENS or any(v.startswith(tok) for tok in INTERNAL_TOKENS)
def extract(docx_path: str):
"""Parse the docx and return (litters, animals) lists."""
with zipfile.ZipFile(docx_path) as z:
xml = z.read("word/document.xml").decode("utf-8", errors="replace")
# ---- Paragraphs → litter header blocks ----
paras = re.findall(r"<w:p[ >].*?</w:p>", xml, re.DOTALL)
para_texts = []
for p in paras:
t = re.sub(r"<[^>]+>", "", p)
t = t.replace("&amp;", "&").strip()
t = re.sub(r"\s+", " ", t).strip()
if t:
para_texts.append(t)
litters: list[dict] = []
current_ws: str = ""
current_litter_dob: str = ""
# Build a WS-code → litter index for assigning animals
ws_to_idx: dict[str, int] = {}
for para in para_texts:
m = LITTER_RE.search(para)
if not m:
continue
litter_id = m.group(1).strip()
dob_raw = m.group(2).strip()
mother_raw = m.group(3).strip()
father_raw = m.group(4).strip()
ws_raw = m.group(5).replace(" ", "")
note = (m.group(6) or "").strip()
# For dual-date tokens like "31.05/*01.06.2023", take the last full date.
last_dates = _LAST_DATE_RE.findall(dob_raw)
dob_clean = _norm_dob(last_dates[-1] if last_dates else dob_raw)
litter = {
"litterId": litter_id,
"dob": dob_clean,
"motherName": mother_raw,
"fatherName": father_raw,
"wsCode": ws_raw,
"note": note,
}
ws_to_idx[ws_raw] = len(litters)
litters.append(litter)
# ---- Tables → animal rows ----
# Each table sits after a litter-header paragraph; we sequence tables and
# litter headers together by their byte offset in the XML.
animals: list[dict] = []
# Build ordered sequence of (offset, type, data) events
events: list[tuple[int, str, any]] = []
for m in re.finditer(r"<w:p[ >].*?</w:p>", xml, re.DOTALL):
t = re.sub(r"<[^>]+>", "", m.group()).replace("&amp;", "&").strip()
t = re.sub(r"\s+", " ", t).strip()
lm = LITTER_RE.search(t)
if lm:
ws = lm.group(5).replace(" ", "")
dob_tok = lm.group(2)
last = _LAST_DATE_RE.findall(dob_tok)
dob = _norm_dob(last[-1] if last else dob_tok)
events.append((m.start(), "litter", (ws, dob)))
for m in re.finditer(r"<w:tbl[ >].*?</w:tbl>", xml, re.DOTALL):
events.append((m.start(), "table", m.group()))
events.sort(key=lambda e: e[0])
active_ws = ""
active_dob = ""
for _, etype, edata in events:
if etype == "litter":
active_ws, active_dob = edata
elif etype == "table" and active_ws:
# Parse all rows in this table
rows = re.findall(r"<w:tr[ >].*?</w:tr>", edata, re.DOTALL)
for row in rows:
cells_xml = re.findall(r"<w:tc[ >].*?</w:tc>", row, re.DOTALL)
ct = [_cell_text(c) for c in cells_xml]
if not ct:
continue
# Skip header rows
if ct[0] == "G" and len(ct) > 1 and "Farbe" in ct[1]:
continue
# Column positions: G | Farbe | Name | Partner | Abnehmer | ABD | T.D
# Some newer tables add ABGew between ABD and T.D (7 or 8 cols)
g_col = ct[0] if len(ct) > 0 else ""
farbe_raw = ct[1] if len(ct) > 1 else ""
name = ct[2] if len(ct) > 2 else ""
partner = ct[3] if len(ct) > 3 else ""
owner = ct[4] if len(ct) > 4 else ""
abd_raw = ct[5] if len(ct) > 5 else ""
# If 8 cols, col 6 = ABGew, col 7 = T.D; if 7 cols, col 6 = T.D
if len(ct) >= 8:
abgew = ct[6]
tod_raw = ct[7]
elif len(ct) >= 7:
abgew = ""
tod_raw = ct[6]
else:
abgew = ""
tod_raw = ""
# Skip placeholders
name = name.strip()
if name in PLACEHOLDER_NAMES:
continue
if not farbe_raw.strip() and not name:
continue
# Gender: explicit marker in G column, or * suffix on Farbschlag
is_male = bool(g_col.strip() == "*" or farbe_raw.endswith("*"))
farbschlag = farbe_raw.rstrip("*").strip()
# Owner: strip internal sentinels
owner_clean = owner.strip()
if _is_internal(owner_clean):
owner_clean = ""
# For multi-owner ("1.) Julia2.) RG:"), take first
m1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", owner_clean)
if m1:
owner_clean = m1.group(1).strip()
# ABD (Abgabe-Datum)
abgabe_date = _norm_dob(abd_raw.strip())
# T.D column: may start with a date followed by cause
death_date = ""
death_cause = ""
if tod_raw:
dm = DATE_START_RE.match(tod_raw.strip())
if dm:
death_date = _norm_dob(dm.group(1))
death_cause = dm.group(2).strip()
else:
death_cause = tod_raw.strip()
# Partner name and DOB
partner_clean = partner.strip()
partner_dob = ""
pdob_m = PARTNER_DOB_RE.search(partner_clean)
if pdob_m:
partner_dob = _norm_dob(pdob_m.group(1))
partner_clean = PARTNER_DOB_RE.sub("", partner_clean).strip()
# Strip VG:/ZT/etc. prefixes
partner_clean = re.sub(r"^(?:VG\*?:|ZT\s*)", "", partner_clean).strip()
# Take first partner in numbered list
pm1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", partner_clean)
if pm1:
partner_clean = pm1.group(1).strip()
animals.append({
"wsCode": active_ws,
"litterDob": active_dob,
"name": name,
"farbschlag": farbschlag,
"gender": "male" if is_male else "female",
"owner": owner_clean,
"abgabeDate": abgabe_date,
"abgabeWeight": abgew.strip(),
"deathDate": death_date,
"deathCause": death_cause,
"partnerName": partner_clean,
"partnerDob": partner_dob,
})
return litters, animals
def main():
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
ap = argparse.ArgumentParser(description="FEAT-8d docx extractor")
ap.add_argument("--docx", default=DEFAULT_DOCX,
help="Pfad zur 'im Detail.docx'")
args = ap.parse_args()
if not os.path.isfile(args.docx):
print(f"Fehler: Datei nicht gefunden: {args.docx}", file=sys.stderr)
sys.exit(1)
os.makedirs(OUT, exist_ok=True)
print(f"Lese: {args.docx}")
litters, animals = extract(args.docx)
litters_path = os.path.join(OUT, "docx_litters.json")
animals_path = os.path.join(OUT, "docx_animals.json")
with open(litters_path, "w", encoding="utf-8") as f:
json.dump(litters, f, ensure_ascii=False, indent=2)
with open(animals_path, "w", encoding="utf-8") as f:
json.dump(animals, f, ensure_ascii=False, indent=2)
# Stats
named = sum(1 for a in animals if a["name"])
with_owner = sum(1 for a in animals if a["owner"])
with_death = sum(1 for a in animals if a["deathDate"])
with_abgabe = sum(1 for a in animals if a["abgabeDate"])
print(f"Würfe: {len(litters)}")
print(f"Tiere: {len(animals)} (benannt: {named})")
print(f" mit Abnehmer: {with_owner}")
print(f" mit Abgabe-Dat: {with_abgabe}")
print(f" mit Tod-Datum: {with_death}")
print(f"Ausgabe: {OUT}")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,77 @@
"""Tests for extract_docx.py — run: python test_extract_docx.py"""
import sys
import os
# Require the docx to exist; skip if not present (CI won't have it)
DOCX = os.path.join(r"C:\Users\gulum\dev",
"Wurfchronik der Kleinen Chaoten im Detail.docx")
SKIP = not os.path.isfile(DOCX)
import extract_docx as ed
failed = 0
def check(name, cond):
global failed
print(("ok: " if cond else "FAIL: ") + name)
if not cond:
failed += 1
# --- _norm_dob ---
check("norm_dob 2-digit year", ed._norm_dob("12.09.21") == "12.09.2021")
check("norm_dob 4-digit year", ed._norm_dob("07.04.2019") == "07.04.2019")
check("norm_dob empty", ed._norm_dob("") == "")
# --- _LAST_DATE_RE ---
check("last date: simple", ed._LAST_DATE_RE.findall("07.04.2019") == ["07.04.2019"])
check("last date: dual-day", ed._LAST_DATE_RE.findall("16./17.03.2021") == ["17.03.2021"])
check("last date: dual-month", ed._LAST_DATE_RE.findall("31.05/*01.06.2023") == ["01.06.2023"])
# --- _is_internal ---
check("ZT is internal", ed._is_internal("ZT"))
check("BLEIBT is internal", ed._is_internal("BLEIBT"))
check("VG: is internal", ed._is_internal("VG: Partner"))
check("real name not internal", not ed._is_internal("Marion Teichmann"))
# --- LITTER_RE ---
cases = [
("D19-Wurf *07.04.2019Von: Xhemile gen. Chanel v.d. Kleinen Chaoten & Omero v.d. Kleinen Chaoten WS: 2/4Notiz:", "2/4", "07.04.2019"),
("-Wurf *16./17.03.2021Von: Victoria Welby v.d. Kleinen Chaoten & Patch v.d. Kleinen Chaoten WS: /5Notiz:", "/5", "16./17.03.2021"),
("S22-Wurf *31.05/*01.06.2023Von: Velvet v.d. Kleinen Chaoten & Vance Sohn v.d. Kleinen ChaotenWS: 3/3", "3/3", "31.05/*01.06.2023"),
("Q21-Wurf *21.03.2022Von: Belica gen. Emi v.d. Kleinen Chaoten & Zac gen. Action v.d. Kleinen ChaotenWS: 2/4Notiz:", "2/4", "21.03.2022"),
]
for para, expected_ws, _ in cases:
m = ed.LITTER_RE.search(para)
ws = m.group(5).replace(" ", "") if m else None
check(f"LITTER_RE matches: {para[:50]}...", ws == expected_ws)
if SKIP:
print("(Skipping live-docx tests: file not found)")
else:
litters, animals = ed.extract(DOCX)
check("93 litters extracted", len(litters) == 93)
check("All litters have wsCode", all(l["wsCode"] for l in litters))
check("All litters have dob", all(l["dob"] for l in litters))
check(">200 named animals", len(animals) >= 200)
check(">150 animals with owner", sum(1 for a in animals if a["owner"]) >= 150)
check(">20 animals with death date", sum(1 for a in animals if a["deathDate"]) >= 20)
# Verify first litter
d19 = next((l for l in litters if l["litterId"] == "D19-Wurf"), None)
check("D19-Wurf found", d19 is not None)
check("D19-Wurf dob correct", d19 and d19["dob"] == "07.04.2019")
check("D19-Wurf wsCode = 2/4", d19 and d19["wsCode"] == "2/4")
check("D19-Wurf mother contains Xhemile", d19 and "Xhemile" in d19["motherName"])
# Verify Eddie in animals
eddie = next((a for a in animals if a["name"] == "Eddie" and a["wsCode"] == "2/4"), None)
check("Eddie found in D19-Wurf", eddie is not None)
check("Eddie gender=male (Zobel* suffix)", eddie and eddie["gender"] == "male")
check("Eddie abgabeDate", eddie and eddie["abgabeDate"] == "12.09.2021")
# Flash death date
flash = next((a for a in animals if a["name"] == "Flash" and a["wsCode"] == "3/3"), None)
check("Flash death date extracted", flash and flash["deathDate"] == "16.09.2023")
check("Flash death cause extracted", flash and "Tumor" in flash["deathCause"])
if failed:
print(f"\n{failed} test(s) FAILED")
sys.exit(1)
print("\nALL PASS")