#!/usr/bin/env python3 """FEAT-8d Stage 1 — Wurfchronik-Detail-Dokument (.docx) extrahieren. Liest 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML, stdlib-Python, kein pip) und erzeugt: output/docx_litters.json — Wurf-Kopfdaten (WS-Code, DOB, Eltern, Notiz) output/docx_animals.json — Tier-Zeilen (Name, Farbe, Abnehmer, ABD, Tod) Format der Ausgabe ist so gestaltet, dass ImportDocxService.cs in C# direkt darüber laden kann. Idempotent: mehrfaches Ausführen überschreibt denselben Output. Bekannte Sonderwerte im Dokument: ZT = Zucht-Tier (bleibt in Zucht, kein externer Abnehmer) BLEIBT = vorläufig beim Züchter FREI = noch verfügbar VG: = Verpaarungs-Geschichte (bisherige Partner; nicht als Abnehmer werten) RG: = Rückgabe BEW = Bewerbung (Adoptionsinteressent in Prüfung) -- ??? = Platzhalter, kein echter Name Feld 'gender': '' = weiblich (kein Marker), '*' auf Farbschlag oder 'G'-Spalte = männlich. Ausführung: python extract_docx.py [--docx PFAD] """ import os import re import sys import json import zipfile import argparse HERE = os.path.dirname(os.path.abspath(__file__)) DEFAULT_DOCX = os.path.join( r"C:\Users\gulum\dev", "Wurfchronik der Kleinen Chaoten im Detail.docx", ) OUT = os.path.join(HERE, "output") # --- Regex patterns ------------------------------------------------------- # Litter header paragraph (after whitespace-collapsing). # Edge cases handled: # - Dual birth date: "*16./17.03.2021" # - WS without numerator: "WS: /5" # - WS with trailing text: "WS: 4/4, davon 1 später..." # - No space before WS: "...ChaotenWS: 2/4" # Date part allows simple DD.MM.YYYY, dual-day (16./17.03.2021), or dual-month (31.05/*01.06.2023). # We capture the LAST complete DD.MM.YYYY in the date token as the birth date. _DATE_TOKEN = r"[\d./\*]+" # Full litter header regex LITTER_RE = re.compile( r"([A-Za-z\d\-]*Wurf)\s*\*\s*(" + _DATE_TOKEN + r")" r"\s*Von:\s*(.+?)\s*&\s*(.+?)\s*WS:\s*(\d*\s*/\s*\d+)" r"(?:[,\s].*?)?(?:Notiz:\s*(.*?))?$", re.IGNORECASE, ) # Used to extract the canonical date from a date token like "31.05/*01.06.2023" _LAST_DATE_RE = re.compile(r"(\d{1,2}\.\d{2}\.\d{4})(?![\d.])") # Death/adoption date at start of combined T.D column: "16.09.23Tumor am After" DATE_START_RE = re.compile(r"^(\d{1,2}\.\d{1,2}\.\d{2,4})\s*(.*)") # Partner birth date: "Crow (*25.12.20)" or "Tom (*05.01.21)" PARTNER_DOB_RE = re.compile(r"\(\s*\*\s*(\d{2}\.\d{2}\.\d{2,4})\s*\)") # Special-value sentinel names to skip PLACEHOLDER_NAMES = {"--", "???", ""} INTERNAL_TOKENS = {"ZT", "BLEIBT", "FREI", "VG:", "VG*:", "RG:", "BEW"} def _norm_dob(d: str) -> str: """Normalise German date to DD.MM.YYYY.""" if not d: return "" p = d.strip().split(".") if len(p) == 3: y = p[2].strip() if len(y) == 2: y = "20" + y return f"{p[0].zfill(2)}.{p[1].zfill(2)}.{y}" return d.strip() def _cell_text(cell_xml: str) -> str: """Strip XML from a cell and return clean text.""" t = re.sub(r"<[^>]+>", "", cell_xml) t = t.replace("&", "&").replace("<", "<").replace(">", ">") t = t.replace("'", "'").replace(""", '"') return re.sub(r"\s+", " ", t).strip() def _is_internal(value: str) -> bool: """True if the owner/name field holds an internal sentinel, not a real person.""" v = value.strip() return v in INTERNAL_TOKENS or any(v.startswith(tok) for tok in INTERNAL_TOKENS) def extract(docx_path: str): """Parse the docx and return (litters, animals) lists.""" with zipfile.ZipFile(docx_path) as z: xml = z.read("word/document.xml").decode("utf-8", errors="replace") # ---- Paragraphs → litter header blocks ---- paras = re.findall(r"].*?", xml, re.DOTALL) para_texts = [] for p in paras: t = re.sub(r"<[^>]+>", "", p) t = t.replace("&", "&").strip() t = re.sub(r"\s+", " ", t).strip() if t: para_texts.append(t) litters: list[dict] = [] current_ws: str = "" current_litter_dob: str = "" # Build a WS-code → litter index for assigning animals ws_to_idx: dict[str, int] = {} for para in para_texts: m = LITTER_RE.search(para) if not m: continue litter_id = m.group(1).strip() dob_raw = m.group(2).strip() mother_raw = m.group(3).strip() father_raw = m.group(4).strip() ws_raw = m.group(5).replace(" ", "") note = (m.group(6) or "").strip() # For dual-date tokens like "31.05/*01.06.2023", take the last full date. last_dates = _LAST_DATE_RE.findall(dob_raw) dob_clean = _norm_dob(last_dates[-1] if last_dates else dob_raw) litter = { "litterId": litter_id, "dob": dob_clean, "motherName": mother_raw, "fatherName": father_raw, "wsCode": ws_raw, "note": note, } ws_to_idx[ws_raw] = len(litters) litters.append(litter) # ---- Tables → animal rows ---- # Each table sits after a litter-header paragraph; we sequence tables and # litter headers together by their byte offset in the XML. animals: list[dict] = [] # Build ordered sequence of (offset, type, data) events events: list[tuple[int, str, any]] = [] for m in re.finditer(r"].*?", xml, re.DOTALL): t = re.sub(r"<[^>]+>", "", m.group()).replace("&", "&").strip() t = re.sub(r"\s+", " ", t).strip() lm = LITTER_RE.search(t) if lm: ws = lm.group(5).replace(" ", "") dob_tok = lm.group(2) last = _LAST_DATE_RE.findall(dob_tok) dob = _norm_dob(last[-1] if last else dob_tok) events.append((m.start(), "litter", (ws, dob))) for m in re.finditer(r"].*?", xml, re.DOTALL): events.append((m.start(), "table", m.group())) events.sort(key=lambda e: e[0]) active_ws = "" active_dob = "" for _, etype, edata in events: if etype == "litter": active_ws, active_dob = edata elif etype == "table" and active_ws: # Parse all rows in this table rows = re.findall(r"].*?", edata, re.DOTALL) for row in rows: cells_xml = re.findall(r"].*?", row, re.DOTALL) ct = [_cell_text(c) for c in cells_xml] if not ct: continue # Skip header rows if ct[0] == "G" and len(ct) > 1 and "Farbe" in ct[1]: continue # Column positions: G | Farbe | Name | Partner | Abnehmer | ABD | T.D # Some newer tables add ABGew between ABD and T.D (7 or 8 cols) g_col = ct[0] if len(ct) > 0 else "" farbe_raw = ct[1] if len(ct) > 1 else "" name = ct[2] if len(ct) > 2 else "" partner = ct[3] if len(ct) > 3 else "" owner = ct[4] if len(ct) > 4 else "" abd_raw = ct[5] if len(ct) > 5 else "" # If 8 cols, col 6 = ABGew, col 7 = T.D; if 7 cols, col 6 = T.D if len(ct) >= 8: abgew = ct[6] tod_raw = ct[7] elif len(ct) >= 7: abgew = "" tod_raw = ct[6] else: abgew = "" tod_raw = "" # Skip placeholders name = name.strip() if name in PLACEHOLDER_NAMES: continue if not farbe_raw.strip() and not name: continue # Gender: explicit marker in G column, or * suffix on Farbschlag is_male = bool(g_col.strip() == "*" or farbe_raw.endswith("*")) farbschlag = farbe_raw.rstrip("*").strip() # Owner: strip internal sentinels owner_clean = owner.strip() if _is_internal(owner_clean): owner_clean = "" # For multi-owner ("1.) Julia2.) RG:"), take first m1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", owner_clean) if m1: owner_clean = m1.group(1).strip() # ABD (Abgabe-Datum) abgabe_date = _norm_dob(abd_raw.strip()) # T.D column: may start with a date followed by cause death_date = "" death_cause = "" if tod_raw: dm = DATE_START_RE.match(tod_raw.strip()) if dm: death_date = _norm_dob(dm.group(1)) death_cause = dm.group(2).strip() else: death_cause = tod_raw.strip() # Partner name and DOB partner_clean = partner.strip() partner_dob = "" pdob_m = PARTNER_DOB_RE.search(partner_clean) if pdob_m: partner_dob = _norm_dob(pdob_m.group(1)) partner_clean = PARTNER_DOB_RE.sub("", partner_clean).strip() # Strip VG:/ZT/etc. prefixes partner_clean = re.sub(r"^(?:VG\*?:|ZT\s*)", "", partner_clean).strip() # Take first partner in numbered list pm1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", partner_clean) if pm1: partner_clean = pm1.group(1).strip() animals.append({ "wsCode": active_ws, "litterDob": active_dob, "name": name, "farbschlag": farbschlag, "gender": "male" if is_male else "female", "owner": owner_clean, "abgabeDate": abgabe_date, "abgabeWeight": abgew.strip(), "deathDate": death_date, "deathCause": death_cause, "partnerName": partner_clean, "partnerDob": partner_dob, }) return litters, animals def main(): try: sys.stdout.reconfigure(encoding="utf-8", errors="replace") except Exception: pass ap = argparse.ArgumentParser(description="FEAT-8d docx extractor") ap.add_argument("--docx", default=DEFAULT_DOCX, help="Pfad zur 'im Detail.docx'") args = ap.parse_args() if not os.path.isfile(args.docx): print(f"Fehler: Datei nicht gefunden: {args.docx}", file=sys.stderr) sys.exit(1) os.makedirs(OUT, exist_ok=True) print(f"Lese: {args.docx}") litters, animals = extract(args.docx) litters_path = os.path.join(OUT, "docx_litters.json") animals_path = os.path.join(OUT, "docx_animals.json") with open(litters_path, "w", encoding="utf-8") as f: json.dump(litters, f, ensure_ascii=False, indent=2) with open(animals_path, "w", encoding="utf-8") as f: json.dump(animals, f, ensure_ascii=False, indent=2) # Stats named = sum(1 for a in animals if a["name"]) with_owner = sum(1 for a in animals if a["owner"]) with_death = sum(1 for a in animals if a["deathDate"]) with_abgabe = sum(1 for a in animals if a["abgabeDate"]) print(f"Würfe: {len(litters)}") print(f"Tiere: {len(animals)} (benannt: {named})") print(f" mit Abnehmer: {with_owner}") print(f" mit Abgabe-Dat: {with_abgabe}") print(f" mit Tod-Datum: {with_death}") print(f"Ausgabe: {OUT}") if __name__ == "__main__": main()