Deployment: - custom-app.compose.yaml: self-contained Compose fuer TrueNAS "Custom App" (absolute Host-Bind-Pfade, postgres:18, pull_policy always, Port 8090) - scripts/truenas-deploy.sh: Host-Skript create/redeploy via midclt (App bleibt unter Apps sichtbar) inkl. Image-Pull + Health-Check - ci.yml Deploy-Job: laeuft auf ubuntu-latest-Runner, kopiert Deploy-Dateien per SSH auf den NAS-Host und triggert truenas-deploy.sh (statt runs-on goldeye) - compose.yaml/.env.example: postgres:18 (Locale-Match zur Quell-DB), Port 8090 - .gitignore: .agents/, tools/rag/, deploy/truenas/.env (Secrets/Scratch) Aufgelaufene Feature-Arbeit (verified/Freeze, Migrationen, Import-Triage): - GerbilOverride/VerifiedGerbil-Endpoints + GerbilSnapshotService + Tests - EF-Migrationen (ShowInChronicle, Stillborn, BirthOrder, ManualFlag, DSGVO) - Frontend VerifizierteTierePage + verified-API + e2e-Spec - diverse Import-/Triage-Skripte und -Tests Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
342 lines
13 KiB
Python
342 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""FEAT-8d Stage 1 — Wurfchronik-Detail-Dokument (.docx) extrahieren.
|
|
|
|
Liest 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML, stdlib-Python,
|
|
kein pip) und erzeugt:
|
|
output/docx_litters.json — Wurf-Kopfdaten (WS-Code, DOB, Eltern, Notiz)
|
|
output/docx_animals.json — Tier-Zeilen (Name, Farbe, Abnehmer, ABD, Tod)
|
|
|
|
Format der Ausgabe ist so gestaltet, dass ImportDocxService.cs in C# direkt
|
|
darüber laden kann. Idempotent: mehrfaches Ausführen überschreibt denselben Output.
|
|
|
|
Bekannte Sonderwerte im Dokument:
|
|
ZT = Zucht-Tier (bleibt in Zucht, kein externer Abnehmer)
|
|
BLEIBT = vorläufig beim Züchter
|
|
FREI = noch verfügbar
|
|
VG: = Verpaarungs-Geschichte (bisherige Partner; nicht als Abnehmer werten)
|
|
RG: = Rückgabe
|
|
BEW = Bewerbung (Adoptionsinteressent in Prüfung)
|
|
-- ??? = Platzhalter, kein echter Name
|
|
|
|
Feld 'gender': '' = weiblich (kein Marker), '*' auf Farbschlag oder 'G'-Spalte = männlich.
|
|
|
|
Ausführung: python extract_docx.py [--docx PFAD]
|
|
"""
|
|
import os
|
|
import re
|
|
import sys
|
|
import json
|
|
import zipfile
|
|
import argparse
|
|
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
DEFAULT_DOCX = os.path.join(
|
|
r"C:\Users\gulum\dev",
|
|
"Wurfchronik der Kleinen Chaoten im Detail.docx",
|
|
)
|
|
OUT = os.path.join(HERE, "output")
|
|
|
|
# --- Regex patterns -------------------------------------------------------
|
|
|
|
# Litter header paragraph (after whitespace-collapsing).
|
|
# Edge cases handled:
|
|
# - Dual birth date: "*16./17.03.2021"
|
|
# - WS without numerator: "WS: /5"
|
|
# - WS with trailing text: "WS: 4/4, davon 1 später..."
|
|
# - No space before WS: "...ChaotenWS: 2/4"
|
|
# Date part allows simple DD.MM.YYYY, dual-day (16./17.03.2021), or dual-month (31.05/*01.06.2023).
|
|
# We capture the LAST complete DD.MM.YYYY in the date token as the birth date.
|
|
_DATE_TOKEN = r"[\d./\*]+"
|
|
# Full litter header regex
|
|
LITTER_RE = re.compile(
|
|
r"([A-Za-z\d\-]*Wurf)\s*\*\s*(" + _DATE_TOKEN + r")"
|
|
r"\s*Von:\s*(.+?)\s*&\s*(.+?)\s*WS:\s*(\d*\s*/\s*\d+)"
|
|
r"(?:[,\s].*?)?(?:Notiz:\s*(.*?))?$",
|
|
re.IGNORECASE,
|
|
)
|
|
# Used to extract the canonical date from a date token like "31.05/*01.06.2023"
|
|
_LAST_DATE_RE = re.compile(r"(\d{1,2}\.\d{2}\.\d{4})(?![\d.])")
|
|
# Death/adoption date at start of combined T.D column: "16.09.23Tumor am After"
|
|
DATE_START_RE = re.compile(r"^(\d{1,2}\.\d{1,2}\.\d{2,4})\s*(.*)")
|
|
# Partner birth date: "Crow (*25.12.20)" or "Tom (*05.01.21)"
|
|
PARTNER_DOB_RE = re.compile(r"\(\s*\*\s*(\d{2}\.\d{2}\.\d{2,4})\s*\)")
|
|
# Special-value sentinel names to skip
|
|
PLACEHOLDER_NAMES = {"--", "???", ""}
|
|
INTERNAL_TOKENS = {"ZT", "BLEIBT", "FREI", "VG:", "VG*:", "RG:", "BEW"}
|
|
|
|
|
|
def _norm_dob(d: str) -> str:
|
|
"""Normalise German date to DD.MM.YYYY."""
|
|
if not d:
|
|
return ""
|
|
p = d.strip().split(".")
|
|
if len(p) == 3:
|
|
y = p[2].strip()
|
|
if len(y) == 2:
|
|
y = "20" + y
|
|
return f"{p[0].zfill(2)}.{p[1].zfill(2)}.{y}"
|
|
return d.strip()
|
|
|
|
|
|
def _cell_text(cell_xml: str) -> str:
|
|
"""Strip XML from a <w:tc> cell and return clean text."""
|
|
t = re.sub(r"<[^>]+>", "", cell_xml)
|
|
t = t.replace("&", "&").replace("<", "<").replace(">", ">")
|
|
t = t.replace("'", "'").replace(""", '"')
|
|
return re.sub(r"\s+", " ", t).strip()
|
|
|
|
|
|
def _is_internal(value: str) -> bool:
|
|
"""True if the owner/name field holds an internal sentinel, not a real person."""
|
|
v = value.strip()
|
|
return v in INTERNAL_TOKENS or any(v.startswith(tok) for tok in INTERNAL_TOKENS)
|
|
|
|
|
|
def extract(docx_path: str):
|
|
"""Parse the docx and return (litters, animals) lists."""
|
|
with zipfile.ZipFile(docx_path) as z:
|
|
xml = z.read("word/document.xml").decode("utf-8", errors="replace")
|
|
|
|
# ---- Paragraphs → litter header blocks ----
|
|
paras = re.findall(r"<w:p[ >].*?</w:p>", xml, re.DOTALL)
|
|
para_texts = []
|
|
for p in paras:
|
|
t = re.sub(r"<[^>]+>", "", p)
|
|
t = t.replace("&", "&").strip()
|
|
t = re.sub(r"\s+", " ", t).strip()
|
|
if t:
|
|
para_texts.append(t)
|
|
|
|
litters: list[dict] = []
|
|
current_ws: str = ""
|
|
current_litter_dob: str = ""
|
|
|
|
# Build a WS-code → litter index for assigning animals
|
|
ws_to_idx: dict[str, int] = {}
|
|
|
|
for para in para_texts:
|
|
m = LITTER_RE.search(para)
|
|
if not m:
|
|
continue
|
|
litter_id = m.group(1).strip()
|
|
dob_raw = m.group(2).strip()
|
|
mother_raw = m.group(3).strip()
|
|
father_raw = m.group(4).strip()
|
|
ws_raw = m.group(5).replace(" ", "")
|
|
note = (m.group(6) or "").strip()
|
|
|
|
# For dual-date tokens like "31.05/*01.06.2023", take the last full date.
|
|
last_dates = _LAST_DATE_RE.findall(dob_raw)
|
|
dob_clean = _norm_dob(last_dates[-1] if last_dates else dob_raw)
|
|
|
|
# Map 22-Wurf and 23-Wurf to their correct alphabetical sequence name based on DOB
|
|
if litter_id == "22-Wurf":
|
|
dob_map = {
|
|
"17.06.2023": "V22-Wurf",
|
|
"13.07.2023": "X22-Wurf",
|
|
"20.07.2023": "Y22-Wurf",
|
|
"24.07.2023": "Z22-Wurf"
|
|
}
|
|
if dob_clean in dob_map:
|
|
litter_id = dob_map[dob_clean]
|
|
elif litter_id == "23-Wurf":
|
|
dob_map = {
|
|
"03.10.2023": "F23-Wurf",
|
|
"06.10.2023": "G23-Wurf",
|
|
"10.10.2023": "H23-Wurf",
|
|
"17.10.2023": "I23-Wurf",
|
|
"31.10.2023": "J23-Wurf",
|
|
"04.11.2023": "K23-Wurf",
|
|
"11.11.2023": "L23-Wurf",
|
|
"17.11.2023": "M23-Wurf",
|
|
"20.11.2023": "N23-Wurf",
|
|
"01.12.2023": "O23-Wurf",
|
|
"02.12.2023": "P23-Wurf",
|
|
"03.12.2023": "Q23-Wurf",
|
|
"05.12.2023": "R23-Wurf"
|
|
}
|
|
if dob_clean in dob_map:
|
|
litter_id = dob_map[dob_clean]
|
|
|
|
litter = {
|
|
"litterId": litter_id,
|
|
"dob": dob_clean,
|
|
"motherName": mother_raw,
|
|
"fatherName": father_raw,
|
|
"wsCode": ws_raw,
|
|
"note": note,
|
|
}
|
|
ws_to_idx[ws_raw] = len(litters)
|
|
litters.append(litter)
|
|
|
|
# ---- Tables → animal rows ----
|
|
# Each table sits after a litter-header paragraph; we sequence tables and
|
|
# litter headers together by their byte offset in the XML.
|
|
animals: list[dict] = []
|
|
|
|
# Build ordered sequence of (offset, type, data) events
|
|
events: list[tuple[int, str, any]] = []
|
|
for m in re.finditer(r"<w:p[ >].*?</w:p>", xml, re.DOTALL):
|
|
t = re.sub(r"<[^>]+>", "", m.group()).replace("&", "&").strip()
|
|
t = re.sub(r"\s+", " ", t).strip()
|
|
lm = LITTER_RE.search(t)
|
|
if lm:
|
|
ws = lm.group(5).replace(" ", "")
|
|
dob_tok = lm.group(2)
|
|
last = _LAST_DATE_RE.findall(dob_tok)
|
|
dob = _norm_dob(last[-1] if last else dob_tok)
|
|
events.append((m.start(), "litter", (ws, dob)))
|
|
|
|
for m in re.finditer(r"<w:tbl[ >].*?</w:tbl>", xml, re.DOTALL):
|
|
events.append((m.start(), "table", m.group()))
|
|
|
|
events.sort(key=lambda e: e[0])
|
|
|
|
active_ws = ""
|
|
active_dob = ""
|
|
|
|
for _, etype, edata in events:
|
|
if etype == "litter":
|
|
active_ws, active_dob = edata
|
|
elif etype == "table" and active_ws:
|
|
# Parse all rows in this table
|
|
rows = re.findall(r"<w:tr[ >].*?</w:tr>", edata, re.DOTALL)
|
|
for row in rows:
|
|
cells_xml = re.findall(r"<w:tc[ >].*?</w:tc>", row, re.DOTALL)
|
|
ct = [_cell_text(c) for c in cells_xml]
|
|
if not ct:
|
|
continue
|
|
# Skip header rows
|
|
if ct[0] == "G" and len(ct) > 1 and "Farbe" in ct[1]:
|
|
continue
|
|
|
|
# Column positions: G | Farbe | Name | Partner | Abnehmer | ABD | T.D
|
|
# Some newer tables add ABGew between ABD and T.D (7 or 8 cols)
|
|
g_col = ct[0] if len(ct) > 0 else ""
|
|
farbe_raw = ct[1] if len(ct) > 1 else ""
|
|
name = ct[2] if len(ct) > 2 else ""
|
|
partner = ct[3] if len(ct) > 3 else ""
|
|
owner = ct[4] if len(ct) > 4 else ""
|
|
abd_raw = ct[5] if len(ct) > 5 else ""
|
|
# If 8 cols, col 6 = ABGew, col 7 = T.D; if 7 cols, col 6 = T.D
|
|
if len(ct) >= 8:
|
|
abgew = ct[6]
|
|
tod_raw = ct[7]
|
|
elif len(ct) >= 7:
|
|
abgew = ""
|
|
tod_raw = ct[6]
|
|
else:
|
|
abgew = ""
|
|
tod_raw = ""
|
|
|
|
# Skip placeholders
|
|
name = name.strip()
|
|
if name in PLACEHOLDER_NAMES:
|
|
continue
|
|
if not farbe_raw.strip() and not name:
|
|
continue
|
|
|
|
# Gender: explicit marker in G column, or * suffix on Farbschlag
|
|
is_male = bool(g_col.strip() == "*" or farbe_raw.endswith("*"))
|
|
farbschlag = farbe_raw.rstrip("*").strip()
|
|
|
|
# Owner: strip internal sentinels
|
|
owner_clean = owner.strip()
|
|
if _is_internal(owner_clean):
|
|
owner_clean = ""
|
|
# For multi-owner ("1.) Julia2.) RG:"), take first
|
|
m1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", owner_clean)
|
|
if m1:
|
|
owner_clean = m1.group(1).strip()
|
|
|
|
# ABD (Abgabe-Datum)
|
|
abgabe_date = _norm_dob(abd_raw.strip())
|
|
|
|
# T.D column: may start with a date followed by cause
|
|
death_date = ""
|
|
death_cause = ""
|
|
if tod_raw:
|
|
dm = DATE_START_RE.match(tod_raw.strip())
|
|
if dm:
|
|
death_date = _norm_dob(dm.group(1))
|
|
death_cause = dm.group(2).strip()
|
|
else:
|
|
death_cause = tod_raw.strip()
|
|
|
|
# Partner name and DOB
|
|
partner_clean = partner.strip()
|
|
partner_dob = ""
|
|
pdob_m = PARTNER_DOB_RE.search(partner_clean)
|
|
if pdob_m:
|
|
partner_dob = _norm_dob(pdob_m.group(1))
|
|
partner_clean = PARTNER_DOB_RE.sub("", partner_clean).strip()
|
|
# Strip VG:/ZT/etc. prefixes
|
|
partner_clean = re.sub(r"^(?:VG\*?:|ZT\s*)", "", partner_clean).strip()
|
|
# Take first partner in numbered list
|
|
pm1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", partner_clean)
|
|
if pm1:
|
|
partner_clean = pm1.group(1).strip()
|
|
|
|
animals.append({
|
|
"wsCode": active_ws,
|
|
"litterDob": active_dob,
|
|
"name": name,
|
|
"farbschlag": farbschlag,
|
|
"gender": "male" if is_male else "female",
|
|
"owner": owner_clean,
|
|
"abgabeDate": abgabe_date,
|
|
"abgabeWeight": abgew.strip(),
|
|
"deathDate": death_date,
|
|
"deathCause": death_cause,
|
|
"partnerName": partner_clean,
|
|
"partnerDob": partner_dob,
|
|
})
|
|
|
|
return litters, animals
|
|
|
|
|
|
def main():
|
|
try:
|
|
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
except Exception:
|
|
pass
|
|
|
|
ap = argparse.ArgumentParser(description="FEAT-8d docx extractor")
|
|
ap.add_argument("--docx", default=DEFAULT_DOCX,
|
|
help="Pfad zur 'im Detail.docx'")
|
|
args = ap.parse_args()
|
|
|
|
if not os.path.isfile(args.docx):
|
|
print(f"Fehler: Datei nicht gefunden: {args.docx}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
os.makedirs(OUT, exist_ok=True)
|
|
|
|
print(f"Lese: {args.docx}")
|
|
litters, animals = extract(args.docx)
|
|
|
|
litters_path = os.path.join(OUT, "docx_litters.json")
|
|
animals_path = os.path.join(OUT, "docx_animals.json")
|
|
|
|
with open(litters_path, "w", encoding="utf-8") as f:
|
|
json.dump(litters, f, ensure_ascii=False, indent=2)
|
|
with open(animals_path, "w", encoding="utf-8") as f:
|
|
json.dump(animals, f, ensure_ascii=False, indent=2)
|
|
|
|
# Stats
|
|
named = sum(1 for a in animals if a["name"])
|
|
with_owner = sum(1 for a in animals if a["owner"])
|
|
with_death = sum(1 for a in animals if a["deathDate"])
|
|
with_abgabe = sum(1 for a in animals if a["abgabeDate"])
|
|
|
|
print(f"Würfe: {len(litters)}")
|
|
print(f"Tiere: {len(animals)} (benannt: {named})")
|
|
print(f" mit Abnehmer: {with_owner}")
|
|
print(f" mit Abgabe-Dat: {with_abgabe}")
|
|
print(f" mit Tod-Datum: {with_death}")
|
|
print(f"Ausgabe: {OUT}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|