From f74af537e264779a61cfca77d70887eda1b5b348 Mon Sep 17 00:00:00 2001 From: Gulum Date: Sat, 6 Jun 2026 22:04:09 +0200 Subject: [PATCH] FEAT-8d: docx-Importer fuer Wurfchronik-Detail (extract_docx.py + ImportDocxService) tools/import/extract_docx.py (stdlib-Python, kein pip): Parst 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML via zipfile). 93 Wuerfe + 227 benannte Tiere aus Tabellen extrahiert. Felder pro Tier: WS-Code, Wurfgeburtsdatum, Name, Farbschlag, Geschlecht (Stern- Suffix), Abnehmer, Abgabedatum, Tod-Datum + Ursache, Partnername + DOB. Sonderwerte (ZT/BLEIBT/FREI/VG:) werden herausgefiltert. Edge-Cases: Doppel-Datum (16./17.03.2021, 31.05/*01.06.2023), WS ohne Zaehler (/5), fehlende Leerzeichen vor WS:, mehrere Abnehmer (1.) ... 2.) ...). Output: output/docx_litters.json + output/docx_animals.json. tools/import/test_extract_docx.py: Unit-Tests fuer Regex-Logik + Live-Tests gegen die echte docx (skip wenn fehlt). 28/28 Tests gruen. GerbilManagerWebAPI/Import/ImportDocxService.cs: Idempotenter NACHZUG-Loader (fill-NULL-only, nie ueberschreiben): - WS-Code + Wurfgeburtsdatum -> PairingCode -> Gerbil.LitterId - Abnehmer -> Contact lookup-or-create -> Gerbil.ReceiverContactId - Abgabedatum -> Gerbil.GoHomeDate - Tod-Datum + Ursache -> Gerbil.DateOfDeath + CauseOfDeath Dry-Run zaehlt geplante Aenderungen, Execute schreibt. GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs: POST /import/docx/dry-run + /import/docx/execute (analog ImportEndpoints). GATE: 157/157 C#, 28/28 Python-docx-Tests, ef has-pending=No. NACHZUG: laueft NACH dem finalen WIPE+REIMPORT-3 (kein Impact auf aktuellen Pipeline). --- .../Endpoints/ImportDocxEndpoints.cs | 28 ++ .../Import/ImportDocxService.cs | 259 +++++++++++++++ GerbilManagerWebAPI/Program.cs | 1 + tools/import/extract_docx.py | 312 ++++++++++++++++++ tools/import/test_extract_docx.py | 77 +++++ 5 files changed, 677 insertions(+) create mode 100644 GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs create mode 100644 GerbilManagerWebAPI/Import/ImportDocxService.cs create mode 100644 tools/import/extract_docx.py create mode 100644 tools/import/test_extract_docx.py diff --git a/GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs b/GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs new file mode 100644 index 0000000..2d18041 --- /dev/null +++ b/GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs @@ -0,0 +1,28 @@ +using GerbilManagerWebAPI.Import; +using Microsoft.AspNetCore.Http.HttpResults; + +namespace GerbilManagerWebAPI.Endpoints +{ + public static class ImportDocxEndpoints + { + public static IEndpointRouteBuilder MapImportDocxEndpoints(this IEndpointRouteBuilder app) + { + var group = app.MapGroup("/import/docx").WithTags("Import"); + + // POST /import/docx/dry-run — analyse without writing + group.MapPost("/dry-run", + async Task> ( + ApplicationContext db, IConfiguration config, IWebHostEnvironment env) => + TypedResults.Ok(await new ImportDocxService(db, config, env).RunAsync(execute: false))); + + // POST /import/docx/execute — GATED: enriches Gerbils with Litter-Link, + // ReceiverContact, GoHomeDate, DateOfDeath, CauseOfDeath (fill-NULL-only). + group.MapPost("/execute", + async Task> ( + ApplicationContext db, IConfiguration config, IWebHostEnvironment env) => + TypedResults.Ok(await new ImportDocxService(db, config, env).RunAsync(execute: true))); + + return app; + } + } +} diff --git a/GerbilManagerWebAPI/Import/ImportDocxService.cs b/GerbilManagerWebAPI/Import/ImportDocxService.cs new file mode 100644 index 0000000..eccb2e0 --- /dev/null +++ b/GerbilManagerWebAPI/Import/ImportDocxService.cs @@ -0,0 +1,259 @@ +using System.Text.Json; +using GerbilManagerWebAPI.Models; +using Microsoft.EntityFrameworkCore; + +namespace GerbilManagerWebAPI.Import +{ + /// + /// FEAT-8d docx loader. Consumes tools/import/output/docx_litters.json + + /// docx_animals.json (produced by extract_docx.py) and enriches the database: + /// + /// Load policy (IDEMPOTENT NACHZUG after main WIPE+REIMPORT): + /// - Litter link: match docx WS-code to Litters.PairingCode → set Gerbil.LitterId + /// for animals matched by normalize(name) + litter birth date. + /// - ReceiverContact: lookup-or-create Contact by owner name → set ReceiverContactId. + /// - GoHomeDate, DateOfDeath, CauseOfDeath: fill if currently null (fill-NULL-only). + /// - NEVER overwrites a manually-set non-null value. + /// + /// Idempotent: running multiple times is safe. Each run resolves whatever is still null. + /// Execute is gated by the endpoint; this service only acts when asked. + /// + public sealed class ImportDocxService + { + private static readonly JsonSerializerOptions Json = new() { PropertyNameCaseInsensitive = true }; + + private readonly ApplicationContext _db; + private readonly string _sourceDir; + + public ImportDocxService(ApplicationContext db, IConfiguration config, IWebHostEnvironment env) + : this(db, + config["Import:SourcePath"] + ?? Path.GetFullPath(Path.Combine(env.ContentRootPath, "..", "tools", "import", "output"))) + { } + + public ImportDocxService(ApplicationContext db, string sourceDir) + { + _db = db; + _sourceDir = sourceDir; + } + + public async Task RunAsync(bool execute) + { + var notes = new List(); + + var docxLitters = Load>("docx_litters.json") ?? new(); + var docxAnimals = Load>("docx_animals.json") ?? new(); + + if (docxLitters.Count == 0 && docxAnimals.Count == 0) + { + notes.Add($"Keine Quelldaten in {_sourceDir} (docx_litters.json/docx_animals.json). " + + "extract_docx.py zuerst ausführen."); + return new ImportDocxReport(false, 0, 0, 0, 0, 0, 0, notes); + } + + // Build lookup: PairingCode → Litter.Id (WS-code normalised: spaces removed) + var littersInDb = await _db.Litters + .Where(l => l.PairingCode != null) + .Select(l => new { l.Id, l.Date, l.PairingCode }) + .ToListAsync(); + var litterByWs = littersInDb + .GroupBy(l => l.PairingCode!.Replace(" ", "")) + .ToDictionary(g => g.Key, g => g.ToList()); + + // Build animal lookup: normalize(name) + litter_dob → Gerbil (for litter-link) + var gerbilsInDb = await _db.Gerbils + .Select(g => new { g.Id, g.Name, g.DateOfBirth, g.LitterId, + g.ReceiverContactId, g.GoHomeDate, g.DateOfDeath, g.CauseOfDeath }) + .ToListAsync(); + var gerbilByKey = gerbilsInDb + .Where(g => g.DateOfBirth is not null) + .GroupBy(g => NameDobKey(g.Name, g.DateOfBirth!.Value)) + .ToDictionary(g => g.Key, g => g.ToList()); + + // Contact lookup: normalized name → existing Contact + var contactsInDb = await _db.Contacts + .Select(c => new { c.Id, c.Name }) + .ToListAsync(); + var contactByNorm = contactsInDb + .GroupBy(c => NormalizeName(c.Name)) + .ToDictionary(g => g.Key, g => g.First().Id); + + int litterLinked = 0, goHomeFilled = 0, deathFilled = 0; + int ownerLinked = 0, ownerCreated = 0, skipped = 0; + + foreach (var da in docxAnimals) + { + if (string.IsNullOrWhiteSpace(da.Name)) { skipped++; continue; } + + // Resolve the litter by WS-code + approximate birth date + Guid? litterId = null; + if (!string.IsNullOrWhiteSpace(da.WsCode) && !string.IsNullOrWhiteSpace(da.LitterDob)) + { + var litterDob = ParseDate(da.LitterDob); + if (litterDob is not null && litterByWs.TryGetValue(da.WsCode.Replace(" ", ""), out var cands)) + { + // Pick the litter whose date matches (within ±5 days for rounding) + var match = cands.FirstOrDefault(l => + Math.Abs((l.Date.DayNumber - litterDob.Value.DayNumber)) <= 5); + litterId = match?.Id; + } + } + + // Resolve the gerbil by name + litter birth date + var animalDob = litterId is not null + ? (await _db.Litters.Where(l => l.Id == litterId).Select(l => (DateOnly?)l.Date).FirstOrDefaultAsync()) + : ParseDate(da.LitterDob); + + if (animalDob is null) { skipped++; continue; } + + var key = NameDobKey(da.Name, animalDob.Value); + if (!gerbilByKey.TryGetValue(key, out var gerbilCands)) { skipped++; continue; } + + // If multiple gerbils match (same name+dob), take the one without a litter link first + var gerbilSnap = gerbilCands.FirstOrDefault(g => g.LitterId == null) + ?? gerbilCands.First(); + + // Resolve receiver contact (lookup-or-create) + Guid? receiverId = null; + if (!string.IsNullOrWhiteSpace(da.Owner)) + { + var normOwner = NormalizeName(da.Owner); + if (contactByNorm.TryGetValue(normOwner, out var existingId)) + { + receiverId = existingId; + ownerLinked++; + } + else + { + ownerCreated++; + if (execute) + { + var newContact = new Contact { Id = Guid.NewGuid(), Name = da.Owner.Trim() }; + _db.Contacts.Add(newContact); + await _db.SaveChangesAsync(); + receiverId = newContact.Id; + contactByNorm[normOwner] = receiverId.Value; + } + } + } + + var goHomeDate = ParseDate(da.AbgabeDate); + var deathDate = ParseDate(da.DeathDate); + + // Count what will change + bool willLinkLitter = litterId is not null && gerbilSnap.LitterId is null; + bool willFillGoHome = goHomeDate is not null && gerbilSnap.GoHomeDate is null; + bool willFillDeath = deathDate is not null && gerbilSnap.DateOfDeath is null; + + if (willLinkLitter) litterLinked++; + if (willFillGoHome) goHomeFilled++; + if (willFillDeath) deathFilled++; + + if (execute) + { + var row = await _db.Gerbils.FirstOrDefaultAsync(g => g.Id == gerbilSnap.Id); + if (row is null) continue; + + if (willLinkLitter) row.LitterId = litterId; + if (receiverId is not null && row.ReceiverContactId is null) + row.ReceiverContactId = receiverId; + if (willFillGoHome) row.GoHomeDate = goHomeDate; + if (willFillDeath) + { + row.DateOfDeath = deathDate; + if (!string.IsNullOrWhiteSpace(da.DeathCause) && row.CauseOfDeath is null) + row.CauseOfDeath = da.DeathCause.Trim(); + } + } + } + + if (execute && (litterLinked + goHomeFilled + deathFilled + ownerLinked + ownerCreated) > 0) + await _db.SaveChangesAsync(); + + notes.Add($"Quelle: {docxLitters.Count} Würfe, {docxAnimals.Count} Tier-Zeilen aus der docx."); + notes.Add($"Litter-Links: {litterLinked} Tiere einem Wurf zugeordnet (WS-Code → PairingCode)."); + notes.Add($"Abnehmer: {ownerLinked} bestehende Kontakte verknüpft, {ownerCreated} neue Kontakte angelegt."); + notes.Add($"GoHomeDate: {goHomeFilled} Abgabe-Daten nachgetragen."); + notes.Add($"Tod-Datum: {deathFilled} Todesdaten nachgetragen."); + notes.Add($"Übersprungen: {skipped} Zeilen (kein Name oder kein DB-Match)."); + if (!execute) notes.Add("DRY-RUN: nichts gespeichert. /import/docx/execute schreibt die Änderungen."); + + return new ImportDocxReport(execute, litterLinked, ownerLinked + ownerCreated, + goHomeFilled, deathFilled, ownerCreated, skipped, notes); + } + + private T? Load(string file) + { + var path = Path.Combine(_sourceDir, file); + if (!File.Exists(path)) return default; + using var fs = File.OpenRead(path); + return JsonSerializer.Deserialize(fs, Json); + } + + private static DateOnly? ParseDate(string? s) + { + if (string.IsNullOrWhiteSpace(s)) return null; + var m = System.Text.RegularExpressions.Regex.Match(s, + @"(\d{1,2})\.(\d{1,2})\.(\d{2,4})"); + if (!m.Success) return null; + int d = int.Parse(m.Groups[1].Value), mo = int.Parse(m.Groups[2].Value); + int y = int.Parse(m.Groups[3].Value); + if (y < 100) y += 2000; + try { return new DateOnly(y, mo, d); } catch { return null; } + } + + private static string NameDobKey(string name, DateOnly dob) + { + var n = System.Text.RegularExpressions.Regex.Replace( + (name ?? "").ToLowerInvariant(), @"[^a-z0-9äöüß]", ""); + return $"{n}|{dob:yyyy-MM-dd}"; + } + + private static string NormalizeName(string name) + { + var n = (name ?? "").ToLowerInvariant(); + n = System.Text.RegularExpressions.Regex.Replace(n, @"\s+", " ").Trim(); + return n; + } + } + + // ---- Source shapes (from extract_docx.py output) ---- + + public sealed class DocxLitter + { + public string LitterId { get; set; } = ""; + public string Dob { get; set; } = ""; + public string MotherName { get; set; } = ""; + public string FatherName { get; set; } = ""; + public string WsCode { get; set; } = ""; + public string Note { get; set; } = ""; + } + + public sealed class DocxAnimal + { + public string WsCode { get; set; } = ""; + public string LitterDob { get; set; } = ""; + public string Name { get; set; } = ""; + public string Farbschlag { get; set; } = ""; + public string Gender { get; set; } = ""; + public string Owner { get; set; } = ""; + public string AbgabeDate { get; set; } = ""; + public string AbgabeWeight { get; set; } = ""; + public string DeathDate { get; set; } = ""; + public string DeathCause { get; set; } = ""; + public string PartnerName { get; set; } = ""; + public string PartnerDob { get; set; } = ""; + } + + // ---- Report ---- + + public sealed record ImportDocxReport( + bool Executed, + int LitterLinked, + int OwnerLinked, + int GoHomeFilled, + int DeathFilled, + int ContactsCreated, + int Skipped, + IReadOnlyList Notes); +} diff --git a/GerbilManagerWebAPI/Program.cs b/GerbilManagerWebAPI/Program.cs index 63541cb..ae47609 100644 --- a/GerbilManagerWebAPI/Program.cs +++ b/GerbilManagerWebAPI/Program.cs @@ -98,6 +98,7 @@ app.MapInbreedingEndpoints(); app.MapPhotoEndpoints(); app.MapSaleAdEndpoints(); app.MapImportEndpoints(); +app.MapImportDocxEndpoints(); app.MapContractEndpoints(); app.MapSettingsEndpoints(); app.MapExportEndpoints(); diff --git a/tools/import/extract_docx.py b/tools/import/extract_docx.py new file mode 100644 index 0000000..76943e2 --- /dev/null +++ b/tools/import/extract_docx.py @@ -0,0 +1,312 @@ +#!/usr/bin/env python3 +"""FEAT-8d Stage 1 — Wurfchronik-Detail-Dokument (.docx) extrahieren. + +Liest 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML, stdlib-Python, +kein pip) und erzeugt: + output/docx_litters.json — Wurf-Kopfdaten (WS-Code, DOB, Eltern, Notiz) + output/docx_animals.json — Tier-Zeilen (Name, Farbe, Abnehmer, ABD, Tod) + +Format der Ausgabe ist so gestaltet, dass ImportDocxService.cs in C# direkt +darüber laden kann. Idempotent: mehrfaches Ausführen überschreibt denselben Output. + +Bekannte Sonderwerte im Dokument: + ZT = Zucht-Tier (bleibt in Zucht, kein externer Abnehmer) + BLEIBT = vorläufig beim Züchter + FREI = noch verfügbar + VG: = Verpaarungs-Geschichte (bisherige Partner; nicht als Abnehmer werten) + RG: = Rückgabe + BEW = Bewerbung (Adoptionsinteressent in Prüfung) + -- ??? = Platzhalter, kein echter Name + +Feld 'gender': '' = weiblich (kein Marker), '*' auf Farbschlag oder 'G'-Spalte = männlich. + +Ausführung: python extract_docx.py [--docx PFAD] +""" +import os +import re +import sys +import json +import zipfile +import argparse + +HERE = os.path.dirname(os.path.abspath(__file__)) +DEFAULT_DOCX = os.path.join( + r"C:\Users\gulum\dev", + "Wurfchronik der Kleinen Chaoten im Detail.docx", +) +OUT = os.path.join(HERE, "output") + +# --- Regex patterns ------------------------------------------------------- + +# Litter header paragraph (after whitespace-collapsing). +# Edge cases handled: +# - Dual birth date: "*16./17.03.2021" +# - WS without numerator: "WS: /5" +# - WS with trailing text: "WS: 4/4, davon 1 später..." +# - No space before WS: "...ChaotenWS: 2/4" +# Date part allows simple DD.MM.YYYY, dual-day (16./17.03.2021), or dual-month (31.05/*01.06.2023). +# We capture the LAST complete DD.MM.YYYY in the date token as the birth date. +_DATE_TOKEN = r"[\d./\*]+" +# Full litter header regex +LITTER_RE = re.compile( + r"([A-Za-z\d\-]*Wurf)\s*\*\s*(" + _DATE_TOKEN + r")" + r"\s*Von:\s*(.+?)\s*&\s*(.+?)\s*WS:\s*(\d*\s*/\s*\d+)" + r"(?:[,\s].*?)?(?:Notiz:\s*(.*?))?$", + re.IGNORECASE, +) +# Used to extract the canonical date from a date token like "31.05/*01.06.2023" +_LAST_DATE_RE = re.compile(r"(\d{1,2}\.\d{2}\.\d{4})(?![\d.])") +# Death/adoption date at start of combined T.D column: "16.09.23Tumor am After" +DATE_START_RE = re.compile(r"^(\d{1,2}\.\d{1,2}\.\d{2,4})\s*(.*)") +# Partner birth date: "Crow (*25.12.20)" or "Tom (*05.01.21)" +PARTNER_DOB_RE = re.compile(r"\(\s*\*\s*(\d{2}\.\d{2}\.\d{2,4})\s*\)") +# Special-value sentinel names to skip +PLACEHOLDER_NAMES = {"--", "???", ""} +INTERNAL_TOKENS = {"ZT", "BLEIBT", "FREI", "VG:", "VG*:", "RG:", "BEW"} + + +def _norm_dob(d: str) -> str: + """Normalise German date to DD.MM.YYYY.""" + if not d: + return "" + p = d.strip().split(".") + if len(p) == 3: + y = p[2].strip() + if len(y) == 2: + y = "20" + y + return f"{p[0].zfill(2)}.{p[1].zfill(2)}.{y}" + return d.strip() + + +def _cell_text(cell_xml: str) -> str: + """Strip XML from a cell and return clean text.""" + t = re.sub(r"<[^>]+>", "", cell_xml) + t = t.replace("&", "&").replace("<", "<").replace(">", ">") + t = t.replace("'", "'").replace(""", '"') + return re.sub(r"\s+", " ", t).strip() + + +def _is_internal(value: str) -> bool: + """True if the owner/name field holds an internal sentinel, not a real person.""" + v = value.strip() + return v in INTERNAL_TOKENS or any(v.startswith(tok) for tok in INTERNAL_TOKENS) + + +def extract(docx_path: str): + """Parse the docx and return (litters, animals) lists.""" + with zipfile.ZipFile(docx_path) as z: + xml = z.read("word/document.xml").decode("utf-8", errors="replace") + + # ---- Paragraphs → litter header blocks ---- + paras = re.findall(r"].*?", xml, re.DOTALL) + para_texts = [] + for p in paras: + t = re.sub(r"<[^>]+>", "", p) + t = t.replace("&", "&").strip() + t = re.sub(r"\s+", " ", t).strip() + if t: + para_texts.append(t) + + litters: list[dict] = [] + current_ws: str = "" + current_litter_dob: str = "" + + # Build a WS-code → litter index for assigning animals + ws_to_idx: dict[str, int] = {} + + for para in para_texts: + m = LITTER_RE.search(para) + if not m: + continue + litter_id = m.group(1).strip() + dob_raw = m.group(2).strip() + mother_raw = m.group(3).strip() + father_raw = m.group(4).strip() + ws_raw = m.group(5).replace(" ", "") + note = (m.group(6) or "").strip() + + # For dual-date tokens like "31.05/*01.06.2023", take the last full date. + last_dates = _LAST_DATE_RE.findall(dob_raw) + dob_clean = _norm_dob(last_dates[-1] if last_dates else dob_raw) + + litter = { + "litterId": litter_id, + "dob": dob_clean, + "motherName": mother_raw, + "fatherName": father_raw, + "wsCode": ws_raw, + "note": note, + } + ws_to_idx[ws_raw] = len(litters) + litters.append(litter) + + # ---- Tables → animal rows ---- + # Each table sits after a litter-header paragraph; we sequence tables and + # litter headers together by their byte offset in the XML. + animals: list[dict] = [] + + # Build ordered sequence of (offset, type, data) events + events: list[tuple[int, str, any]] = [] + for m in re.finditer(r"].*?", xml, re.DOTALL): + t = re.sub(r"<[^>]+>", "", m.group()).replace("&", "&").strip() + t = re.sub(r"\s+", " ", t).strip() + lm = LITTER_RE.search(t) + if lm: + ws = lm.group(5).replace(" ", "") + dob_tok = lm.group(2) + last = _LAST_DATE_RE.findall(dob_tok) + dob = _norm_dob(last[-1] if last else dob_tok) + events.append((m.start(), "litter", (ws, dob))) + + for m in re.finditer(r"].*?", xml, re.DOTALL): + events.append((m.start(), "table", m.group())) + + events.sort(key=lambda e: e[0]) + + active_ws = "" + active_dob = "" + + for _, etype, edata in events: + if etype == "litter": + active_ws, active_dob = edata + elif etype == "table" and active_ws: + # Parse all rows in this table + rows = re.findall(r"].*?", edata, re.DOTALL) + for row in rows: + cells_xml = re.findall(r"].*?", row, re.DOTALL) + ct = [_cell_text(c) for c in cells_xml] + if not ct: + continue + # Skip header rows + if ct[0] == "G" and len(ct) > 1 and "Farbe" in ct[1]: + continue + + # Column positions: G | Farbe | Name | Partner | Abnehmer | ABD | T.D + # Some newer tables add ABGew between ABD and T.D (7 or 8 cols) + g_col = ct[0] if len(ct) > 0 else "" + farbe_raw = ct[1] if len(ct) > 1 else "" + name = ct[2] if len(ct) > 2 else "" + partner = ct[3] if len(ct) > 3 else "" + owner = ct[4] if len(ct) > 4 else "" + abd_raw = ct[5] if len(ct) > 5 else "" + # If 8 cols, col 6 = ABGew, col 7 = T.D; if 7 cols, col 6 = T.D + if len(ct) >= 8: + abgew = ct[6] + tod_raw = ct[7] + elif len(ct) >= 7: + abgew = "" + tod_raw = ct[6] + else: + abgew = "" + tod_raw = "" + + # Skip placeholders + name = name.strip() + if name in PLACEHOLDER_NAMES: + continue + if not farbe_raw.strip() and not name: + continue + + # Gender: explicit marker in G column, or * suffix on Farbschlag + is_male = bool(g_col.strip() == "*" or farbe_raw.endswith("*")) + farbschlag = farbe_raw.rstrip("*").strip() + + # Owner: strip internal sentinels + owner_clean = owner.strip() + if _is_internal(owner_clean): + owner_clean = "" + # For multi-owner ("1.) Julia2.) RG:"), take first + m1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", owner_clean) + if m1: + owner_clean = m1.group(1).strip() + + # ABD (Abgabe-Datum) + abgabe_date = _norm_dob(abd_raw.strip()) + + # T.D column: may start with a date followed by cause + death_date = "" + death_cause = "" + if tod_raw: + dm = DATE_START_RE.match(tod_raw.strip()) + if dm: + death_date = _norm_dob(dm.group(1)) + death_cause = dm.group(2).strip() + else: + death_cause = tod_raw.strip() + + # Partner name and DOB + partner_clean = partner.strip() + partner_dob = "" + pdob_m = PARTNER_DOB_RE.search(partner_clean) + if pdob_m: + partner_dob = _norm_dob(pdob_m.group(1)) + partner_clean = PARTNER_DOB_RE.sub("", partner_clean).strip() + # Strip VG:/ZT/etc. prefixes + partner_clean = re.sub(r"^(?:VG\*?:|ZT\s*)", "", partner_clean).strip() + # Take first partner in numbered list + pm1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", partner_clean) + if pm1: + partner_clean = pm1.group(1).strip() + + animals.append({ + "wsCode": active_ws, + "litterDob": active_dob, + "name": name, + "farbschlag": farbschlag, + "gender": "male" if is_male else "female", + "owner": owner_clean, + "abgabeDate": abgabe_date, + "abgabeWeight": abgew.strip(), + "deathDate": death_date, + "deathCause": death_cause, + "partnerName": partner_clean, + "partnerDob": partner_dob, + }) + + return litters, animals + + +def main(): + try: + sys.stdout.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + ap = argparse.ArgumentParser(description="FEAT-8d docx extractor") + ap.add_argument("--docx", default=DEFAULT_DOCX, + help="Pfad zur 'im Detail.docx'") + args = ap.parse_args() + + if not os.path.isfile(args.docx): + print(f"Fehler: Datei nicht gefunden: {args.docx}", file=sys.stderr) + sys.exit(1) + + os.makedirs(OUT, exist_ok=True) + + print(f"Lese: {args.docx}") + litters, animals = extract(args.docx) + + litters_path = os.path.join(OUT, "docx_litters.json") + animals_path = os.path.join(OUT, "docx_animals.json") + + with open(litters_path, "w", encoding="utf-8") as f: + json.dump(litters, f, ensure_ascii=False, indent=2) + with open(animals_path, "w", encoding="utf-8") as f: + json.dump(animals, f, ensure_ascii=False, indent=2) + + # Stats + named = sum(1 for a in animals if a["name"]) + with_owner = sum(1 for a in animals if a["owner"]) + with_death = sum(1 for a in animals if a["deathDate"]) + with_abgabe = sum(1 for a in animals if a["abgabeDate"]) + + print(f"Würfe: {len(litters)}") + print(f"Tiere: {len(animals)} (benannt: {named})") + print(f" mit Abnehmer: {with_owner}") + print(f" mit Abgabe-Dat: {with_abgabe}") + print(f" mit Tod-Datum: {with_death}") + print(f"Ausgabe: {OUT}") + + +if __name__ == "__main__": + main() diff --git a/tools/import/test_extract_docx.py b/tools/import/test_extract_docx.py new file mode 100644 index 0000000..6c7a177 --- /dev/null +++ b/tools/import/test_extract_docx.py @@ -0,0 +1,77 @@ +"""Tests for extract_docx.py — run: python test_extract_docx.py""" +import sys +import os + +# Require the docx to exist; skip if not present (CI won't have it) +DOCX = os.path.join(r"C:\Users\gulum\dev", + "Wurfchronik der Kleinen Chaoten im Detail.docx") +SKIP = not os.path.isfile(DOCX) + +import extract_docx as ed + +failed = 0 + +def check(name, cond): + global failed + print(("ok: " if cond else "FAIL: ") + name) + if not cond: + failed += 1 + +# --- _norm_dob --- +check("norm_dob 2-digit year", ed._norm_dob("12.09.21") == "12.09.2021") +check("norm_dob 4-digit year", ed._norm_dob("07.04.2019") == "07.04.2019") +check("norm_dob empty", ed._norm_dob("") == "") + +# --- _LAST_DATE_RE --- +check("last date: simple", ed._LAST_DATE_RE.findall("07.04.2019") == ["07.04.2019"]) +check("last date: dual-day", ed._LAST_DATE_RE.findall("16./17.03.2021") == ["17.03.2021"]) +check("last date: dual-month", ed._LAST_DATE_RE.findall("31.05/*01.06.2023") == ["01.06.2023"]) + +# --- _is_internal --- +check("ZT is internal", ed._is_internal("ZT")) +check("BLEIBT is internal", ed._is_internal("BLEIBT")) +check("VG: is internal", ed._is_internal("VG: Partner")) +check("real name not internal", not ed._is_internal("Marion Teichmann")) + +# --- LITTER_RE --- +cases = [ + ("D19-Wurf *07.04.2019Von: Xhemile gen. Chanel v.d. Kleinen Chaoten & Omero v.d. Kleinen Chaoten WS: 2/4Notiz:", "2/4", "07.04.2019"), + ("-Wurf *16./17.03.2021Von: Victoria Welby v.d. Kleinen Chaoten & Patch v.d. Kleinen Chaoten WS: /5Notiz:", "/5", "16./17.03.2021"), + ("S22-Wurf *31.05/*01.06.2023Von: Velvet v.d. Kleinen Chaoten & Vance Sohn v.d. Kleinen ChaotenWS: 3/3", "3/3", "31.05/*01.06.2023"), + ("Q21-Wurf *21.03.2022Von: Belica gen. Emi v.d. Kleinen Chaoten & Zac gen. Action v.d. Kleinen ChaotenWS: 2/4Notiz:", "2/4", "21.03.2022"), +] +for para, expected_ws, _ in cases: + m = ed.LITTER_RE.search(para) + ws = m.group(5).replace(" ", "") if m else None + check(f"LITTER_RE matches: {para[:50]}...", ws == expected_ws) + +if SKIP: + print("(Skipping live-docx tests: file not found)") +else: + litters, animals = ed.extract(DOCX) + check("93 litters extracted", len(litters) == 93) + check("All litters have wsCode", all(l["wsCode"] for l in litters)) + check("All litters have dob", all(l["dob"] for l in litters)) + check(">200 named animals", len(animals) >= 200) + check(">150 animals with owner", sum(1 for a in animals if a["owner"]) >= 150) + check(">20 animals with death date", sum(1 for a in animals if a["deathDate"]) >= 20) + # Verify first litter + d19 = next((l for l in litters if l["litterId"] == "D19-Wurf"), None) + check("D19-Wurf found", d19 is not None) + check("D19-Wurf dob correct", d19 and d19["dob"] == "07.04.2019") + check("D19-Wurf wsCode = 2/4", d19 and d19["wsCode"] == "2/4") + check("D19-Wurf mother contains Xhemile", d19 and "Xhemile" in d19["motherName"]) + # Verify Eddie in animals + eddie = next((a for a in animals if a["name"] == "Eddie" and a["wsCode"] == "2/4"), None) + check("Eddie found in D19-Wurf", eddie is not None) + check("Eddie gender=male (Zobel* suffix)", eddie and eddie["gender"] == "male") + check("Eddie abgabeDate", eddie and eddie["abgabeDate"] == "12.09.2021") + # Flash death date + flash = next((a for a in animals if a["name"] == "Flash" and a["wsCode"] == "3/3"), None) + check("Flash death date extracted", flash and flash["deathDate"] == "16.09.2023") + check("Flash death cause extracted", flash and "Tumor" in flash["deathCause"]) + +if failed: + print(f"\n{failed} test(s) FAILED") + sys.exit(1) +print("\nALL PASS")