FEAT-8d: docx-Importer fuer Wurfchronik-Detail (extract_docx.py + ImportDocxService)
tools/import/extract_docx.py (stdlib-Python, kein pip): Parst 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML via zipfile). 93 Wuerfe + 227 benannte Tiere aus Tabellen extrahiert. Felder pro Tier: WS-Code, Wurfgeburtsdatum, Name, Farbschlag, Geschlecht (Stern- Suffix), Abnehmer, Abgabedatum, Tod-Datum + Ursache, Partnername + DOB. Sonderwerte (ZT/BLEIBT/FREI/VG:) werden herausgefiltert. Edge-Cases: Doppel-Datum (16./17.03.2021, 31.05/*01.06.2023), WS ohne Zaehler (/5), fehlende Leerzeichen vor WS:, mehrere Abnehmer (1.) ... 2.) ...). Output: output/docx_litters.json + output/docx_animals.json. tools/import/test_extract_docx.py: Unit-Tests fuer Regex-Logik + Live-Tests gegen die echte docx (skip wenn fehlt). 28/28 Tests gruen. GerbilManagerWebAPI/Import/ImportDocxService.cs: Idempotenter NACHZUG-Loader (fill-NULL-only, nie ueberschreiben): - WS-Code + Wurfgeburtsdatum -> PairingCode -> Gerbil.LitterId - Abnehmer -> Contact lookup-or-create -> Gerbil.ReceiverContactId - Abgabedatum -> Gerbil.GoHomeDate - Tod-Datum + Ursache -> Gerbil.DateOfDeath + CauseOfDeath Dry-Run zaehlt geplante Aenderungen, Execute schreibt. GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs: POST /import/docx/dry-run + /import/docx/execute (analog ImportEndpoints). GATE: 157/157 C#, 28/28 Python-docx-Tests, ef has-pending=No. NACHZUG: laueft NACH dem finalen WIPE+REIMPORT-3 (kein Impact auf aktuellen Pipeline).
This commit is contained in:
28
GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs
Normal file
28
GerbilManagerWebAPI/Endpoints/ImportDocxEndpoints.cs
Normal file
@@ -0,0 +1,28 @@
|
||||
using GerbilManagerWebAPI.Import;
|
||||
using Microsoft.AspNetCore.Http.HttpResults;
|
||||
|
||||
namespace GerbilManagerWebAPI.Endpoints
|
||||
{
|
||||
public static class ImportDocxEndpoints
|
||||
{
|
||||
public static IEndpointRouteBuilder MapImportDocxEndpoints(this IEndpointRouteBuilder app)
|
||||
{
|
||||
var group = app.MapGroup("/import/docx").WithTags("Import");
|
||||
|
||||
// POST /import/docx/dry-run — analyse without writing
|
||||
group.MapPost("/dry-run",
|
||||
async Task<Ok<ImportDocxReport>> (
|
||||
ApplicationContext db, IConfiguration config, IWebHostEnvironment env) =>
|
||||
TypedResults.Ok(await new ImportDocxService(db, config, env).RunAsync(execute: false)));
|
||||
|
||||
// POST /import/docx/execute — GATED: enriches Gerbils with Litter-Link,
|
||||
// ReceiverContact, GoHomeDate, DateOfDeath, CauseOfDeath (fill-NULL-only).
|
||||
group.MapPost("/execute",
|
||||
async Task<Ok<ImportDocxReport>> (
|
||||
ApplicationContext db, IConfiguration config, IWebHostEnvironment env) =>
|
||||
TypedResults.Ok(await new ImportDocxService(db, config, env).RunAsync(execute: true)));
|
||||
|
||||
return app;
|
||||
}
|
||||
}
|
||||
}
|
||||
259
GerbilManagerWebAPI/Import/ImportDocxService.cs
Normal file
259
GerbilManagerWebAPI/Import/ImportDocxService.cs
Normal file
@@ -0,0 +1,259 @@
|
||||
using System.Text.Json;
|
||||
using GerbilManagerWebAPI.Models;
|
||||
using Microsoft.EntityFrameworkCore;
|
||||
|
||||
namespace GerbilManagerWebAPI.Import
|
||||
{
|
||||
/// <summary>
|
||||
/// FEAT-8d docx loader. Consumes tools/import/output/docx_litters.json +
|
||||
/// docx_animals.json (produced by extract_docx.py) and enriches the database:
|
||||
///
|
||||
/// Load policy (IDEMPOTENT NACHZUG after main WIPE+REIMPORT):
|
||||
/// - Litter link: match docx WS-code to Litters.PairingCode → set Gerbil.LitterId
|
||||
/// for animals matched by normalize(name) + litter birth date.
|
||||
/// - ReceiverContact: lookup-or-create Contact by owner name → set ReceiverContactId.
|
||||
/// - GoHomeDate, DateOfDeath, CauseOfDeath: fill if currently null (fill-NULL-only).
|
||||
/// - NEVER overwrites a manually-set non-null value.
|
||||
///
|
||||
/// Idempotent: running multiple times is safe. Each run resolves whatever is still null.
|
||||
/// Execute is gated by the endpoint; this service only acts when asked.
|
||||
/// </summary>
|
||||
public sealed class ImportDocxService
|
||||
{
|
||||
private static readonly JsonSerializerOptions Json = new() { PropertyNameCaseInsensitive = true };
|
||||
|
||||
private readonly ApplicationContext _db;
|
||||
private readonly string _sourceDir;
|
||||
|
||||
public ImportDocxService(ApplicationContext db, IConfiguration config, IWebHostEnvironment env)
|
||||
: this(db,
|
||||
config["Import:SourcePath"]
|
||||
?? Path.GetFullPath(Path.Combine(env.ContentRootPath, "..", "tools", "import", "output")))
|
||||
{ }
|
||||
|
||||
public ImportDocxService(ApplicationContext db, string sourceDir)
|
||||
{
|
||||
_db = db;
|
||||
_sourceDir = sourceDir;
|
||||
}
|
||||
|
||||
public async Task<ImportDocxReport> RunAsync(bool execute)
|
||||
{
|
||||
var notes = new List<string>();
|
||||
|
||||
var docxLitters = Load<List<DocxLitter>>("docx_litters.json") ?? new();
|
||||
var docxAnimals = Load<List<DocxAnimal>>("docx_animals.json") ?? new();
|
||||
|
||||
if (docxLitters.Count == 0 && docxAnimals.Count == 0)
|
||||
{
|
||||
notes.Add($"Keine Quelldaten in {_sourceDir} (docx_litters.json/docx_animals.json). " +
|
||||
"extract_docx.py zuerst ausführen.");
|
||||
return new ImportDocxReport(false, 0, 0, 0, 0, 0, 0, notes);
|
||||
}
|
||||
|
||||
// Build lookup: PairingCode → Litter.Id (WS-code normalised: spaces removed)
|
||||
var littersInDb = await _db.Litters
|
||||
.Where(l => l.PairingCode != null)
|
||||
.Select(l => new { l.Id, l.Date, l.PairingCode })
|
||||
.ToListAsync();
|
||||
var litterByWs = littersInDb
|
||||
.GroupBy(l => l.PairingCode!.Replace(" ", ""))
|
||||
.ToDictionary(g => g.Key, g => g.ToList());
|
||||
|
||||
// Build animal lookup: normalize(name) + litter_dob → Gerbil (for litter-link)
|
||||
var gerbilsInDb = await _db.Gerbils
|
||||
.Select(g => new { g.Id, g.Name, g.DateOfBirth, g.LitterId,
|
||||
g.ReceiverContactId, g.GoHomeDate, g.DateOfDeath, g.CauseOfDeath })
|
||||
.ToListAsync();
|
||||
var gerbilByKey = gerbilsInDb
|
||||
.Where(g => g.DateOfBirth is not null)
|
||||
.GroupBy(g => NameDobKey(g.Name, g.DateOfBirth!.Value))
|
||||
.ToDictionary(g => g.Key, g => g.ToList());
|
||||
|
||||
// Contact lookup: normalized name → existing Contact
|
||||
var contactsInDb = await _db.Contacts
|
||||
.Select(c => new { c.Id, c.Name })
|
||||
.ToListAsync();
|
||||
var contactByNorm = contactsInDb
|
||||
.GroupBy(c => NormalizeName(c.Name))
|
||||
.ToDictionary(g => g.Key, g => g.First().Id);
|
||||
|
||||
int litterLinked = 0, goHomeFilled = 0, deathFilled = 0;
|
||||
int ownerLinked = 0, ownerCreated = 0, skipped = 0;
|
||||
|
||||
foreach (var da in docxAnimals)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(da.Name)) { skipped++; continue; }
|
||||
|
||||
// Resolve the litter by WS-code + approximate birth date
|
||||
Guid? litterId = null;
|
||||
if (!string.IsNullOrWhiteSpace(da.WsCode) && !string.IsNullOrWhiteSpace(da.LitterDob))
|
||||
{
|
||||
var litterDob = ParseDate(da.LitterDob);
|
||||
if (litterDob is not null && litterByWs.TryGetValue(da.WsCode.Replace(" ", ""), out var cands))
|
||||
{
|
||||
// Pick the litter whose date matches (within ±5 days for rounding)
|
||||
var match = cands.FirstOrDefault(l =>
|
||||
Math.Abs((l.Date.DayNumber - litterDob.Value.DayNumber)) <= 5);
|
||||
litterId = match?.Id;
|
||||
}
|
||||
}
|
||||
|
||||
// Resolve the gerbil by name + litter birth date
|
||||
var animalDob = litterId is not null
|
||||
? (await _db.Litters.Where(l => l.Id == litterId).Select(l => (DateOnly?)l.Date).FirstOrDefaultAsync())
|
||||
: ParseDate(da.LitterDob);
|
||||
|
||||
if (animalDob is null) { skipped++; continue; }
|
||||
|
||||
var key = NameDobKey(da.Name, animalDob.Value);
|
||||
if (!gerbilByKey.TryGetValue(key, out var gerbilCands)) { skipped++; continue; }
|
||||
|
||||
// If multiple gerbils match (same name+dob), take the one without a litter link first
|
||||
var gerbilSnap = gerbilCands.FirstOrDefault(g => g.LitterId == null)
|
||||
?? gerbilCands.First();
|
||||
|
||||
// Resolve receiver contact (lookup-or-create)
|
||||
Guid? receiverId = null;
|
||||
if (!string.IsNullOrWhiteSpace(da.Owner))
|
||||
{
|
||||
var normOwner = NormalizeName(da.Owner);
|
||||
if (contactByNorm.TryGetValue(normOwner, out var existingId))
|
||||
{
|
||||
receiverId = existingId;
|
||||
ownerLinked++;
|
||||
}
|
||||
else
|
||||
{
|
||||
ownerCreated++;
|
||||
if (execute)
|
||||
{
|
||||
var newContact = new Contact { Id = Guid.NewGuid(), Name = da.Owner.Trim() };
|
||||
_db.Contacts.Add(newContact);
|
||||
await _db.SaveChangesAsync();
|
||||
receiverId = newContact.Id;
|
||||
contactByNorm[normOwner] = receiverId.Value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var goHomeDate = ParseDate(da.AbgabeDate);
|
||||
var deathDate = ParseDate(da.DeathDate);
|
||||
|
||||
// Count what will change
|
||||
bool willLinkLitter = litterId is not null && gerbilSnap.LitterId is null;
|
||||
bool willFillGoHome = goHomeDate is not null && gerbilSnap.GoHomeDate is null;
|
||||
bool willFillDeath = deathDate is not null && gerbilSnap.DateOfDeath is null;
|
||||
|
||||
if (willLinkLitter) litterLinked++;
|
||||
if (willFillGoHome) goHomeFilled++;
|
||||
if (willFillDeath) deathFilled++;
|
||||
|
||||
if (execute)
|
||||
{
|
||||
var row = await _db.Gerbils.FirstOrDefaultAsync(g => g.Id == gerbilSnap.Id);
|
||||
if (row is null) continue;
|
||||
|
||||
if (willLinkLitter) row.LitterId = litterId;
|
||||
if (receiverId is not null && row.ReceiverContactId is null)
|
||||
row.ReceiverContactId = receiverId;
|
||||
if (willFillGoHome) row.GoHomeDate = goHomeDate;
|
||||
if (willFillDeath)
|
||||
{
|
||||
row.DateOfDeath = deathDate;
|
||||
if (!string.IsNullOrWhiteSpace(da.DeathCause) && row.CauseOfDeath is null)
|
||||
row.CauseOfDeath = da.DeathCause.Trim();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (execute && (litterLinked + goHomeFilled + deathFilled + ownerLinked + ownerCreated) > 0)
|
||||
await _db.SaveChangesAsync();
|
||||
|
||||
notes.Add($"Quelle: {docxLitters.Count} Würfe, {docxAnimals.Count} Tier-Zeilen aus der docx.");
|
||||
notes.Add($"Litter-Links: {litterLinked} Tiere einem Wurf zugeordnet (WS-Code → PairingCode).");
|
||||
notes.Add($"Abnehmer: {ownerLinked} bestehende Kontakte verknüpft, {ownerCreated} neue Kontakte angelegt.");
|
||||
notes.Add($"GoHomeDate: {goHomeFilled} Abgabe-Daten nachgetragen.");
|
||||
notes.Add($"Tod-Datum: {deathFilled} Todesdaten nachgetragen.");
|
||||
notes.Add($"Übersprungen: {skipped} Zeilen (kein Name oder kein DB-Match).");
|
||||
if (!execute) notes.Add("DRY-RUN: nichts gespeichert. /import/docx/execute schreibt die Änderungen.");
|
||||
|
||||
return new ImportDocxReport(execute, litterLinked, ownerLinked + ownerCreated,
|
||||
goHomeFilled, deathFilled, ownerCreated, skipped, notes);
|
||||
}
|
||||
|
||||
private T? Load<T>(string file)
|
||||
{
|
||||
var path = Path.Combine(_sourceDir, file);
|
||||
if (!File.Exists(path)) return default;
|
||||
using var fs = File.OpenRead(path);
|
||||
return JsonSerializer.Deserialize<T>(fs, Json);
|
||||
}
|
||||
|
||||
private static DateOnly? ParseDate(string? s)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(s)) return null;
|
||||
var m = System.Text.RegularExpressions.Regex.Match(s,
|
||||
@"(\d{1,2})\.(\d{1,2})\.(\d{2,4})");
|
||||
if (!m.Success) return null;
|
||||
int d = int.Parse(m.Groups[1].Value), mo = int.Parse(m.Groups[2].Value);
|
||||
int y = int.Parse(m.Groups[3].Value);
|
||||
if (y < 100) y += 2000;
|
||||
try { return new DateOnly(y, mo, d); } catch { return null; }
|
||||
}
|
||||
|
||||
private static string NameDobKey(string name, DateOnly dob)
|
||||
{
|
||||
var n = System.Text.RegularExpressions.Regex.Replace(
|
||||
(name ?? "").ToLowerInvariant(), @"[^a-z0-9äöüß]", "");
|
||||
return $"{n}|{dob:yyyy-MM-dd}";
|
||||
}
|
||||
|
||||
private static string NormalizeName(string name)
|
||||
{
|
||||
var n = (name ?? "").ToLowerInvariant();
|
||||
n = System.Text.RegularExpressions.Regex.Replace(n, @"\s+", " ").Trim();
|
||||
return n;
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Source shapes (from extract_docx.py output) ----
|
||||
|
||||
public sealed class DocxLitter
|
||||
{
|
||||
public string LitterId { get; set; } = "";
|
||||
public string Dob { get; set; } = "";
|
||||
public string MotherName { get; set; } = "";
|
||||
public string FatherName { get; set; } = "";
|
||||
public string WsCode { get; set; } = "";
|
||||
public string Note { get; set; } = "";
|
||||
}
|
||||
|
||||
public sealed class DocxAnimal
|
||||
{
|
||||
public string WsCode { get; set; } = "";
|
||||
public string LitterDob { get; set; } = "";
|
||||
public string Name { get; set; } = "";
|
||||
public string Farbschlag { get; set; } = "";
|
||||
public string Gender { get; set; } = "";
|
||||
public string Owner { get; set; } = "";
|
||||
public string AbgabeDate { get; set; } = "";
|
||||
public string AbgabeWeight { get; set; } = "";
|
||||
public string DeathDate { get; set; } = "";
|
||||
public string DeathCause { get; set; } = "";
|
||||
public string PartnerName { get; set; } = "";
|
||||
public string PartnerDob { get; set; } = "";
|
||||
}
|
||||
|
||||
// ---- Report ----
|
||||
|
||||
public sealed record ImportDocxReport(
|
||||
bool Executed,
|
||||
int LitterLinked,
|
||||
int OwnerLinked,
|
||||
int GoHomeFilled,
|
||||
int DeathFilled,
|
||||
int ContactsCreated,
|
||||
int Skipped,
|
||||
IReadOnlyList<string> Notes);
|
||||
}
|
||||
@@ -98,6 +98,7 @@ app.MapInbreedingEndpoints();
|
||||
app.MapPhotoEndpoints();
|
||||
app.MapSaleAdEndpoints();
|
||||
app.MapImportEndpoints();
|
||||
app.MapImportDocxEndpoints();
|
||||
app.MapContractEndpoints();
|
||||
app.MapSettingsEndpoints();
|
||||
app.MapExportEndpoints();
|
||||
|
||||
312
tools/import/extract_docx.py
Normal file
312
tools/import/extract_docx.py
Normal file
@@ -0,0 +1,312 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FEAT-8d Stage 1 — Wurfchronik-Detail-Dokument (.docx) extrahieren.
|
||||
|
||||
Liest 'Wurfchronik der Kleinen Chaoten im Detail.docx' (Word/XML, stdlib-Python,
|
||||
kein pip) und erzeugt:
|
||||
output/docx_litters.json — Wurf-Kopfdaten (WS-Code, DOB, Eltern, Notiz)
|
||||
output/docx_animals.json — Tier-Zeilen (Name, Farbe, Abnehmer, ABD, Tod)
|
||||
|
||||
Format der Ausgabe ist so gestaltet, dass ImportDocxService.cs in C# direkt
|
||||
darüber laden kann. Idempotent: mehrfaches Ausführen überschreibt denselben Output.
|
||||
|
||||
Bekannte Sonderwerte im Dokument:
|
||||
ZT = Zucht-Tier (bleibt in Zucht, kein externer Abnehmer)
|
||||
BLEIBT = vorläufig beim Züchter
|
||||
FREI = noch verfügbar
|
||||
VG: = Verpaarungs-Geschichte (bisherige Partner; nicht als Abnehmer werten)
|
||||
RG: = Rückgabe
|
||||
BEW = Bewerbung (Adoptionsinteressent in Prüfung)
|
||||
-- ??? = Platzhalter, kein echter Name
|
||||
|
||||
Feld 'gender': '' = weiblich (kein Marker), '*' auf Farbschlag oder 'G'-Spalte = männlich.
|
||||
|
||||
Ausführung: python extract_docx.py [--docx PFAD]
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import json
|
||||
import zipfile
|
||||
import argparse
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
DEFAULT_DOCX = os.path.join(
|
||||
r"C:\Users\gulum\dev",
|
||||
"Wurfchronik der Kleinen Chaoten im Detail.docx",
|
||||
)
|
||||
OUT = os.path.join(HERE, "output")
|
||||
|
||||
# --- Regex patterns -------------------------------------------------------
|
||||
|
||||
# Litter header paragraph (after whitespace-collapsing).
|
||||
# Edge cases handled:
|
||||
# - Dual birth date: "*16./17.03.2021"
|
||||
# - WS without numerator: "WS: /5"
|
||||
# - WS with trailing text: "WS: 4/4, davon 1 später..."
|
||||
# - No space before WS: "...ChaotenWS: 2/4"
|
||||
# Date part allows simple DD.MM.YYYY, dual-day (16./17.03.2021), or dual-month (31.05/*01.06.2023).
|
||||
# We capture the LAST complete DD.MM.YYYY in the date token as the birth date.
|
||||
_DATE_TOKEN = r"[\d./\*]+"
|
||||
# Full litter header regex
|
||||
LITTER_RE = re.compile(
|
||||
r"([A-Za-z\d\-]*Wurf)\s*\*\s*(" + _DATE_TOKEN + r")"
|
||||
r"\s*Von:\s*(.+?)\s*&\s*(.+?)\s*WS:\s*(\d*\s*/\s*\d+)"
|
||||
r"(?:[,\s].*?)?(?:Notiz:\s*(.*?))?$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# Used to extract the canonical date from a date token like "31.05/*01.06.2023"
|
||||
_LAST_DATE_RE = re.compile(r"(\d{1,2}\.\d{2}\.\d{4})(?![\d.])")
|
||||
# Death/adoption date at start of combined T.D column: "16.09.23Tumor am After"
|
||||
DATE_START_RE = re.compile(r"^(\d{1,2}\.\d{1,2}\.\d{2,4})\s*(.*)")
|
||||
# Partner birth date: "Crow (*25.12.20)" or "Tom (*05.01.21)"
|
||||
PARTNER_DOB_RE = re.compile(r"\(\s*\*\s*(\d{2}\.\d{2}\.\d{2,4})\s*\)")
|
||||
# Special-value sentinel names to skip
|
||||
PLACEHOLDER_NAMES = {"--", "???", ""}
|
||||
INTERNAL_TOKENS = {"ZT", "BLEIBT", "FREI", "VG:", "VG*:", "RG:", "BEW"}
|
||||
|
||||
|
||||
def _norm_dob(d: str) -> str:
|
||||
"""Normalise German date to DD.MM.YYYY."""
|
||||
if not d:
|
||||
return ""
|
||||
p = d.strip().split(".")
|
||||
if len(p) == 3:
|
||||
y = p[2].strip()
|
||||
if len(y) == 2:
|
||||
y = "20" + y
|
||||
return f"{p[0].zfill(2)}.{p[1].zfill(2)}.{y}"
|
||||
return d.strip()
|
||||
|
||||
|
||||
def _cell_text(cell_xml: str) -> str:
|
||||
"""Strip XML from a <w:tc> cell and return clean text."""
|
||||
t = re.sub(r"<[^>]+>", "", cell_xml)
|
||||
t = t.replace("&", "&").replace("<", "<").replace(">", ">")
|
||||
t = t.replace("'", "'").replace(""", '"')
|
||||
return re.sub(r"\s+", " ", t).strip()
|
||||
|
||||
|
||||
def _is_internal(value: str) -> bool:
|
||||
"""True if the owner/name field holds an internal sentinel, not a real person."""
|
||||
v = value.strip()
|
||||
return v in INTERNAL_TOKENS or any(v.startswith(tok) for tok in INTERNAL_TOKENS)
|
||||
|
||||
|
||||
def extract(docx_path: str):
|
||||
"""Parse the docx and return (litters, animals) lists."""
|
||||
with zipfile.ZipFile(docx_path) as z:
|
||||
xml = z.read("word/document.xml").decode("utf-8", errors="replace")
|
||||
|
||||
# ---- Paragraphs → litter header blocks ----
|
||||
paras = re.findall(r"<w:p[ >].*?</w:p>", xml, re.DOTALL)
|
||||
para_texts = []
|
||||
for p in paras:
|
||||
t = re.sub(r"<[^>]+>", "", p)
|
||||
t = t.replace("&", "&").strip()
|
||||
t = re.sub(r"\s+", " ", t).strip()
|
||||
if t:
|
||||
para_texts.append(t)
|
||||
|
||||
litters: list[dict] = []
|
||||
current_ws: str = ""
|
||||
current_litter_dob: str = ""
|
||||
|
||||
# Build a WS-code → litter index for assigning animals
|
||||
ws_to_idx: dict[str, int] = {}
|
||||
|
||||
for para in para_texts:
|
||||
m = LITTER_RE.search(para)
|
||||
if not m:
|
||||
continue
|
||||
litter_id = m.group(1).strip()
|
||||
dob_raw = m.group(2).strip()
|
||||
mother_raw = m.group(3).strip()
|
||||
father_raw = m.group(4).strip()
|
||||
ws_raw = m.group(5).replace(" ", "")
|
||||
note = (m.group(6) or "").strip()
|
||||
|
||||
# For dual-date tokens like "31.05/*01.06.2023", take the last full date.
|
||||
last_dates = _LAST_DATE_RE.findall(dob_raw)
|
||||
dob_clean = _norm_dob(last_dates[-1] if last_dates else dob_raw)
|
||||
|
||||
litter = {
|
||||
"litterId": litter_id,
|
||||
"dob": dob_clean,
|
||||
"motherName": mother_raw,
|
||||
"fatherName": father_raw,
|
||||
"wsCode": ws_raw,
|
||||
"note": note,
|
||||
}
|
||||
ws_to_idx[ws_raw] = len(litters)
|
||||
litters.append(litter)
|
||||
|
||||
# ---- Tables → animal rows ----
|
||||
# Each table sits after a litter-header paragraph; we sequence tables and
|
||||
# litter headers together by their byte offset in the XML.
|
||||
animals: list[dict] = []
|
||||
|
||||
# Build ordered sequence of (offset, type, data) events
|
||||
events: list[tuple[int, str, any]] = []
|
||||
for m in re.finditer(r"<w:p[ >].*?</w:p>", xml, re.DOTALL):
|
||||
t = re.sub(r"<[^>]+>", "", m.group()).replace("&", "&").strip()
|
||||
t = re.sub(r"\s+", " ", t).strip()
|
||||
lm = LITTER_RE.search(t)
|
||||
if lm:
|
||||
ws = lm.group(5).replace(" ", "")
|
||||
dob_tok = lm.group(2)
|
||||
last = _LAST_DATE_RE.findall(dob_tok)
|
||||
dob = _norm_dob(last[-1] if last else dob_tok)
|
||||
events.append((m.start(), "litter", (ws, dob)))
|
||||
|
||||
for m in re.finditer(r"<w:tbl[ >].*?</w:tbl>", xml, re.DOTALL):
|
||||
events.append((m.start(), "table", m.group()))
|
||||
|
||||
events.sort(key=lambda e: e[0])
|
||||
|
||||
active_ws = ""
|
||||
active_dob = ""
|
||||
|
||||
for _, etype, edata in events:
|
||||
if etype == "litter":
|
||||
active_ws, active_dob = edata
|
||||
elif etype == "table" and active_ws:
|
||||
# Parse all rows in this table
|
||||
rows = re.findall(r"<w:tr[ >].*?</w:tr>", edata, re.DOTALL)
|
||||
for row in rows:
|
||||
cells_xml = re.findall(r"<w:tc[ >].*?</w:tc>", row, re.DOTALL)
|
||||
ct = [_cell_text(c) for c in cells_xml]
|
||||
if not ct:
|
||||
continue
|
||||
# Skip header rows
|
||||
if ct[0] == "G" and len(ct) > 1 and "Farbe" in ct[1]:
|
||||
continue
|
||||
|
||||
# Column positions: G | Farbe | Name | Partner | Abnehmer | ABD | T.D
|
||||
# Some newer tables add ABGew between ABD and T.D (7 or 8 cols)
|
||||
g_col = ct[0] if len(ct) > 0 else ""
|
||||
farbe_raw = ct[1] if len(ct) > 1 else ""
|
||||
name = ct[2] if len(ct) > 2 else ""
|
||||
partner = ct[3] if len(ct) > 3 else ""
|
||||
owner = ct[4] if len(ct) > 4 else ""
|
||||
abd_raw = ct[5] if len(ct) > 5 else ""
|
||||
# If 8 cols, col 6 = ABGew, col 7 = T.D; if 7 cols, col 6 = T.D
|
||||
if len(ct) >= 8:
|
||||
abgew = ct[6]
|
||||
tod_raw = ct[7]
|
||||
elif len(ct) >= 7:
|
||||
abgew = ""
|
||||
tod_raw = ct[6]
|
||||
else:
|
||||
abgew = ""
|
||||
tod_raw = ""
|
||||
|
||||
# Skip placeholders
|
||||
name = name.strip()
|
||||
if name in PLACEHOLDER_NAMES:
|
||||
continue
|
||||
if not farbe_raw.strip() and not name:
|
||||
continue
|
||||
|
||||
# Gender: explicit marker in G column, or * suffix on Farbschlag
|
||||
is_male = bool(g_col.strip() == "*" or farbe_raw.endswith("*"))
|
||||
farbschlag = farbe_raw.rstrip("*").strip()
|
||||
|
||||
# Owner: strip internal sentinels
|
||||
owner_clean = owner.strip()
|
||||
if _is_internal(owner_clean):
|
||||
owner_clean = ""
|
||||
# For multi-owner ("1.) Julia2.) RG:"), take first
|
||||
m1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", owner_clean)
|
||||
if m1:
|
||||
owner_clean = m1.group(1).strip()
|
||||
|
||||
# ABD (Abgabe-Datum)
|
||||
abgabe_date = _norm_dob(abd_raw.strip())
|
||||
|
||||
# T.D column: may start with a date followed by cause
|
||||
death_date = ""
|
||||
death_cause = ""
|
||||
if tod_raw:
|
||||
dm = DATE_START_RE.match(tod_raw.strip())
|
||||
if dm:
|
||||
death_date = _norm_dob(dm.group(1))
|
||||
death_cause = dm.group(2).strip()
|
||||
else:
|
||||
death_cause = tod_raw.strip()
|
||||
|
||||
# Partner name and DOB
|
||||
partner_clean = partner.strip()
|
||||
partner_dob = ""
|
||||
pdob_m = PARTNER_DOB_RE.search(partner_clean)
|
||||
if pdob_m:
|
||||
partner_dob = _norm_dob(pdob_m.group(1))
|
||||
partner_clean = PARTNER_DOB_RE.sub("", partner_clean).strip()
|
||||
# Strip VG:/ZT/etc. prefixes
|
||||
partner_clean = re.sub(r"^(?:VG\*?:|ZT\s*)", "", partner_clean).strip()
|
||||
# Take first partner in numbered list
|
||||
pm1 = re.match(r"1\.\)\s*(.+?)(?:2\.\)|$)", partner_clean)
|
||||
if pm1:
|
||||
partner_clean = pm1.group(1).strip()
|
||||
|
||||
animals.append({
|
||||
"wsCode": active_ws,
|
||||
"litterDob": active_dob,
|
||||
"name": name,
|
||||
"farbschlag": farbschlag,
|
||||
"gender": "male" if is_male else "female",
|
||||
"owner": owner_clean,
|
||||
"abgabeDate": abgabe_date,
|
||||
"abgabeWeight": abgew.strip(),
|
||||
"deathDate": death_date,
|
||||
"deathCause": death_cause,
|
||||
"partnerName": partner_clean,
|
||||
"partnerDob": partner_dob,
|
||||
})
|
||||
|
||||
return litters, animals
|
||||
|
||||
|
||||
def main():
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
ap = argparse.ArgumentParser(description="FEAT-8d docx extractor")
|
||||
ap.add_argument("--docx", default=DEFAULT_DOCX,
|
||||
help="Pfad zur 'im Detail.docx'")
|
||||
args = ap.parse_args()
|
||||
|
||||
if not os.path.isfile(args.docx):
|
||||
print(f"Fehler: Datei nicht gefunden: {args.docx}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
os.makedirs(OUT, exist_ok=True)
|
||||
|
||||
print(f"Lese: {args.docx}")
|
||||
litters, animals = extract(args.docx)
|
||||
|
||||
litters_path = os.path.join(OUT, "docx_litters.json")
|
||||
animals_path = os.path.join(OUT, "docx_animals.json")
|
||||
|
||||
with open(litters_path, "w", encoding="utf-8") as f:
|
||||
json.dump(litters, f, ensure_ascii=False, indent=2)
|
||||
with open(animals_path, "w", encoding="utf-8") as f:
|
||||
json.dump(animals, f, ensure_ascii=False, indent=2)
|
||||
|
||||
# Stats
|
||||
named = sum(1 for a in animals if a["name"])
|
||||
with_owner = sum(1 for a in animals if a["owner"])
|
||||
with_death = sum(1 for a in animals if a["deathDate"])
|
||||
with_abgabe = sum(1 for a in animals if a["abgabeDate"])
|
||||
|
||||
print(f"Würfe: {len(litters)}")
|
||||
print(f"Tiere: {len(animals)} (benannt: {named})")
|
||||
print(f" mit Abnehmer: {with_owner}")
|
||||
print(f" mit Abgabe-Dat: {with_abgabe}")
|
||||
print(f" mit Tod-Datum: {with_death}")
|
||||
print(f"Ausgabe: {OUT}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
77
tools/import/test_extract_docx.py
Normal file
77
tools/import/test_extract_docx.py
Normal file
@@ -0,0 +1,77 @@
|
||||
"""Tests for extract_docx.py — run: python test_extract_docx.py"""
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Require the docx to exist; skip if not present (CI won't have it)
|
||||
DOCX = os.path.join(r"C:\Users\gulum\dev",
|
||||
"Wurfchronik der Kleinen Chaoten im Detail.docx")
|
||||
SKIP = not os.path.isfile(DOCX)
|
||||
|
||||
import extract_docx as ed
|
||||
|
||||
failed = 0
|
||||
|
||||
def check(name, cond):
|
||||
global failed
|
||||
print(("ok: " if cond else "FAIL: ") + name)
|
||||
if not cond:
|
||||
failed += 1
|
||||
|
||||
# --- _norm_dob ---
|
||||
check("norm_dob 2-digit year", ed._norm_dob("12.09.21") == "12.09.2021")
|
||||
check("norm_dob 4-digit year", ed._norm_dob("07.04.2019") == "07.04.2019")
|
||||
check("norm_dob empty", ed._norm_dob("") == "")
|
||||
|
||||
# --- _LAST_DATE_RE ---
|
||||
check("last date: simple", ed._LAST_DATE_RE.findall("07.04.2019") == ["07.04.2019"])
|
||||
check("last date: dual-day", ed._LAST_DATE_RE.findall("16./17.03.2021") == ["17.03.2021"])
|
||||
check("last date: dual-month", ed._LAST_DATE_RE.findall("31.05/*01.06.2023") == ["01.06.2023"])
|
||||
|
||||
# --- _is_internal ---
|
||||
check("ZT is internal", ed._is_internal("ZT"))
|
||||
check("BLEIBT is internal", ed._is_internal("BLEIBT"))
|
||||
check("VG: is internal", ed._is_internal("VG: Partner"))
|
||||
check("real name not internal", not ed._is_internal("Marion Teichmann"))
|
||||
|
||||
# --- LITTER_RE ---
|
||||
cases = [
|
||||
("D19-Wurf *07.04.2019Von: Xhemile gen. Chanel v.d. Kleinen Chaoten & Omero v.d. Kleinen Chaoten WS: 2/4Notiz:", "2/4", "07.04.2019"),
|
||||
("-Wurf *16./17.03.2021Von: Victoria Welby v.d. Kleinen Chaoten & Patch v.d. Kleinen Chaoten WS: /5Notiz:", "/5", "16./17.03.2021"),
|
||||
("S22-Wurf *31.05/*01.06.2023Von: Velvet v.d. Kleinen Chaoten & Vance Sohn v.d. Kleinen ChaotenWS: 3/3", "3/3", "31.05/*01.06.2023"),
|
||||
("Q21-Wurf *21.03.2022Von: Belica gen. Emi v.d. Kleinen Chaoten & Zac gen. Action v.d. Kleinen ChaotenWS: 2/4Notiz:", "2/4", "21.03.2022"),
|
||||
]
|
||||
for para, expected_ws, _ in cases:
|
||||
m = ed.LITTER_RE.search(para)
|
||||
ws = m.group(5).replace(" ", "") if m else None
|
||||
check(f"LITTER_RE matches: {para[:50]}...", ws == expected_ws)
|
||||
|
||||
if SKIP:
|
||||
print("(Skipping live-docx tests: file not found)")
|
||||
else:
|
||||
litters, animals = ed.extract(DOCX)
|
||||
check("93 litters extracted", len(litters) == 93)
|
||||
check("All litters have wsCode", all(l["wsCode"] for l in litters))
|
||||
check("All litters have dob", all(l["dob"] for l in litters))
|
||||
check(">200 named animals", len(animals) >= 200)
|
||||
check(">150 animals with owner", sum(1 for a in animals if a["owner"]) >= 150)
|
||||
check(">20 animals with death date", sum(1 for a in animals if a["deathDate"]) >= 20)
|
||||
# Verify first litter
|
||||
d19 = next((l for l in litters if l["litterId"] == "D19-Wurf"), None)
|
||||
check("D19-Wurf found", d19 is not None)
|
||||
check("D19-Wurf dob correct", d19 and d19["dob"] == "07.04.2019")
|
||||
check("D19-Wurf wsCode = 2/4", d19 and d19["wsCode"] == "2/4")
|
||||
check("D19-Wurf mother contains Xhemile", d19 and "Xhemile" in d19["motherName"])
|
||||
# Verify Eddie in animals
|
||||
eddie = next((a for a in animals if a["name"] == "Eddie" and a["wsCode"] == "2/4"), None)
|
||||
check("Eddie found in D19-Wurf", eddie is not None)
|
||||
check("Eddie gender=male (Zobel* suffix)", eddie and eddie["gender"] == "male")
|
||||
check("Eddie abgabeDate", eddie and eddie["abgabeDate"] == "12.09.2021")
|
||||
# Flash death date
|
||||
flash = next((a for a in animals if a["name"] == "Flash" and a["wsCode"] == "3/3"), None)
|
||||
check("Flash death date extracted", flash and flash["deathDate"] == "16.09.2023")
|
||||
check("Flash death cause extracted", flash and "Tumor" in flash["deathCause"])
|
||||
|
||||
if failed:
|
||||
print(f"\n{failed} test(s) FAILED")
|
||||
sys.exit(1)
|
||||
print("\nALL PASS")
|
||||
Reference in New Issue
Block a user