Files
GerbilManager/tools/import/extract.py
Gulum 4aca1d528b EXTRACT band-aware Farbschlag: deep bands (gen>=2) have no colour cell
Julian-confirmed against the real xlsx (Kentucky, Akio Kids): pedigree
blocks have two shapes. EARLY bands (gen 0-1, cols B/E/H) are 5-cell
Name/DOB/Farbschlag/Genotype/Zucht — real Farbschlag present. DEEP bands
(gen>=2, cols K/N/Q...) are 3-cell Name/DOB/Genotype with NO Farbschlag —
the colour is derived from the genotype.

The old extractor grabbed the NEXT block's name or a stray health note
(e.g. "DD-Tumor", "Chronische Ohrenentzündung...") as the deep-band
Farbschlag — source of the D1 'name-in-Farbschlag' conflicts and a big
chunk of farbschlagUnmatched. Fix: gen_of(col)>=2 => never assign a
Farbschlag in that block (Name/DOB/Genotype only); the cell is left for the
next block's name search. Deep-band animals load with empty Farbschlag and
the UI computes 'Errechnet' from the stored genotype (lossless).

Dedup already prefers a real (non-empty) early-band Farbschlag, so an animal
appearing in both an early band (Chesnut -> 'Kohlfuchsschimmel') and deep
bands keeps the real value; Chesnut and Tennessee stay separate (distinct
genotype + name+DOB key).

Result: 248/306 loadable animals now have empty Farbschlag (genotype-
derived), 0 suspicious name/note values remain, 0 Farbschlag-based conflicts
(D1 dissolved). Regression test test_extract.py builds a two-band xlsx and
asserts the deep band yields no Farbschlag + the stray note doesn't bleed.

Extractor-only (no C# change). python test_genotype + test_extract green;
dotnet 118/118. Batches with Kevin's GEN-3f seed for one re-import.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-06 11:56:29 +02:00

897 lines
39 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""FEAT-8b stages 1-2 — extract + dedup the GerbilManager source spreadsheets.
Stage 1: parse the 10 Stammbaum pedigree charts and the Wurfchronik litter
chronicle into normalised animals.json + litters.json, and extract
embedded photos (anchor-mapped to animals).
Stage 2: dedup animals (key = normalise(name)+DOB, corroborated by DOB+genotype)
and emit a German-friendly review-report.md for Julian's wife. No DB load.
Re-runnable per file (later Wurfchronik "Teil2+" / more charts just re-run).
Migration tooling — Python, not product code. Zero third-party deps.
See FEAT-8a-format-spec.md (in Pam's hive workspace) for the format analysis.
"""
import os
import re
import sys
import json
import glob
import shutil
import argparse
import unicodedata
from collections import Counter
import xlsx_util as xu
import genotype as gt
HERE = os.path.dirname(os.path.abspath(__file__))
DEFAULT_STAMMBAEUME = r"C:\Users\gulum\dev\Sttammbäume"
DEFAULT_WURFCHRONIK = r"C:\Users\gulum\dev\Wurfchronik der Kleine Chaoten Teil1.xlsx"
OUT = os.path.join(HERE, "output")
DOB = re.compile(r"\*\s?(\d{1,2}\.\d{1,2}\.(?:\d{4}|\d{2}))")
DEATH = re.compile(r"\+\s?(\d{1,2}\.\d{1,2}\.(?:\d{4}|\d{2})|\d{4})")
# ---------------------------------------------------------------- helpers ----
def gen_of(colnum):
"""Map a column number to a generation band (0=proband ... 5=deepest)."""
if colnum <= 6:
return 0 # E band (proband / "Kids")
if colnum <= 9:
return 1 # H band (parents)
if colnum <= 12:
return 2 # K band (grandparents)
if colnum <= 15:
return 3 # N band (great-grandparents)
if colnum <= 17:
return 4 # Q band (gg-grandparents)
return 5 # R/S band (name-pairs)
def norm_name(name):
if not name:
return ""
n = name.lower()
n = re.sub(r"\[.*?\]", " ", n) # drop [line] tags (Wurfchronik)
n = re.sub(r"\bgen\.\b", " ", n) # "gen." nickname marker
n = re.sub(r"\bv\.\s?d\.\b", " von den ", n)
n = re.sub(r"\b(von der|von den|von|of)\b", " ", n) # cattery/line connectors
n = unicodedata.normalize("NFKD", n)
n = re.sub(r"[^a-z0-9äöüß]", "", n)
return n
def norm_dob(d):
if not d:
return ""
p = d.split(".")
if len(p) == 3 and len(p[2]) == 2:
p[2] = "20" + p[2]
return ".".join(x.zfill(2) if i < 2 else x for i, x in enumerate(p))
def slug(name, dob):
base = norm_name(name) or "unbekannt"
d = norm_dob(dob).replace(".", "")
return (base[:40] + ("-" + d if d else "")) or "unbekannt"
def clean_name(raw):
"""Strip detail/markers from a name cell, keep the human name + [line]."""
n = raw.strip().strip(",").strip()
return n
# --- (name, Zucht) canonicalisation -------------------------------------------
# Julian (FEAT-8 ruling): the [brackets] in Wurfchronik names ARE the Zucht
# (breeding line) and are equivalent to the Stammbaum "of/von <line>" suffix.
# Both fold into one canonical (call-name, zucht) pair; the Zucht acts as a
# dedup DISCRIMINATOR (same name+DOB but different Zucht = different animal).
_BRACKET_ZUCHT = re.compile(r"^(.*?)\s*\[([^\]]+)\]\s*$")
_SUFFIX_ZUCHT = re.compile(
r"^(.+?)\s+(?:of|von\s+den|von\s+der|v\.\s?d\.|von)\s+(.+)$", re.IGNORECASE)
# canonical values are post-norm_zucht (word-final 'n' folded: "kleinechaote")
ZUCHT_ALIASES = {
"zdkc": "kleinechaote", # "Zucht der kleinen Chaoten" (home cattery shorthand)
}
# A Farbschlag value must NOT contain cattery/line connectors (v.d./von/of/gen.) — when it
# does, a parent's NAME has bled into the Farbschlag cell (cross-cell chart read, PEDIGREE-LINK
# bug: e.g. "Victoria Welby gen. Welby v.d. Kleinen Chaoten" became a Farbschlag variant and
# spawned a phantom conflict). Reject such values so they don't pollute farbschlag/conflicts.
_NAME_MARKER = re.compile(r"\bv\.\s?d\.|\bvon\b|\bof\b|\bgen\.", re.IGNORECASE)
def looks_like_animal_name(text):
"""True if a candidate Farbschlag cell actually looks like an animal name (has a
cattery/line connector). Real Farbschläge are short colour words without these."""
return bool(_NAME_MARKER.search(text or ""))
def split_name_zucht(raw):
"""'Luna [ZdkC]' -> ('Luna','ZdkC'); 'Pikachu of Black Forest' ->
('Pikachu','Black Forest'); plain names -> (name, '')."""
n = (raw or "").strip()
m = _BRACKET_ZUCHT.match(n)
if m:
return m.group(1).strip(), m.group(2).strip()
m = _SUFFIX_ZUCHT.match(n)
if m:
return m.group(1).strip(), m.group(2).strip()
return n, ""
def norm_zucht(z):
"""Canonical Zucht key: drops Zucht/von/der fillers, folds declension
('kleinen Chaoten' == 'kleine Chaoten'), resolves known shorthands."""
if not z:
return ""
n = z.lower()
n = re.sub(r"\bv\.\s?d\.\b", " ", n)
n = re.sub(r"\b(zucht|privatzucht|der|die|den|des|dem|von|of)\b", " ", n)
n = re.sub(r"[^a-z0-9äöüß ]", " ", n)
words = [w[:-1] if len(w) > 4 and w.endswith("n") else w for w in n.split()]
key = "".join(words)
return ZUCHT_ALIASES.get(key, key)
def canon_pair(raw):
"""Full raw name -> (normalised call-name, canonical zucht)."""
name, zucht = split_name_zucht(raw)
return norm_name(name), norm_zucht(zucht)
# --------------------------------------------------- Stammbaum extraction ----
def parse_detail(text):
"""From a string that contains *DOB and/or genotype, pull (dob, death, geno_str).
For compact lines ("Name,*DOB[/+death], genotype") the genotype is everything
after the date — we must NOT scan from the first locus-looking letter, or stray
name words ("den", "of") get swallowed as genotype tokens.
"""
dob = DOB.search(text)
death = DEATH.search(text)
geno = ""
if dob:
tail = text[dob.end():]
tail = re.sub(r"^\s*/?\+?\s?\d[\d.]*", "", tail) # drop any /+death remnant
tail = tail.lstrip(" ,").strip()
if gt.looks_like_genotype(tail):
geno = tail
return (dob.group(1) if dob else "",
death.group(1) if death else "",
geno)
def extract_stammbaum(path):
"""Return list of animal dicts for one chart file."""
fname = os.path.basename(path)
z = __import__("zipfile").ZipFile(path)
ss = xu.shared_strings(z)
sheets = xu.sheet_paths(z)
cells = xu.read_cells(z, sheets[0], ss)
fillsex = xu.cell_fill_sex(z, sheets[0]) # box colour -> sex (blue=male, white=female)
# group cells by column for block reconstruction
by_col = {}
for (c, r), t in cells.items():
by_col.setdefault(c, []).append((r, t))
for c in by_col:
by_col[c].sort()
animals = []
used = set()
for (c, r), t in sorted(cells.items()):
if (c, r) in used:
continue
compact = re.match(r"^(.+?),\s*\*", t) # "Name,*DOB, genotype"
is_block_dob = bool(re.match(r"^\*\s?\d", t)) # standalone "*DOB"
if not compact and not is_block_dob:
continue
if compact:
name = clean_name(compact.group(1))
dob, death, geno = parse_detail(t)
farbschlag = ""
breeder = ""
used.add((c, r))
else:
# full block: name above, farbschlag/genotype/breeder below
dob, death, geno0 = parse_detail(t)
name = ""
for rr in range(r - 1, r - 4, -1):
if (c, rr) in cells and not re.match(r"^\*?\s?\d", cells[(c, rr)]) \
and not gt.looks_like_genotype(cells[(c, rr)]):
name = clean_name(cells[(c, rr)])
used.add((c, rr))
break
farbschlag = ""
geno = geno0
breeder = ""
# BAND-AWARE (Julian-confirmed): early bands (gen 0-1, cols B/E/H) are 5-cell blocks
# WITH a Farbschlag cell; deep bands (gen >= 2, cols K/N/Q...) are 3-cell blocks
# (Name/DOB/Genotype) with NO Farbschlag — colour is derived from the genotype. So in
# deep bands we must NOT grab the next block's name or a stray health note as Farbschlag.
deep_band = gen_of(c) >= 2
for rr in range(r + 1, r + 4):
cell = cells.get((c, rr))
if not cell:
continue
if gt.looks_like_genotype(cell):
geno = cell
used.add((c, rr))
elif re.search(r"\b(Zucht|Privatzucht)\b", cell) or cell.startswith("("):
breeder = cell
used.add((c, rr))
elif not deep_band and not farbschlag and not re.match(r"^\*?\s?\d", cell) \
and not looks_like_animal_name(cell):
farbschlag = cell
used.add((c, rr))
used.add((c, r))
g = parse_detail(t) if compact else (dob, death, geno)
genodict = gt.parse(geno)
# Zucht: from the name's [tag]/of-von suffix, else from the breeder line
_, zraw = split_name_zucht(name)
if not zraw and breeder:
zraw = breeder
animals.append({
"id": None, # assigned in dedup
"name": name,
"nameVariants": [],
"dob": dob,
"death": death,
"gender": fillsex.get((c, r)), # box colour: blue=male, white=female
"farbschlag": farbschlag,
"genotype": genodict,
"deaf": genodict.get("deaf"),
"tags": genodict.get("tags", []),
"breeder": breeder,
"zucht": zraw,
"parentRefs": [],
"photos": [],
"sourceFiles": [fname],
"_gen": gen_of(c),
"_col": c,
"_row": r,
"_file": fname,
"_zucht": norm_zucht(zraw),
})
# name-pair cells "X & Y" (deepest generation, names only)
for (c, r), t in cells.items():
if (c, r) in used:
continue
if " & " in t and not DOB.search(t) and len(t) < 90 and gen_of(c) >= 4:
for part in t.split(" & "):
part = clean_name(part)
if part:
_, zraw = split_name_zucht(part)
animals.append({
"id": None, "name": part, "nameVariants": [],
"dob": "", "death": "", "gender": None, "farbschlag": "",
"genotype": gt.parse(""), "deaf": None, "tags": [],
"breeder": "", "zucht": zraw,
"parentRefs": [], "photos": [], "sourceFiles": [fname],
"_gen": gen_of(c), "_col": c, "_row": r, "_file": fname,
"_zucht": norm_zucht(zraw),
})
_reconstruct_parents(animals)
_attach_photos(z, sheets, animals, fname)
return animals
def _reconstruct_parents(animals):
"""Positional: an animal's parents are the bracketing blocks one generation
deeper (father = nearest block above, mother = nearest below). Role guess is
by vertical position (German charts: Vater oben) — flagged for review; the
Wurfchronik is authoritative for matched animals (Stage 3)."""
by_gen = {}
for a in animals:
by_gen.setdefault(a["_gen"], []).append(a)
for g, group in by_gen.items():
nxt = sorted(by_gen.get(g + 1, []), key=lambda a: a["_row"])
if not nxt:
continue
for a in group:
r = a["_row"]
above = [x for x in nxt if x["_row"] <= r]
below = [x for x in nxt if x["_row"] > r]
father = above[-1] if above else None
mother = below[0] if below else None
for parent, role in ((father, "father"), (mother, "mother")):
if parent and parent["name"]:
a["parentRefs"].append({
"name": parent["name"],
"dob": parent["dob"],
"roleGuess": role,
"method": "chart-position",
"confidence": "medium",
})
def _attach_photos(z, sheets, animals, fname):
anchors = [a for a in xu.image_anchors(z)]
if not anchors:
return
by_gen = {}
for a in animals:
by_gen.setdefault(a["_gen"], []).append(a)
media_dir = os.path.join(OUT, "photos")
for i, (sp, col, row, media) in enumerate(anchors):
g = gen_of(col)
cands = by_gen.get(g, [])
if not cands:
# fall back to nearest animal by row across all gens
cands = animals
target = min(cands, key=lambda a: abs(a["_row"] - row)) if cands else None
if not target:
continue
ext = os.path.splitext(media)[1] or ".img"
sl = slug(target["name"], target["dob"])
dest_dir = os.path.join(media_dir, sl)
os.makedirs(dest_dir, exist_ok=True)
rel = f"photos/{sl}/{os.path.basename(media)}"
try:
with z.open(media) as src, open(os.path.join(OUT, rel), "wb") as dst:
shutil.copyfileobj(src, dst)
target["photos"].append(rel)
except KeyError:
pass
# -------------------------------------------------- Wurfchronik extraction ---
def extract_wurfchronik(path):
"""Return list of litter dicts. Parses columns BY HEADER (sheets differ)."""
fname = os.path.basename(path)
z = __import__("zipfile").ZipFile(path)
ss = xu.shared_strings(z)
litters = []
for sp in xu.sheet_paths(z):
cells = xu.read_cells(z, sp, ss)
if not cells:
continue
# build row -> {colnum: text}
rows = {}
for (c, r), t in cells.items():
rows.setdefault(r, {})[c] = t
hdr = xu.header_row(cells) # colnum -> header label
hdr_row = min(r for (_, r) in cells) # the header row number, to skip it
# map header label -> colnum (fuzzy by keyword)
def find(*keys):
for c, lbl in hdr.items():
low = lbl.lower()
if any(k in low for k in keys):
return c
return None
col_id = find("wurfbuchstabe", "buchstabe")
col_date = find("geburtsdatum", "datum")
col_dam = find("mutter")
col_sire = find("vater")
col_ws = find("ws", "wurfstärke", "wurfstaerke")
col_breakdown = find("männchen", "maennchen", "weibchen")
col_zn = find("zuchtnummer")
col_note = find("bemerkung", "anmerkung")
# Julian (authoritative): Tabelle1 has an UNLABELED numeric column between
# Vater and Wurfstärke = "Überlebende bis zum Abgabedatum" (survivors to
# go-home). Tabelle2 dropped it. Header-based find() cannot see it, so
# detect it positionally.
col_survived = None
if col_sire and col_ws and col_ws - col_sire > 1:
mapped = {col_id, col_date, col_dam, col_sire, col_ws,
col_breakdown, col_zn, col_note}
for c in range(col_sire + 1, col_ws):
if c not in mapped:
col_survived = c
break
sheet_name = os.path.basename(sp)
for r in sorted(rows):
if r == hdr_row: # skip the header row itself
continue
row = rows[r]
# skip empty-id + "Jahr YYYY" section rows
txt_b = row.get(col_date, "") if col_date else ""
if not row.get(col_id):
continue
if "jahr" in " ".join(row.values()).lower() and not DOB.search(txt_b):
continue
dob = DOB.search(txt_b)
lid = row.get(col_id, "")
datestr = dob.group(1) if dob else ""
dam_raw = row.get(col_dam, "") if col_dam else ""
sire_raw = row.get(col_sire, "") if col_sire else ""
_, dam_zucht = split_name_zucht(dam_raw)
# '&' = multiple sires possible (litter with uncertain/dual sire)
sire_names = [s.strip() for s in sire_raw.split("&") if s.strip()]
_, sire_zucht = split_name_zucht(sire_names[0] if sire_names else "")
# Numeric layout is read VALUE-ADAPTIVELY per row: a handful of
# Tabelle2 rows insert an extra numeric column (Überlebende, T1
# order E,F) before WS and shift the breakdown right of its header.
warnings = []
scan_end = (col_note or (col_sire or 4) + 6) + 1
bd_col = None
for c in range((col_sire or 4) + 1, scan_end):
if re.fullmatch(r"\d+(?:[,;]\d+){1,3}", row.get(c, "")):
bd_col = c
break
if bd_col and col_breakdown and bd_col != col_breakdown:
warnings.append(
f"Spaltenschema-Abweichung: Geschlechter-Aufschlüsselung in "
f"Spalte {xu.num_to_col(bd_col)} statt "
f"{xu.num_to_col(col_breakdown)} gefunden")
# single numeric cells between Vater and the breakdown: E (Überlebende)
# and/or F (Wurfstärke), in T1 order
singles = []
for c in range((col_sire or 4) + 1, bd_col or scan_end):
v = row.get(c, "")
if re.fullmatch(r"\d+", v):
singles.append(int(v))
if len(singles) >= 2:
survived, total_born = singles[0], singles[1]
if not col_survived:
warnings.append(
"Spaltenschema-Abweichung: zusätzliche Zahlenspalte als "
"„Überlebende“ interpretiert (bitte prüfen)")
elif len(singles) == 1:
total_born = singles[0]
survived = None
else:
total_born = _to_int(row.get(col_ws)) if col_ws else None
survived = _to_int(row.get(col_survived)) if col_survived else None
# breakdown "2,0,2,0" = Männchen, Weibchen, Totgeburt, später
# verstorben ('s' = died after birth, before Abgabe — Julian).
bd = row.get(bd_col, "") if bd_col else ""
m = re.findall(r"\d+", bd)
males = females = stillborn = died_later = None
if m:
vals = [int(x) for x in m] + [None] * 4
males, females, stillborn, died_later = vals[:4]
# Validation (Julian): Überlebende E should equal F TG s.
# Mismatch = data-quality signal for the review report, NOT a blocker.
if survived is not None and total_born is not None:
expected = total_born - (stillborn or 0) - (died_later or 0)
if survived != expected:
warnings.append(
f"Überlebende ({survived}) ≠ Wurfstärke ({total_born}) "
f" Totgeburten ({stillborn or 0}) später verstorben "
f"({died_later or 0}) = {expected}")
litters.append({
"id": f"{sheet_name.replace('.xml','')}-{lid}-{norm_dob(datestr)}",
"litterId": lid,
"date": datestr,
"damName": dam_raw,
"damZucht": dam_zucht,
"sireName": sire_raw,
"sireNames": sire_names,
"sireZucht": sire_zucht,
"totalBorn": total_born,
"survivedToGoHome": survived,
"males": males,
"females": females,
"stillborn": stillborn,
"diedLater": died_later,
"breakdownRaw": bd,
"zuchtnummer": row.get(col_zn, "") if col_zn else "",
"note": row.get(col_note, "") if col_note else "",
"warnings": warnings,
"sourceFile": fname,
"sheet": sheet_name,
"row": r,
})
return litters
def _to_int(s):
if not s:
return None
m = re.search(r"\d+", s)
return int(m.group()) if m else None
# ------------------------------------------------------------- stage 2: dedup
def _geno_key(genodict):
"""Canonical, order-independent key of a genotype's mapped loci — used for conflict
detection so Uw==G (and allele ordering) no longer count as a conflict."""
m = genodict.get("mapped8locus", {})
return "|".join(f"{locus}:{','.join(sorted(m[locus]))}" for locus in sorted(m))
def dedup(animals):
"""Merge by normalise(call-name)+DOB, with the canonical Zucht as
DISCRIMINATOR (Julian: same name+DOB+Zucht = same animal; different Zucht =
different animal). Returns (merged, conflicts, orphans, zucht_splits)."""
groups = {}
orphans = []
for a in animals:
call, _ = split_name_zucht(a["name"])
key = (norm_name(call), norm_dob(a["dob"]))
if not key[0] or not key[1]:
orphans.append(a)
# orphans still get a stable id but are not merged
key = ("__orphan__", id(a))
groups.setdefault(key, []).append(a)
# split groups whose members carry DIFFERENT (non-empty) canonical Zuchten;
# members without a Zucht merge into the group only if it is unambiguous.
final_groups = []
zucht_splits = []
for key, grp in groups.items():
by_zucht = {}
for a in grp:
by_zucht.setdefault(a.get("_zucht", ""), []).append(a)
nonempty = sorted(z for z in by_zucht if z)
if len(nonempty) <= 1:
final_groups.append(grp)
continue
# genuine split: same call-name+DOB, different Zucht
for z in nonempty:
final_groups.append(by_zucht[z])
if "" in by_zucht:
# Zucht-less mentions cannot be attributed -> own (flagged) entry
final_groups.append(by_zucht[""])
zucht_splits.append({
"name": grp[0]["name"],
"dob": norm_dob(grp[0]["dob"]),
"zuechte": sorted(set(a["zucht"] for a in grp if a.get("zucht"))),
"files": sorted(set(f for a in grp for f in a["sourceFiles"])),
})
merged = []
conflicts = []
for grp in final_groups:
base = dict(grp[0])
variants = set([base["name"]])
files = set(base["sourceFiles"])
photos = list(base["photos"])
parent_refs = list(base["parentRefs"])
genos = set()
geno_keys = set() # GEN-3b: conflict on NORMALIZED genotype (Uw==G) not raw text
farb = set()
deaths = set()
deaf_seen = set()
tags_set = set()
genders = []
for a in grp:
variants.add(a["name"])
files.update(a["sourceFiles"])
photos.extend(a["photos"])
parent_refs.extend(a["parentRefs"])
if a.get("gender"):
genders.append(a["gender"])
if a["genotype"]["mapped8locus"]:
genos.add(a["genotype"]["rawGenotype"])
geno_keys.add(_geno_key(a["genotype"]))
if a["farbschlag"]:
farb.add(a["farbschlag"])
if a["death"]:
deaths.add(norm_dob(a["death"]))
if a.get("deaf") is not None:
deaf_seen.add(a["deaf"])
tags_set.update(a.get("tags", []))
# pick the richest genotype (most mapped loci, then longest raw)
best = max((a["genotype"] for a in grp),
key=lambda gd: (len(gd["mapped8locus"]), len(gd["rawGenotype"])))
out = {
"id": slug(base["name"], base["dob"]),
"name": base["name"],
"nameVariants": sorted(v for v in variants if v),
"dob": norm_dob(base["dob"]),
"death": sorted(deaths)[0] if deaths else "",
# box-colour sex (blue=male, white=female): majority across mentions, else None.
"gender": Counter(genders).most_common(1)[0][0] if genders else None,
"farbschlag": sorted(farb)[0] if farb else "",
"farbschlagVariants": sorted(farb),
"genotype": best,
"breeder": next((a["breeder"] for a in grp if a["breeder"]), ""),
"zucht": next((a["zucht"] for a in grp if a.get("zucht")), ""),
"zuchtCanon": next((a["_zucht"] for a in grp if a.get("_zucht")), ""),
"parentRefs": _dedup_parentrefs(parent_refs),
"photos": sorted(set(photos)),
"sourceFiles": sorted(files),
"mentions": len(grp),
# GEN-3b: hearing/deaf phenotype flag (deaf wins if any mention says so) + tags.
"deaf": (True if True in deaf_seen else (False if False in deaf_seen else None)),
"tags": sorted(tags_set),
# FEAT-8c: machine-readable quarantine marker so the API loader can skip
# conflicting records without parsing the German review report.
"conflict": False,
}
merged.append(out)
# conflict: same animal, disagreeing NORMALIZED genotype (Uw==G) or farbschlag or death
if len(geno_keys) > 1 or len(farb) > 1 or len(deaths) > 1:
out["conflict"] = True
conflicts.append({
"id": out["id"], "name": base["name"], "dob": out["dob"],
"genotypes": sorted(genos), "farbschlaege": sorted(farb),
"deaths": sorted(deaths), "files": sorted(files),
})
merged.sort(key=lambda a: (a["dob"], a["name"]))
# ids must stay unique even when a Zucht split duplicates name+DOB slugs
seen_ids = {}
for a in merged:
if a["id"] in seen_ids:
seen_ids[a["id"]] += 1
a["id"] = f"{a['id']}-{seen_ids[a['id']]}"
else:
seen_ids[a["id"]] = 1
return merged, conflicts, orphans, zucht_splits
# ------------------------------------------- animal <-> litter matching ------
def match_litters(merged, litters):
"""Attach each animal to its Wurfchronik litter (Pam-validated build order:
litters are canonical, animals match onto them via DOB + (Vater, Mutter)).
Sets a['litterRef']; returns match statistics for the report."""
by_date = {}
for l in litters:
d = norm_dob(l["date"])
if d:
by_date.setdefault(d, []).append(l)
stats = {"parents": 0, "dateOnly": 0, "ambiguous": 0}
for a in merged:
if not a["dob"]:
continue
cands = by_date.get(a["dob"])
if not cands:
continue
a_parents = set()
for ref in a["parentRefs"]:
cn, _ = canon_pair(ref["name"])
if cn:
a_parents.add(cn)
def score(l):
s = 0
for nm in [l["damName"]] + l.get("sireNames", []):
cn, _ = canon_pair(nm)
if cn and cn in a_parents:
s += 1
return s
scored = sorted(((score(l), l["id"]) for l in cands), reverse=True)
best_score, best_id = scored[0]
if best_score > 0 and (len(scored) == 1 or scored[1][0] < best_score):
a["litterRef"] = {"litterId": best_id,
"method": "geburtsdatum+eltern", "confidence": "hoch"}
stats["parents"] += 1
elif len(cands) == 1:
a["litterRef"] = {"litterId": cands[0]["id"],
"method": "nur-geburtsdatum", "confidence": "niedrig"}
stats["dateOnly"] += 1
else:
a["litterRef"] = {"litterId": None, "method": "mehrdeutig",
"candidates": [l["id"] for l in cands]}
stats["ambiguous"] += 1
return stats
def _dedup_parentrefs(refs):
seen = {}
for r in refs:
k = (norm_name(r["name"]), r["roleGuess"])
if k not in seen:
seen[k] = r
return list(seen.values())
# ------------------------------------------------------------------ reporting
def write_report(merged, conflicts, orphans, raw_count, litters, photo_count,
zucht_splits, match_stats):
multi = [a for a in merged if a["mentions"] > 1]
with_dob = [a for a in merged if a["dob"]]
val_warn = [l for l in litters if any("" in w for w in l["warnings"])]
schema_warn = [l for l in litters if any("Abweichung" in w for w in l["warnings"])]
matched = match_stats["parents"] + match_stats["dateOnly"]
L = []
L.append("# FEAT-8b — Import-Vorschau & Prüfbericht (Stammbäume + Wurfchronik)\n")
L.append("_Automatisch erzeugt von `tools/import/extract.py` — **noch nichts in die Datenbank geladen.** "
"Bitte prüfen, bevor importiert wird._\n")
L.append("## Überblick\n")
L.append(f"- Rohe Tier-Einträge aus den Stammbäumen: **{raw_count}**")
L.append(f"- Nach Zusammenführung (eindeutige Tiere): **{len(merged)}**")
L.append(f" - davon mit Geburtsdatum: {len(with_dob)}")
L.append(f" - in mehreren Dateien gefunden (Dubletten zusammengeführt): {len(multi)}")
L.append(f"- Konflikte zur Klärung: **{len(conflicts)}**")
L.append(f"- Mehrdeutige / unvollständige Einträge (ohne Name+Datum): **{len(orphans)}**")
L.append(f"- Fotos zugeordnet: **{photo_count}**")
L.append(f"- Würfe aus der Wurfchronik: **{len(litters)}**")
L.append(f" - Tiere mit Wurf verknüpft: **{matched}** "
f"(davon über Geburtsdatum **und** Eltern: {match_stats['parents']}, "
f"nur über Geburtsdatum: {match_stats['dateOnly']}; "
f"mehrdeutig: {match_stats['ambiguous']})")
L.append(f" - Würfe mit Datenqualitäts-Hinweisen: {len(val_warn)} "
f"(+ {len(schema_warn)} Zeilen mit abweichendem Spaltenschema)\n")
L.append("## Zusammenführungs-Schlüssel\n")
L.append("Tiere wurden zusammengeführt über **normalisierter Rufname + Geburtsdatum**, "
"mit der **Zucht als Unterscheidungsmerkmal** (Julians Regel: die `[Klammern]` "
"in der Wurfchronik und das `of/von <Linie>`-Suffix der Stammbäume bezeichnen "
"beide die Zucht und werden zusammengeführt — z. B. `[ZdkC]` ≙ "
"`von den Kleinen Chaoten`). Namensvarianten (z. B. `v.d.` ↔ `von den`, "
"`gen.`-Spitznamen) werden als `nameVariants` erhalten.\n")
if zucht_splits:
L.append("### Gleicher Name + Geburtsdatum, aber unterschiedliche Zucht "
"(NICHT zusammengeführt — bitte prüfen)\n")
L.append("| Tier | Geburtsdatum | Zuchten | Dateien |")
L.append("|---|---|---|---|")
for s in zucht_splits[:50]:
L.append("| {} | {} | {} | {} |".format(
split_name_zucht(s["name"])[0], s["dob"],
" // ".join(s["zuechte"]),
", ".join(os.path.splitext(f)[0] for f in s["files"])))
L.append("")
L.append("## ⚠️ Konflikte (bitte prüfen)\n")
if conflicts:
L.append("Gleiches Tier (Name+Datum), aber widersprüchliche Angaben in verschiedenen Dateien:\n")
L.append("| Tier | Geburtsdatum | abweichende Genotypen | abweichende Farbschläge | Sterbedaten | Dateien |")
L.append("|---|---|---|---|---|---|")
for c in conflicts[:200]:
L.append("| {} | {} | {} | {} | {} | {} |".format(
c["name"], c["dob"],
" // ".join(c["genotypes"]) or "",
" // ".join(c["farbschlaege"]) or "",
" // ".join(c["deaths"]) or "",
", ".join(os.path.splitext(f)[0] for f in c["files"])))
else:
L.append("_Keine._\n")
L.append("\n## Mehrdeutige / unvollständige Einträge\n")
L.append(f"{len(orphans)} Einträge ohne sichere Name+Datum-Kombination "
"(z. B. `Name1 & Name2`-Paarzellen der tiefsten Generation, oder Zellen ohne Datum). "
"Diese werden NICHT automatisch zusammengeführt.\n")
sample = [o for o in orphans if o["name"]][:40]
for o in sample:
L.append(f"- {o['name']} · {o.get('_file','')}")
# orphan -> likely same-named full record (soft hint, not auto-merged)
from collections import Counter
name_index = {}
for a in merged:
if a["dob"]:
name_index.setdefault(norm_name(a["name"]), []).append(a)
matchable = []
for o in orphans:
if not o["name"]:
continue
cands = name_index.get(norm_name(o["name"]))
if cands:
matchable.append((o, cands))
L.append("\n## Wahrscheinliche Zuordnungen unvollständiger Einträge\n")
L.append(f"{len(matchable)} namenlose/datenlose Einträge tragen denselben Namen wie ein "
"vollständiges Tier — vermutlich dasselbe Tier (zur Bestätigung):\n")
for o, cands in matchable[:60]:
opts = "; ".join(f"{c['name']} (*{c['dob']})" for c in cands[:3])
L.append(f"- „{o['name']}“ → {opts}")
# unmapped-token summary (for Kevin / GEN-2 + the wife)
tok = Counter()
for a in merged:
for t in a["genotype"]["unmappedTokens"]:
tok[t] += 1
L.append("\n## Nicht ins 8-Loci-Modell abgebildete Tokens (verbatim erhalten)\n")
L.append("Diese Tokens stehen weiter in `rawGenotype`/`unmappedTokens` — Entscheidung "
"(Modell erweitern vs. als Notiz) liegt bei Julian/Kevin:\n")
L.append("| Token | Vorkommen | Bedeutung (Vermutung) |")
L.append("|---|---|---|")
hint = {"Uwuw[d]": "9. Locus Uw (nicht im Modell)", "UwUw": "9. Locus Uw",
"[WFNZ]": "Marker", "[DP]": "Marker (Dunkelpigment?)", "DP": "Marker",
"WP": "Marker", "[WP]": "Marker", "C(C)": "Schreibweise (C trägt c)",
"chmchm": "Schreibweise (c[chm]c[chm])"}
for t, n in tok.most_common(25):
L.append(f"| `{t}` | {n} | {hint.get(t, '?')} |")
# Wurfchronik data quality (Julian: E sollte = F TG s sein)
L.append("\n## Wurfchronik — Datenqualitäts-Hinweise\n")
L.append("Julians Spaltenregel: **Überlebende bis Abgabe (E) = Wurfstärke (F) "
"Totgeburten (TG) später verstorben (s)**. Bei diesen Würfen geht die "
"Rechnung nicht auf — kein Import-Hindernis, aber ein Hinweis auf "
"Tippfehler oder fehlende Einträge:\n")
if val_warn:
L.append("| Wurf | Datum | Mutter × Vater | Hinweis |")
L.append("|---|---|---|---|")
for l in val_warn[:120]:
L.append("| {} | {} | {} × {} | {} |".format(
l["litterId"], l["date"], l["damName"], l["sireName"],
"; ".join(w for w in l["warnings"] if "" in w)))
if len(val_warn) > 120:
L.append(f"\n… und {len(val_warn) - 120} weitere (siehe `litters.json`).")
else:
L.append("_Keine — alle Würfe sind rechnerisch konsistent._")
L.append("\n### Zeilen mit abweichendem Spaltenschema (automatisch interpretiert)\n")
L.append(f"{len(schema_warn)} Zeilen (überwiegend Tabelle2 ab 2014) tragen eine "
"zusätzliche Zahlenspalte vor der Wurfstärke bzw. eine verschobene "
"Geschlechter-Aufschlüsselung. Sie wurden nach dem Muster von Tabelle1 "
"gelesen (**Überlebende, Wurfstärke, Aufschlüsselung**) — bei "
f"{sum(1 for l in schema_warn if not any('' in w for w in l['warnings']))} "
"davon geht die Rechnung E = F TG s damit exakt auf, was die Lesart "
"bestätigt. Alle betroffenen Zeilen sind in `litters.json` mit "
"`warnings` markiert. Beispiele:\n")
for l in schema_warn[:8]:
L.append(f"- Wurf {l['litterId']} ({l['date']}): E={l['survivedToGoHome']}, "
f"F={l['totalBorn']}, Aufschlüsselung `{l['breakdownRaw']}`")
L.append("\n## Hinweise für den Import (Stufe 3, später)\n")
L.append("- **Wurfchronik = Quelle der Würfe** (Datum, Wurfstärke, Eltern, Zuchtnummer); "
"**Stammbäume = Abstammung + Genotyp + Fotos**. Verknüpfung über Geburtsdatum + Elternnamen.")
L.append("- Eltern-Verknüpfungen (`parentRefs`) stammen aus der **Position im Stammbaum** "
"(Vater oben / Mutter unten, mittlere Konfidenz) — die Wurfchronik korrigiert dies maßgeblich.")
L.append("- Genotyp: `mapped8locus` (A C D E G P Sp Re), `rawGenotype` (wortgetreu), "
"`unmappedTokens` (z. B. `Uw`, `Sls`, `Dea`, Marker wie `WFNZ/WP/DP`) — **nichts geht verloren**.")
L.append("- `-` (unbekanntes zweites Allel) → `?` (Platzhalter; Annahme, bitte bestätigen).")
with open(os.path.join(OUT, "review-report.md"), "w", encoding="utf-8") as f:
f.write("\n".join(L) + "\n")
# ------------------------------------------------------------------------ main
def main():
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
ap = argparse.ArgumentParser(description="FEAT-8b extractor (stages 1-2)")
ap.add_argument("--stammbaeume", default=DEFAULT_STAMMBAEUME)
ap.add_argument("--wurfchronik", default=DEFAULT_WURFCHRONIK)
args = ap.parse_args()
if os.path.isdir(OUT):
# keep the report stable across reruns but refresh data/photos
for sub in ("photos",):
p = os.path.join(OUT, sub)
if os.path.isdir(p):
shutil.rmtree(p)
os.makedirs(OUT, exist_ok=True)
raw_animals = []
# Skip Excel lock/owner files ("~$...") that appear while a workbook is open.
files = sorted(f for f in glob.glob(os.path.join(args.stammbaeume, "*.xlsx"))
if not os.path.basename(f).startswith("~$"))
print(f"Stammbaum-Dateien: {len(files)}")
for path in files:
got = extract_stammbaum(path)
print(f" {len(got):4d} {os.path.basename(path)}")
raw_animals.extend(got)
litters = []
if os.path.isfile(args.wurfchronik):
litters = extract_wurfchronik(args.wurfchronik)
print(f"Wurfchronik: {len(litters)} Würfe")
merged, conflicts, orphans, zucht_splits = dedup(raw_animals)
match_stats = match_litters(merged, litters)
photo_count = sum(len(a["photos"]) for a in merged)
# strip private (_) fields from the JSON output
def clean(a):
return {k: v for k, v in a.items() if not k.startswith("_")}
with open(os.path.join(OUT, "animals.json"), "w", encoding="utf-8") as f:
json.dump([clean(a) for a in merged], f, ensure_ascii=False, indent=2)
with open(os.path.join(OUT, "litters.json"), "w", encoding="utf-8") as f:
json.dump(litters, f, ensure_ascii=False, indent=2)
write_report(merged, conflicts, orphans, len(raw_animals), litters, photo_count,
zucht_splits, match_stats)
print(f"\nRoh: {len(raw_animals)} → eindeutig: {len(merged)} "
f"| Konflikte: {len(conflicts)} | Zucht-Splits: {len(zucht_splits)} "
f"| Orphans: {len(orphans)} | Fotos: {photo_count}")
print(f"Wurf-Verknüpfung: {match_stats['parents']} (Datum+Eltern), "
f"{match_stats['dateOnly']} (nur Datum), {match_stats['ambiguous']} mehrdeutig "
f"| Wurf-Warnungen: {sum(1 for l in litters if l['warnings'])}")
print(f"Ausgabe in {OUT}")
if __name__ == "__main__":
main()