IMPORT-POLISH: 4 Importer-Fixes nach Re-Import #2

FIX-1 decision-matching: apply_conflict_decisions/apply_dob_remaps
nutzen jetzt canon_pair(name)[0] als Match-Key (Dedup-Identitaet:
call-name ohne Zucht, v.d.<->von den gefaltet). Workaround-Spelling
v.d. in Victoria Welbys Decision bleibt erhalten; beide Formen
matchen jetzt. Kommentar im decision-Eintrag aktualisiert.

FIX-2 specific-wins: _alleles_compatible aendert '? vs x = False'
-> '? vs x = True' (spezifischer Wert gewinnt). C- vs CC, G- vs Gg,
P? vs PP sind kein Konflikt mehr. Echte Wert-Widersprueche (DD vs Dd,
Ee vs ee, PP vs Pp) bleiben Konflikte. Loest Enya, Ella, Zac
automatisch (Konflikte 8->5 erwartet). 2 bestehende Tests angepasst,
7 neue Tests.

FIX-3 parent-FK backfill: nach dem Wurfchronik-Rueckverknuepfungs-
Block iteriert ImportService.RunAsync ueber bereits importierte
Wuerfe mit null Father/MotherId und setzt fehlende FKs wenn das
Elterntier jetzt ladbar ist. Trockenlauf zaehlt, Execute schreibt.
LitterSummary.ParentFksBackfilled + 2 neue C#-Tests (SQLite).

FIX-4 Skarlett-Artefakt: parse_detail() strippt trailing / +YEAR
aus dem Genotyp-Tail (re.sub). Sterbejahr bleibt als death-Date
erhalten -> Skarlett erscheint als reiner Sterbedatum-Konflikt.
2 neue Python-Tests.

Gate: 124/124 C#-Tests, Python test_extract/test_genotype ALL PASS,
has-pending-model-changes = No.

Co-Authored-By: Claude Sonnet 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-06-06 14:53:27 +02:00
parent 5eadd89bb6
commit 0c94cfcbf1
6 changed files with 248 additions and 21 deletions

View File

@@ -81,7 +81,7 @@
{
"name": "Victoria Welby gen. Welby v.d. Kleinen Chaoten",
"dob": "16.01.2023",
"decision": "E-locus = ee[f] (Fuchs). NOTE: this is the mother of animal 'C' (c-29042024) — un-quarantining her links C's second parent. Name kept in the merged record's v.d. spelling: extract.py decision matching uses norm_name (no v.d.<->von den fold) — workaround until the canon_pair matching fix lands.",
"decision": "E-locus = ee[f] (Fuchs). This is the mother of animal 'C' (c-29042024) — un-quarantining her links C's second parent. Name in v.d. spelling (workaround from Re-Import #2); both spellings now match after FIX-1 (canon_pair identity).",
"genotype": "Aa CC D- ee[f] Gg pp Spsp [DP]",
"source": "Julian 2026-06-06 — HUMANQUESTION D4"
}

View File

@@ -158,6 +158,10 @@ def parse_detail(text):
tail = text[dob.end():]
tail = re.sub(r"^\s*/?\+?\s?\d[\d.]*", "", tail) # drop any /+death remnant
tail = tail.lstrip(" ,").strip()
# FIX-4 (Skarlett): strip trailing "/ +YEAR" death-year artifacts leaked from compact
# chart cells (e.g. "… rere / +2018"). The DEATH regex still captures the year from
# the full cell text, so it appears as a death-date conflict — not a genotype conflict.
tail = re.sub(r"\s*/\s*\+\d{4}\s*$", "", tail).strip()
if gt.looks_like_genotype(tail):
geno = tail
return (dob.group(1) if dob else "",
@@ -518,10 +522,11 @@ def _alleles_compatible(a, b):
if a == b:
return True
if a == "?" or b == "?":
return False # unknown vs filled = contradiction (D- vs DD)
return True # specific-wins: unknown allele is compatible with any
# specified value (C- vs CC -> CC; G- vs Gg -> Gg)
(ba, ma), (bb, mb) = _split_allele(a), _split_allele(b)
if ba != bb:
return False # different base allele = real value diff (E vs e)
return False # different base allele = real value diff (E vs e, D vs d)
return ma == "" or mb == "" # same base, modifier present-vs-absent -> presence wins
@@ -887,21 +892,22 @@ def apply_dob_remaps(raw_animals, path):
"""PRE-dedup: a conflict-decision carrying `correctDob` marks a record as a DUPLICATE with a
wrong birthdate — remap that raw record's DOB to correctDob so dedup MERGES it into the
canonical same-named animal (e.g. Chelsea *15.10.2021 -> *02.04.2021). Match =
norm_name(name)+norm_dob(dob). Tolerates a missing/garbled file. Returns the remap count.
canon_pair(name)[0]+norm_dob(dob) (same identity as dedup — strips zucht suffix, folds
v.d.<->von den). Tolerates a missing/garbled file. Returns the remap count.
Must run BEFORE dedup (it changes the dedup identity). (god/HUMANQUESTION D — Dubletten.)"""
remaps = {}
try:
with open(path, encoding="utf-8") as fh:
for r in (json.load(fh).get("resolutions") or []):
if r.get("correctDob"):
remaps[(norm_name(r.get("name", "")), norm_dob(r.get("dob", "")))] = r["correctDob"]
remaps[(canon_pair(r.get("name", ""))[0], norm_dob(r.get("dob", "")))] = r["correctDob"]
except (OSError, ValueError):
return 0
if not remaps:
return 0
n = 0
for a in raw_animals:
new = remaps.get((norm_name(a.get("name", "")), norm_dob(a.get("dob", ""))))
new = remaps.get((canon_pair(a.get("name", ""))[0], norm_dob(a.get("dob", ""))))
if new and a.get("dob") != new:
a["dob"] = new
n += 1
@@ -911,15 +917,18 @@ def apply_dob_remaps(raw_animals, path):
def apply_conflict_decisions(merged, conflicts, path):
"""Consume human conflict resolutions (tools/import/conflict-decisions.json) so the wife's
answers UN-QUARANTINE animals. Schema: {"resolutions":[{name, dob, decision, genotype?,
farbschlag?, source}]}. Match = norm_name(name)+norm_dob(dob) (same identity as dedup). A
matching animal: clear its conflict, mark resolvedByDecision; an explicit `genotype`
(breeder notation) is parsed and becomes authoritative, `farbschlag` overrides too. Tolerates
a missing/empty/garbled file. Returns the number of conflicts resolved. (god/HUMANQUESTION D.)"""
farbschlag?, source}]}. Match = canon_pair(name)[0]+norm_dob(dob) — the same dedup identity
(call-name only, zucht stripped, v.d.<->von den folded). A matching animal: clear its
conflict, mark resolvedByDecision; an explicit `genotype` (breeder notation) is parsed and
becomes authoritative, `farbschlag` overrides too. Tolerates a missing/empty/garbled file.
Returns the number of conflicts resolved. (god/HUMANQUESTION D.)"""
decisions = {}
try:
with open(path, encoding="utf-8") as fh:
for r in (json.load(fh).get("resolutions") or []):
decisions[(norm_name(r.get("name", "")), norm_dob(r.get("dob", "")))] = r
# FIX-1: use dedup identity (call-name only, zucht stripped) so that e.g.
# a decision written as "von den" matches a merged record with "v.d." spelling.
decisions[(canon_pair(r.get("name", ""))[0], norm_dob(r.get("dob", "")))] = r
except (OSError, ValueError):
return 0
if not decisions:
@@ -927,7 +936,7 @@ def apply_conflict_decisions(merged, conflicts, path):
resolved = 0
for a in merged:
d = decisions.get((norm_name(a["name"]), norm_dob(a["dob"])))
d = decisions.get((canon_pair(a["name"])[0], norm_dob(a["dob"])))
if not d:
continue
a["resolvedByDecision"] = True

View File

@@ -102,6 +102,39 @@ check("apply_conflict_decisions returns resolved count", n == 2)
check("missing decisions file tolerated (returns 0)",
e.apply_conflict_decisions([], [], os.path.join(tempfile.gettempdir(), "does-not-exist.json")) == 0)
# FIX-1: decision matching uses canon_pair identity -> 'von den' decision matches 'v.d.' record
dec_vd = os.path.join(tempfile.gettempdir(), "decisions-vd.json")
_json.dump({"resolutions": [
{"name": "Victoria Welby gen. Welby von den Kleinen Chaoten", # written with 'von den'
"dob": "16.01.2023", "decision": "E-locus = ee[f]",
"genotype": "Aa CC D- ee[f] Gg pp Spsp", "source": "test"},
]}, open(dec_vd, "w", encoding="utf-8"))
merged_vd = [
{"id": "vw", "name": "Victoria Welby gen. Welby v.d. Kleinen Chaoten", # record has 'v.d.'
"dob": "16.01.2023", "conflict": True, "farbschlag": "", "death": "",
"genotype": {"mapped8locus": {}, "rawGenotype": "", "unmappedTokens": []}},
]
conflicts_vd = [{"id": "vw"}]
n_vd = e.apply_conflict_decisions(merged_vd, conflicts_vd, dec_vd)
check("FIX-1: 'von den' decision matches 'v.d.' record (canon_pair identity)", n_vd == 1)
check("FIX-1: conflict cleared for v.d. record", merged_vd[0]["conflict"] is False)
# Also verify the workaround spelling (v.d. in decision) matches a 'von den' record
_json.dump({"resolutions": [
{"name": "Victoria Welby gen. Welby v.d. Kleinen Chaoten", # workaround: v.d. in decision
"dob": "16.01.2023", "decision": "E-locus = ee[f]",
"genotype": "Aa CC D- ee[f] Gg pp Spsp", "source": "test"},
]}, open(dec_vd, "w", encoding="utf-8"))
merged_vd2 = [
{"id": "vw2", "name": "Victoria Welby gen. Welby von den Kleinen Chaoten", # record 'von den'
"dob": "16.01.2023", "conflict": True, "farbschlag": "", "death": "",
"genotype": {"mapped8locus": {}, "rawGenotype": "", "unmappedTokens": []}},
]
conflicts_vd2 = [{"id": "vw2"}]
n_vd2 = e.apply_conflict_decisions(merged_vd2, conflicts_vd2, dec_vd)
check("FIX-1: v.d. decision also matches 'von den' record (both spellings match)", n_vd2 == 1)
try: os.remove(dec_vd)
except OSError: pass
# --- correctDob: a wrong-birthdate duplicate is remapped BEFORE dedup so it merges ---
dec2 = os.path.join(tempfile.gettempdir(), "decisions-dob.json")
_json.dump({"resolutions": [
@@ -128,23 +161,52 @@ except OSError: pass
try: os.remove(dec_path)
except OSError: pass
# --- "presence wins" conflict rule (Julian) ---
# present-vs-absent (whole locus or [f] modifier) is NOT a conflict; differing filled values are.
# --- "presence wins" + "specific wins" conflict rules (Julian) ---
# present-vs-absent (whole locus or [f] modifier) is NOT a conflict; differing FILLED values are.
# FIX-2 (specific-wins): unknown allele '?' vs any specified value is also NOT a conflict —
# the specific value wins (C- vs CC -> CC; G- vs Gg -> Gg; P? vs PP -> PP).
check("spsp present vs locus absent -> no conflict",
not e._genotype_conflict([{"Sp": ["sp", "sp"]}, {}]))
check("ee[f] vs ee ([f] modifier present/absent) -> no conflict",
not e._genotype_conflict([{"E": ["e", "e^f"]}, {"E": ["e", "e"]}]))
check("DD vs D- (unknown vs filled) -> conflict",
e._genotype_conflict([{"D": ["D", "D"]}, {"D": ["D", "?"]}]))
# FIX-2: '?' vs specified = specific wins (was: contradiction)
check("FIX-2: DD vs D- (specific wins: DD wins) -> NOT conflict",
not e._genotype_conflict([{"D": ["D", "D"]}, {"D": ["D", "?"]}]))
check("FIX-2: C- vs Cc[h] (specific wins: c^h wins) -> NOT conflict",
not e._genotype_conflict([{"C": ["C", "?"]}, {"C": ["C", "c^h"]}]))
check("FIX-2: C- vs CC (specific wins: CC) -> NOT conflict",
not e._genotype_conflict([{"C": ["C", "?"]}, {"C": ["C", "C"]}]))
check("FIX-2: G- vs Gg (specific wins) -> NOT conflict",
not e._genotype_conflict([{"G": ["G", "?"]}, {"G": ["G", "g"]}]))
check("FIX-2: PP vs P? (specific wins: PP) -> NOT conflict",
not e._genotype_conflict([{"P": ["P", "P"]}, {"P": ["P", "?"]}]))
# Genuine value contradictions (both alleles specified but different) still quarantine
check("Ee vs ee (different base allele) -> conflict",
e._genotype_conflict([{"E": ["E", "e"]}, {"E": ["e", "e"]}]))
check("C- vs Cc[h] -> conflict",
e._genotype_conflict([{"C": ["C", "?"]}, {"C": ["C", "c^h"]}]))
check("c[h] vs c[chm] (different modifiers) -> conflict",
check("DD vs Dd (both specified, D vs d) -> conflict",
e._genotype_conflict([{"D": ["D", "D"]}, {"D": ["D", "d"]}]))
check("PP vs Pp (both specified) -> conflict",
e._genotype_conflict([{"P": ["P", "P"]}, {"P": ["P", "p"]}]))
check("c[h] vs c[chm] (different modifiers, both specified) -> conflict",
not e._alleles_compatible("c^h", "c^chm"))
check("identical genotypes -> no conflict",
not e._genotype_conflict([{"A": ["A", "a"]}, {"A": ["A", "a"]}]))
# --- FIX-4: Skarlett parse artifact — trailing "/ +YEAR" stripped from geno, death captured ---
dob4, death4, geno4 = e.parse_detail("Skarlett,*17.04.2016, aa C- DD ee Gg PP spsp rere / +2018")
check("FIX-4: '/ +YEAR' artifact stripped from geno tail",
geno4 == "aa C- DD ee Gg PP spsp rere")
check("FIX-4: death year still captured from full cell text",
death4 == "2018")
check("FIX-4: DOB still correct",
dob4 == "17.04.2016")
# Without artifact — must be unchanged
dob5, death5, geno5 = e.parse_detail("*01.01.2020, aa C- DD ee Gg PP spsp rere")
check("FIX-4: no artifact -> geno unchanged",
geno5 == "aa C- DD ee Gg PP spsp rere")
check("FIX-4: no artifact -> no spurious death",
death5 == "")
# --- name-bleed guard (a parent name is not a Farbschlag) ---
check("v.d. name rejected", e.looks_like_animal_name("Tennessee von den Kleinen Chaoten"))
check("gen.+v.d. name rejected", e.looks_like_animal_name("Victoria Welby gen. Welby v.d. Kleinen Chaoten"))