"""Fix gender resolution for a rotating first-person POV corpus. Neither existing method works on Yarros, and they fail for opposite structural reasons: * TITLE-FIRST (what Brontë needed) finds almost nothing -- 3 gendered entities per work. Contemporary romance does not say "Miss Sorrengail", it says "Violet". * PRONOUN PROXIMITY is wrong specifically on the people who matter most. Measured against six names whose gender I verified in the text: 3 of 18 WRONG, and the three are Violet, Leah and Landon -- each of them the first-person NARRATOR of the book where they were misgendered. A narrator is "I" in her own book, so her name appears mostly inside the other character's dialogue, surrounded by HIS pronouns. This is the Brontë "Jane called male" pathology, and it is worse here because Yarros rotates POV, so every book has a narrator set up to fail. The signal this corpus actually offers is the POV header: chapters open "Chapter One / Leah / Port of Miami", naming their narrator. So resolve each name's gender from the chapters it does NOT narrate -- where other narrators refer to it in the third person and the pronouns are trustworthy. Refuses to write unless it beats the method it replaces on the verified control, because a fix that is merely different is not a fix. """ from __future__ import annotations import argparse, collections, json, re from pathlib import Path MASC = {"he", "him", "his", "himself"} FEM = {"she", "her", "hers", "herself"} # The POV name sits on its own short line just after the chapter heading. HEAD = re.compile(r"^[ \t]*((?:Chapter|CHAPTER)[ \t]+(?:[A-Za-z-]+|\d+)|Prologue|Epilogue)" r"[ \t]*\.?[ \t]*\n+[ \t]*([A-Z][A-Za-z'’-]{1,18})[ \t]*$", re.M) ap = argparse.ArgumentParser() ap.add_argument("corpus") ap.add_argument("--entities", required=True) ap.add_argument("--out", required=True) ap.add_argument("--window", type=int, default=60, help="chars either side of a mention") ap.add_argument("--min-hits", type=int, default=6) ap.add_argument("--ratio", type=float, default=2.5) ap.add_argument("--control", required=True, help="Name=g,Name=g -- verified in the text") a = ap.parse_args() corpus = Path(a.corpus) man = json.loads((corpus / "manifest.json").read_text()) ents = json.loads(Path(a.entities).read_text()) truth = dict(p.split("=") for p in a.control.split(",")) chapters: dict[str, list[tuple[str | None, str]]] = {} for w in man["works"]: rows = [json.loads(l) for l in (corpus / w["path"]).read_text(encoding="utf-8").splitlines() if l.strip()] out = [] for r in rows: m = HEAD.search(r["text"][:400]) out.append((m.group(2) if m else None, r["text"])) chapters[w["slug"]] = out povs = collections.Counter(p for p, _ in out if p) print(f" {w['slug']:14} {len(out):>3} chapters · POV headers found in " f"{sum(1 for p, _ in out if p):>3} · narrators: {dict(povs.most_common(6))}") def resolve(slug: str, name: str, exclude_own_pov: bool) -> str | None: m = f = 0 for pov, text in chapters[slug]: if exclude_own_pov and pov == name: continue for mt in re.finditer(rf"\b{re.escape(name)}\b", text): ctx = text[max(0, mt.start() - a.window): mt.end() + a.window].lower() for w in re.findall(r"[a-z]+", ctx): if w in MASC: m += 1 elif w in FEM: f += 1 if m + f < a.min_hits: return None if m >= a.ratio * max(f, 1): return "m" if f >= a.ratio * max(m, 1): return "f" return None def score(exclude: bool): ok = wrong = held = 0 detail = [] for slug in chapters: for key, ent in ents[slug]["entities"].items(): s = ent.get("surface") or key if s not in truth: continue g = resolve(slug, s, exclude) t = truth[s] if g == t: ok += 1 elif g is None: held += 1 else: wrong += 1; detail.append(f"{slug}/{s}={g} (truth {t})") return ok, held, wrong, detail base_ok, base_held, base_wrong, base_d = score(False) new_ok, new_held, new_wrong, new_d = score(True) print(f"\n control, WITHOUT excluding own-POV chapters: {base_ok} correct · {base_held} held · {base_wrong} WRONG {base_d}") print(f" control, EXCLUDING own-POV chapters: {new_ok} correct · {new_held} held · {new_wrong} WRONG {new_d}") if new_wrong > base_wrong or (new_wrong == base_wrong and new_ok <= base_ok): raise SystemExit("\n REFUSING to write: excluding own-POV chapters did not beat the " "method it replaces on the verified control. A fix that is merely " "different is not a fix.") applied = 0 for slug in chapters: for key, ent in ents[slug]["entities"].items(): g = resolve(slug, ent.get("surface") or key, True) if g and g != ent.get("gender"): applied += 1 if g: ent["gender"] = g Path(a.out).write_text(json.dumps(ents, indent=1), encoding="utf-8") tot = sum(1 for w in ents.values() for e in w["entities"].values() if e.get("gender")) print(f"\n wrote {a.out}: {applied} genders changed/added · {tot} entities now gendered")