Files
esh-pfi-infrastructure/scripts/hemingway-corpus/gender_by_proximity.py
T

185 lines
7.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""D2b for BabyHemingway: resolve entity gender by MAJORITY VOTE over nearby pronouns.
The honorific-and-local-window resolver inherited from the Bronte/Yarros line fails badly
here. Measured on this corpus before writing a line of replacement: **397 male, 20 female**
across 1,102 entity records, with Catherine Barkley, Brett Ashley, Pilar, Maria, Marita and
Mary all held neutral and `Helen` and `Audrey` resolved outright WRONG. A corpus containing
those characters does not have twenty women in it.
Why it fails is the same mechanism Yarros exposed from the other side: Hemingway's women
appear mostly inside male characters' scenes, so the pronouns nearest their names are
predominantly `he`. Yarros solved its version with the POV chapter header; Hemingway's
editions have no such header, so that fix does not transfer and a different signal is needed.
⭐ THE SIGNAL THAT WORKS IS VOLUME. A major character is named hundreds of times, so instead
of trusting the nearest pronoun in one window, every occurrence votes and the majority wins.
A single window is dominated by whoever else is in the scene; three hundred windows are
dominated by the person being written about.
⚠ THE INSTRUMENT REFUSES TO WRITE UNLESS IT BEATS WHAT IT REPLACES, scored against a
hand-verified control list. That is the same guard `pov_gender.py` carried, and it is the
only reason to believe a replacement is an improvement rather than a different set of errors.
"""
from __future__ import annotations
import argparse, json, re, sys
from collections import Counter
from pathlib import Path
MALE = {"he", "him", "his", "himself"}
FEMALE = {"she", "her", "hers", "herself"}
WINDOW = 12 # words either side of the mention
MIN_VOTES = 6 # below this the evidence is too thin to overrule a hold
MARGIN = 0.60 # winning share required, else HELD neutral
def load_text(corpus: Path) -> str:
out = []
for f in sorted((corpus / "works").glob("*.jsonl")):
for line in f.read_text(encoding="utf-8").splitlines():
out.append(json.loads(line)["text"])
return "\n".join(out)
def tally_all(text: str, surfaces: set) -> dict:
"""One pass over the corpus for EVERY surface at once.
The obvious shape -- rescan the text once per surface -- is 1,102 surfaces x 995,000
words and does not finish in any useful time. Tokenise once, walk once, and carry running
prefix counts of male and female pronouns so a window costs two subtractions instead of a
25-word inner loop.
"""
words = [w.lower() for w in re.findall(r"[A-Za-z'’]+", text)]
n = len(words)
pm = [0] * (n + 1)
pf = [0] * (n + 1)
for i, w in enumerate(words):
pm[i + 1] = pm[i] + (1 if w in MALE else 0)
pf[i + 1] = pf[i] + (1 if w in FEMALE else 0)
want = {s.lower(): s for s in surfaces}
out = {s: [0, 0] for s in surfaces}
for i, w in enumerate(words):
s = want.get(w)
if s is None:
continue
lo, hi = max(0, i - WINDOW), min(n, i + WINDOW + 1)
out[s][0] += pm[hi] - pm[lo]
out[s][1] += pf[hi] - pf[lo]
return {k: (v[0], v[1]) for k, v in out.items()}
def decide(m: int, f: int, base_m: float = 0.5):
"""Score a name's local pronoun mix AGAINST THE CORPUS BASE RATE, not against 50:50.
⚠ MEASURED, and it is why the first version of this was refused by its own gate: a raw
majority vote scored 18 correct but FIVE wrong against the incumbent's one, and every
error was female-read-as-male -- Pilar m=426 f=243, Brett m=249 f=137. Both are strongly
female-associated; they merely appear in a corpus where male pronouns outnumber female
ones several times over, so a bare majority is dominated by the background rate rather
than by the character.
The correction is to ask whether a name's neighbourhood is male-heavy RELATIVE TO the
corpus, which is what `base_m` supplies. A hold stays the safe outcome: the gate counts a
wrong answer as worse than no answer, because rename can leave a held entity neutral but
cannot undo a man's name given to a woman.
"""
tot = m + f
if tot < MIN_VOTES:
return None
base_f = 1.0 - base_m
# odds of the observed mix under each hypothesis, expressed as a share after dividing
# out the background. lift_m > lift_f means male-heavy beyond what the corpus explains.
lift_m = (m / tot) / base_m if base_m else 0.0
lift_f = (f / tot) / base_f if base_f else 0.0
s = lift_m + lift_f
if not s:
return None
if lift_m / s >= MARGIN:
return "m"
if lift_f / s >= MARGIN:
return "f"
return None
def score(resolved: dict, truth: dict) -> tuple[int, int, int]:
ok = bad = held = 0
for name, want in truth.items():
got = resolved.get(name, "__absent__")
if got == "__absent__":
continue
if got is None:
held += 1
elif got == want:
ok += 1
else:
bad += 1
return ok, held, bad
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("corpus")
ap.add_argument("--entities", required=True)
ap.add_argument("--out", required=True)
ap.add_argument("--control", required=True,
help="Name=g,Name=g ... hand-verified, the gate this must beat")
a = ap.parse_args()
truth = {}
for item in a.control.split(","):
n, _, g = item.partition("=")
truth[n.strip()] = g.strip()
ents = json.loads(Path(a.entities).read_text(encoding="utf-8"))
text = load_text(Path(a.corpus))
incumbent = {}
for blob in ents.values():
for v in blob["entities"].values():
incumbent.setdefault(v["surface"], v.get("gender"))
surfaces = {v["surface"] for blob in ents.values() for v in blob["entities"].values()}
tallies = tally_all(text, surfaces)
toks = [w.lower() for w in re.findall(r"[A-Za-z'’]+", text)]
gm = sum(1 for w in toks if w in MALE)
gf = sum(1 for w in toks if w in FEMALE)
base_m = gm / (gm + gf)
print(f" corpus pronoun base rate: male {gm:,} / female {gf:,} -> base_m {base_m:.3f}")
resolved, detail = {}, {}
for s in sorted(surfaces):
m, f = tallies[s]
g = decide(m, f, base_m)
resolved[s] = g
detail[s] = {"gender": g, "male_votes": m, "female_votes": f}
ok_i, held_i, bad_i = score(incumbent, truth)
ok_n, held_n, bad_n = score(resolved, truth)
print(f" incumbent (honorific/window): correct {ok_i} held {held_i} WRONG {bad_i}")
print(f" proximity majority vote : correct {ok_n} held {held_n} WRONG {bad_n}")
for n, want in truth.items():
if resolved.get(n) not in (want, None):
print(f" ⚠ still wrong: {n} want={want} got={resolved.get(n)} "
f"(m={detail[n]['male_votes']} f={detail[n]['female_votes']})")
if bad_n > bad_i or (bad_n == bad_i and ok_n <= ok_i):
print(" REFUSING to write: does not beat the incumbent", file=sys.stderr)
return 1
counts = Counter(v["gender"] for v in detail.values())
print(f" gender distribution: {dict(counts)}")
out = json.loads(Path(a.entities).read_text(encoding="utf-8"))
changed = 0
for blob in out.values():
for v in blob["entities"].values():
g = resolved.get(v["surface"])
if g != v.get("gender"):
changed += 1
v["gender"] = g
Path(a.out).write_text(json.dumps(out, indent=2, ensure_ascii=False), encoding="utf-8")
print(f" wrote {a.out} ({changed} gender fields changed)")
return 0
if __name__ == "__main__":
raise SystemExit(main())