BabyHemingway D2+D3: entities, base-rate gender resolver, rename preset, leak gate passes

This commit is contained in:
Vuong Hoang
2026-09-16 07:55:34 -07:00
parent 9598d0b4a7
commit 03b4a3f62c
5 changed files with 329 additions and 1 deletions
@@ -214,6 +214,44 @@ def split_by_contents(text: str):
return units or None
# ⚠⚠ PUBLISHER BACK MATTER RIDES INSIDE THE LAST UNIT, in 8 of 10 works -- the identical
# defect the Yarros build hit, and it is worth naming again because nothing about the source
# changed to cause it: a splitter cuts on headings, and nothing follows the last one, so the
# "About the Author" block lands inside the final chapter. Here it carries ERNEST HEMINGWAY'S
# OWN NAME 95 times across 7 works -- "Ernest Hemingway was one of America's foremost
# journalists... died in 1961" -- which is precisely the leak the rename pipeline and its gate
# exist to prevent, sitting in the training text before either one runs.
# Bounded: last unit only, and refuses if it would take more than 2% of the corpus.
# ⚠ The EDITOR'S apparatus counts as back matter too, and it is the bigger leak. Stripping
# only the publisher block left 18 "Hemingway" mentions in `true-at-first-light` -- all of
# them inside a CAST OF CHARACTERS and SWAHILI GLOSSARY written by Patrick Hemingway
# ("Mary Ernest Hemingway's fourth and last wife", "Ngui Hemingway's gun bearer"). That is
# an editor describing the author's real household, not the author's prose, and it names him
# directly. The markers are matched in FILE ORDER and the earliest one wins, so the whole
# apparatus goes rather than just its last section.
BACKMATTER = re.compile(
r"^[ \t]*(THE END|About the Author|ABOUT THE AUTHOR|About the Publisher|"
r"CAST OF CHARACTERS|SWAHILI GLOSSARY|GLOSSARY|EDITOR.S ACKNOWLEDGMENTS|"
r"ACKNOWLEDGMENTS|ACKNOWLEDGEMENTS|"
r"Books by [A-Z]|BOOKS BY |Copyright|COPYRIGHT|Also by [A-Z])[ \t]*$", re.M)
def strip_backmatter(units, title):
if not units:
return units, 0
head, body = units[-1]
m = BACKMATTER.search(body)
if not m:
return units, 0
cut = body[:m.start()].rstrip()
removed = len(body.split()) - len(cut.split())
if len(cut.split()) < MIN_UNIT_WORDS:
print(f" ⚠ REFUSING back-matter strip on {title}: last unit would fall below the "
f"{MIN_UNIT_WORDS}-word floor", file=sys.stderr)
return units, 0
return units[:-1] + [(head, cut)], removed
def split_units(text: str):
"""Return (pattern_name, [(heading, body)]).
@@ -270,6 +308,7 @@ for w in works:
text, fixed_lines = repair_lines(text)
text, front_removed = strip_foreign_front(text, w["title"])
pattern, units = split_units(text)
units, back_removed = strip_backmatter(units, w["title"])
alphabet.update(ch for ch in text if ch.isalpha())
words = sum(len(b.split()) for _, b in units)
total_words += words; total_units += len(units)
@@ -279,7 +318,8 @@ for w in works:
if w["title"] in CONTINUOUS else "")
post = " (posthumous/edited)" if w["title"] in POSTHUMOUS_EDITED else ""
extra = (f" [smallcaps {fixed_lines}]" if fixed_lines else "") + \
(f" [front -{front_removed}w]" if front_removed else "")
(f" [front -{front_removed}w]" if front_removed else "") + \
(f" [back -{back_removed}w]" if back_removed else "")
print(f" {w['slug']:26} {len(units):>4} units {words:>8,} words "
f"via {pattern:<14}{post}{flag}{extra}")
manifest["works"].append({"slug": w["slug"], "title": w["title"], "rights": w["rights"],
@@ -287,6 +327,7 @@ for w in works:
"heading_pattern": pattern,
"smallcaps_lines_repaired": fixed_lines,
"foreign_front_matter_words_removed": front_removed,
"publisher_back_matter_words_removed": back_removed,
"posthumous_editor_shaped": w["title"] in POSTHUMOUS_EDITED,
"path": f"works/{w['slug']}.jsonl"})
if not a.survey: