fix(semif): 0.1.4 — object states ending in ) ; } no longer 422 (INV-7)

SemIf's shared scorer trims one token at the state boundary. When an object
state's last value ends in ')', ';' or '}', the JSON that follows re-merges two
tokens back, so score_shared refused the request with 422. The engine now wraps
semif_phase1.shared._state_prefix to keep only the tokens the full prompts
share. Each row scores the same token sequence; only the prefill/suffix split
moves.

Startup proves the fix is in effect, not just installed (heid bug hunt SKAL,
folded). It checks that the hook is callable and is what score_shared resolves,
that an ordinary state keeps upstream's whole prefix, and that a merge-prone
state scores through the shared path.

Real tokenizer: 154 states, 23 refused before and 0 after, with no ordinary or
authored144 prefix changed. Acceptance: 144/144 parity. Shared vs direct 71/72;
the miss is a bf16 tie that flipped across a plain restart (see README).
This commit is contained in:
vh
2026-09-27 10:23:14 -07:00
parent 7e11cf247b
commit 47cad33dd1
11 changed files with 384 additions and 20 deletions
@@ -0,0 +1,37 @@
"""Test doubles for SemIf's tokenizer seam (INV-7), shared by test_prefix and test_engine_load.
MergeTokenizer reproduces the real failure's shape without the Qwen vocabulary; the real
tokenizer is checked on the card, in acceptance."""
import json
class MergeTokenizer:
"""Char-level with three merges, in the shape of the real failure. ')"' and '"}' are single
tokens, but ')"},' splits as [')', '"},']. So a state tail ')"}' encodes as [')"', '}'] when
nothing follows it (the prefix), and the full prompt's ')"},' encodes as [')', '"},'].
Dropping one token from the prefix leaves ')"', which the full prompt does not contain. A
tail like '."}' stays ['.', '"}'] either way, which is the ordinary case."""
def apply_chat_template(self, messages, tokenize, add_generation_prompt, enable_thinking):
assert tokenize is False and add_generation_prompt is True and enable_thinking is False
return "<s>" + "|".join(m["content"] for m in messages) + "<a>"
def encode(self, text, add_special_tokens):
assert add_special_tokens is False
out, i = [], 0
while i < len(text):
if text.startswith(')"},', i):
out += [')', '"},']; i += 4
elif text.startswith(')"', i) or text.startswith('"}', i):
out.append(text[i:i + 2]); i += 2
else:
out.append(text[i]); i += 1
return out
def messages(row): # the shape of semif_phase1.core.direct_messages: evidence first, then the criterion
return [{"role": "user", "content": json.dumps(
{"evidence": row["state"], "criterion": row["question"], "options": row["options"]}, ensure_ascii=False)}]
def upstream_prefix(tokenizer, state): # what SemIf's _state_prefix does: through the state, minus one token
return tokenizer.encode("<s>" + json.dumps({"evidence": state}, ensure_ascii=False)[:-1], add_special_tokens=False)[:-1]
+100 -3
View File
@@ -6,6 +6,8 @@ import types
import pytest
from fake_tokenizer import MergeTokenizer, messages, upstream_prefix
from semif_serve.config import Settings
TOKEN = "t" * 40
@@ -24,6 +26,25 @@ class Model:
yield Param(self._device)
# The shape of SemIf's score_shared, compiled INTO the fake module so that, like the real one, it
# resolves _state_prefix and direct_messages as module globals at call time and refuses (ValueError)
# a prefix that is not a token-prefix of every row.
SHARED_SRC = """
def score_shared(model, tokenizer, rows, metadata, max_tokens=4096):
prefix = _state_prefix(tokenizer, rows[0]["state"])
for row in rows:
ids = tokenizer.encode(tokenizer.apply_chat_template(direct_messages(row), tokenize=False,
add_generation_prompt=True, enable_thinking=False), add_special_tokens=False)
if not prefix or ids[:len(prefix)] != prefix:
raise ValueError("The fixed state prefix does not match every full prompt")
CALLS.append(("shared", rows[0]["state"]))
return [{"id": row["id"]} for row in rows], {}
"""
# A SemIf revision that binds the helper when score_shared is defined: the module global can be
# replaced all day and scoring never sees it.
SHARED_SRC_EARLY_BOUND = SHARED_SRC.replace("max_tokens=4096):", "max_tokens=4096, _state_prefix=_state_prefix):")
@pytest.fixture
def fakes(monkeypatch):
calls = []
@@ -42,7 +63,7 @@ def fakes(monkeypatch):
def load_causal_model(model, revision, device, dtype):
calls.append(("load", model, revision, device, dtype))
return Model(state["device"]), object(), {"source": model}
return Model(state["device"]), MergeTokenizer(), {"source": model}
def score(model, tok, row, meta, max_tokens):
calls.append(("score", row["id"]))
@@ -52,10 +73,12 @@ def fakes(monkeypatch):
core = types.ModuleType("semif_phase1.core")
core.load_causal_model = load_causal_model
core.direct_messages = messages
direct = types.ModuleType("semif_phase1.direct")
direct.score = score
shared = types.ModuleType("semif_phase1.shared")
shared.score_shared = lambda *a: ([], {})
shared._state_prefix, shared.direct_messages, shared.CALLS = upstream_prefix, messages, calls
exec(SHARED_SRC, shared.__dict__)
pkg = types.ModuleType("semif_phase1")
for name, mod in {"torch": torch, "semif_phase1": pkg, "semif_phase1.core": core,
"semif_phase1.direct": direct, "semif_phase1.shared": shared}.items():
@@ -67,7 +90,7 @@ def test_load_caps_before_the_weights_land_then_warms_up(fakes):
from semif_serve.engine import TorchEngine
_torch, calls, _ = fakes
TorchEngine.load(Settings(api_token=TOKEN, vram_cap_gib=12.0))
assert [c[0] for c in calls] == ["cap", "load", "score"]
assert [c[0] for c in calls] == ["cap", "load", "score", "shared"]
assert calls[0] == ("cap", round(12 / 96, 4))
assert calls[2] == ("score", "semif-serve-warmup")
@@ -108,3 +131,77 @@ def test_the_entry_point_forces_offline_mode_before_the_engine_loads(fakes, monk
classmethod(lambda cls, s: seen.update(offline=os.environ.get("HF_HUB_OFFLINE")) or object()))
main.app_from_env()
assert seen["offline"] == "1"
def test_load_wraps_the_upstream_state_prefix_once_even_across_reloads(fakes):
"""INV-7: score_shared looks _state_prefix up as a module global, so the wrapper must replace
it there. A second load must wrap the ORIGINAL again, not stack wrapper on wrapper."""
import semif_phase1.shared as upstream
from semif_serve.engine import TorchEngine
for _ in range(2):
TorchEngine.load(Settings(api_token=TOKEN))
assert upstream._state_prefix is not upstream_prefix
assert upstream._state_prefix.semif_serve_wraps is upstream_prefix
def test_load_fails_closed_when_the_pinned_semif_no_longer_has_the_prefix_hook(fakes, monkeypatch):
import semif_phase1.shared as upstream
from semif_serve.engine import TorchEngine
_torch, calls, _ = fakes
monkeypatch.delattr(upstream, "_state_prefix")
with pytest.raises(RuntimeError, match="_state_prefix"):
TorchEngine.load(Settings(api_token=TOKEN))
assert not any(c[0] == "load" for c in calls)
def test_load_proves_the_hook_on_a_merge_prone_state_through_the_real_shared_path(fakes):
"""Bug hunt SKAL (R2/H1, H2): existence is not effect. Startup scores one state that upstream
alone refuses, through score_shared itself."""
import semif_phase1.shared as upstream
from semif_serve.engine import BOUNDARY_ROW, TorchEngine
_torch, calls, _ = fakes
with pytest.raises(ValueError): # the probe really is merge-prone here
upstream.score_shared(None, MergeTokenizer(), [BOUNDARY_ROW], {})
TorchEngine.load(Settings(api_token=TOKEN))
assert ("shared", BOUNDARY_ROW["state"]) in calls
def test_load_fails_closed_when_score_shared_binds_the_prefix_before_the_patch(fakes):
import semif_phase1.shared as upstream
from semif_serve.engine import TorchEngine
exec(SHARED_SRC_EARLY_BOUND, upstream.__dict__)
with pytest.raises(RuntimeError, match="INV-7"):
TorchEngine.load(Settings(api_token=TOKEN))
def test_load_fails_closed_when_score_shared_resolves_its_globals_elsewhere(fakes):
import semif_phase1.shared as upstream
from semif_serve.engine import TorchEngine
_torch, calls, _ = fakes
elsewhere = {"_state_prefix": upstream_prefix, "direct_messages": messages, "CALLS": calls}
exec(SHARED_SRC, elsewhere)
upstream.score_shared = elsewhere["score_shared"] # re-exported from another module
with pytest.raises(RuntimeError, match="INV-7"):
TorchEngine.load(Settings(api_token=TOKEN))
assert not any(c[0] == "load" for c in calls)
def test_load_fails_closed_when_the_hook_is_not_callable(fakes, monkeypatch):
import semif_phase1.shared as upstream
from semif_serve.engine import TorchEngine
_torch, calls, _ = fakes
monkeypatch.setattr(upstream, "_state_prefix", "not a function")
with pytest.raises(RuntimeError, match="_state_prefix"):
TorchEngine.load(Settings(api_token=TOKEN))
assert not any(c[0] == "load" for c in calls)
def test_load_fails_closed_when_the_wrapper_renders_the_wrong_prompt(fakes, monkeypatch):
"""The SURVIVED row of bug hunt SKAL: a wrapper that renders nothing still yields a valid
(degenerate) prefix, so scoring stays correct and silently loses all sharing. Startup
requires the wrapper to keep upstream's whole prefix on an ordinary state."""
import semif_phase1.core as core
from semif_serve.engine import TorchEngine
monkeypatch.setattr(core, "direct_messages", lambda row: [])
with pytest.raises(RuntimeError, match="INV-7"):
TorchEngine.load(Settings(api_token=TOKEN))
+39
View File
@@ -0,0 +1,39 @@
"""The shared prefix must never cover a token that the full prompts do not share (0.1.4, INV-7).
Found 2026-09-27 by the Cicada mood spike. An object state whose last value ends in ")", ";" or
"}" was refused with 422 "The fixed state prefix does not match every full prompt". SemIf's
_state_prefix encodes the prompt up to the end of the state and drops ONE token, because the
JSON punctuation that follows can merge with it. Here the merge reaches two tokens back.
MergeTokenizer reproduces that shape without the real Qwen vocabulary. The real tokenizer is
checked on the card, in acceptance.
"""
from fake_tokenizer import MergeTokenizer, messages, upstream_prefix
from semif_serve.engine import boundary_safe_prefix
ROW = {"question": "Which pose?", "options": [{"id": "a", "description": "A"}, {"id": "b", "description": "B"}]}
def full(tok, state, row=ROW):
return tok.encode(tok.apply_chat_template(messages({**row, "state": state}), tokenize=False,
add_generation_prompt=True, enable_thinking=False), add_special_tokens=False)
def test_the_merge_that_broke_upstream_is_reproduced():
tok, state = MergeTokenizer(), {"person_said": "ok :)"}
ids = upstream_prefix(tok, state)
assert full(tok, state)[:len(ids)] != ids # the 422: upstream's prefix is not a prefix
def test_a_state_ending_in_a_merging_character_gets_a_prefix_every_row_shares():
tok, state = MergeTokenizer(), {"person_said": "ok :)"}
ids = boundary_safe_prefix(upstream_prefix, messages)(tok, state)
for row in (ROW, {"question": "A different criterion entirely?", "options": ROW["options"][::-1]}):
assert full(tok, state, row)[:len(ids)] == ids
assert ids == upstream_prefix(tok, state)[:-1] # it gave up exactly the one mismatched token
def test_an_ordinary_state_keeps_the_whole_upstream_prefix():
tok = MergeTokenizer()
for state in ({"person_said": "Turn off the lights."}, "ok :)", "Nothing was said."):
assert boundary_safe_prefix(upstream_prefix, messages)(tok, state) == upstream_prefix(tok, state)