fix(semif): 0.1.4 — object states ending in ) ; } no longer 422 (INV-7)
SemIf's shared scorer trims one token at the state boundary. When an object state's last value ends in ')', ';' or '}', the JSON that follows re-merges two tokens back, so score_shared refused the request with 422. The engine now wraps semif_phase1.shared._state_prefix to keep only the tokens the full prompts share. Each row scores the same token sequence; only the prefill/suffix split moves. Startup proves the fix is in effect, not just installed (heid bug hunt SKAL, folded). It checks that the hook is callable and is what score_shared resolves, that an ordinary state keeps upstream's whole prefix, and that a merge-prone state scores through the shared path. Real tokenizer: 154 states, 23 refused before and 0 after, with no ordinary or authored144 prefix changed. Acceptance: 144/144 parity. Shared vs direct 71/72; the miss is a bf16 tie that flipped across a plain restart (see README).
This commit is contained in:
@@ -28,6 +28,44 @@ WARMUP_ROW = {
|
||||
}
|
||||
|
||||
|
||||
# INV-7 startup proof: a state SemIf's own prefix refuses (measured with the real tokenizer,
|
||||
# 2026-09-27), sent through score_shared itself, so the fix is shown to be IN EFFECT, not just
|
||||
# installed (bug hunt SKAL, R2/H1/H2).
|
||||
BOUNDARY_ROW = {**WARMUP_ROW, "id": "semif-serve-boundary-check", "state": {"person_said": "ok :)"}}
|
||||
PREFIX_PROBE_ROW = {
|
||||
"id": "semif-serve-prefix-probe",
|
||||
"question": "prefix boundary placeholder",
|
||||
"options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}],
|
||||
}
|
||||
|
||||
|
||||
def boundary_safe_prefix(state_prefix: Callable, messages: Callable) -> Callable:
|
||||
"""INV-7: wrap SemIf's `shared._state_prefix` so the prefill never covers a token that the full
|
||||
prompts do not share.
|
||||
|
||||
Upstream encodes the prompt up to the end of the state and drops ONE token, because the JSON
|
||||
punctuation that follows the state can merge with it. One is not always enough: an object
|
||||
state whose last value ends in ")", ";" or "}" re-tokenises TWO tokens back once `, "criterion"`
|
||||
follows, and score_shared then refused the request with 422 (found 2026-09-27). This keeps only
|
||||
the leading tokens that upstream's prefix shares with a real full prompt for the same state.
|
||||
The suffix starts that much earlier and scores the same token sequence; the cost is a few
|
||||
tokens of lost sharing. Only the punctuation run at the boundary can merge, and it is the same
|
||||
in every row, so a probe row stands for all of them. score_shared still checks every real row
|
||||
and fails closed if that ever stops holding."""
|
||||
def prefix(tokenizer, state):
|
||||
ids = state_prefix(tokenizer, state)
|
||||
full = tokenizer.encode(tokenizer.apply_chat_template(
|
||||
messages({**PREFIX_PROBE_ROW, "state": state}), tokenize=False, add_generation_prompt=True,
|
||||
enable_thinking=False), add_special_tokens=False)
|
||||
shared = 0
|
||||
while shared < min(len(ids), len(full)) and ids[shared] == full[shared]:
|
||||
shared += 1
|
||||
return ids[:shared]
|
||||
|
||||
prefix.semif_serve_wraps = state_prefix
|
||||
return prefix
|
||||
|
||||
|
||||
def _first_line(exc: BaseException) -> str:
|
||||
lines = str(exc).splitlines()
|
||||
return lines[0] if lines else ""
|
||||
@@ -44,10 +82,24 @@ class TorchEngine:
|
||||
@classmethod
|
||||
def load(cls, settings: Settings) -> "TorchEngine":
|
||||
import torch
|
||||
from semif_phase1.core import load_causal_model
|
||||
import semif_phase1.shared as upstream_shared
|
||||
from semif_phase1.core import direct_messages, load_causal_model
|
||||
from semif_phase1.direct import score
|
||||
from semif_phase1.shared import score_shared
|
||||
|
||||
# INV-7: score_shared looks `_state_prefix` up as a module global, so the wrapper goes there.
|
||||
# A SemIf bump that renames it, or moves score_shared so it resolves its globals elsewhere,
|
||||
# must stop startup, not silently bring the 422 back.
|
||||
original = getattr(upstream_shared, "_state_prefix", None)
|
||||
if not callable(original):
|
||||
raise RuntimeError("semif_phase1.shared._state_prefix is gone or not callable at this SemIf "
|
||||
"commit: re-check the boundary-safe prefix (INV-7) before serving")
|
||||
original = getattr(original, "semif_serve_wraps", original) # one wrapper, however many loads
|
||||
upstream_shared._state_prefix = boundary_safe_prefix(original, direct_messages)
|
||||
if getattr(score_shared, "__globals__", {}).get("_state_prefix") is not upstream_shared._state_prefix:
|
||||
raise RuntimeError("INV-7: score_shared does not resolve semif_phase1.shared._state_prefix, "
|
||||
"so the boundary-safe prefix would be inert")
|
||||
|
||||
if settings.device == "cuda":
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device")
|
||||
@@ -70,10 +122,23 @@ class TorchEngine:
|
||||
raise RuntimeError(f"model landed on {placed}, expected {settings.device}")
|
||||
engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared)
|
||||
engine.direct(WARMUP_ROW) # INV-3: one decision must score
|
||||
engine._prove_prefix_hook(upstream_shared._state_prefix, original)
|
||||
if settings.device == "cuda": # INV-4: the resting footprint
|
||||
engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES
|
||||
return engine
|
||||
|
||||
def _prove_prefix_hook(self, hook: Callable, original: Callable) -> None:
|
||||
"""INV-7, at startup: the wrapper keeps upstream's whole prefix on an ordinary state (a wrapper
|
||||
rendering the wrong prompt would silently give up all sharing), and a state upstream alone
|
||||
refuses scores through score_shared itself (a hook score_shared never calls would not)."""
|
||||
state = WARMUP_ROW["state"]
|
||||
if hook(self._tokenizer, state) != original(self._tokenizer, state):
|
||||
raise RuntimeError("INV-7: the boundary-safe prefix does not keep upstream's prefix on an ordinary state")
|
||||
try:
|
||||
self.shared([BOUNDARY_ROW])
|
||||
except ValueError as exc:
|
||||
raise RuntimeError(f"INV-7: shared scoring still refuses a merge-prone state: {exc}") from None
|
||||
|
||||
def health(self) -> dict:
|
||||
info = dict(self._metadata)
|
||||
if self._settings.device == "cuda":
|
||||
|
||||
Reference in New Issue
Block a user