Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
35 lines
1.5 KiB
Python
35 lines
1.5 KiB
Python
"""Build the ~3,900-token state for the long shapes: JevBench v1.2.16's longest public hard state
|
|
(hard-opus-a-long_policy-01), topped up with the next longest until SemIf's shared prefix (template
|
|
+ evidence, as semif-serve's timing.prefix_tokens counts it) reaches ~3,900 Qwen3.5 tokens.
|
|
Runs in the semif-serve image (tokenizer only, CPU)."""
|
|
import json
|
|
import sys
|
|
|
|
from transformers import AutoTokenizer
|
|
from semif_phase1.core import direct_messages
|
|
|
|
tok = AutoTokenizer.from_pretrained("Qwen/Qwen3.5-4B", revision="851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a")
|
|
rows = {json.loads(l)["id"]: json.loads(l) for l in open(sys.argv[1])}
|
|
text = rows["hard-opus-a-long_policy-01"]["state"] + "\n\n" + rows["hard-sol-b-long_policy-01"]["state"]
|
|
|
|
|
|
def prefix_tokens(state):
|
|
msgs = direct_messages({"id": "x", "state": state, "question": "q", "options": [
|
|
{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]})
|
|
prompt = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True, enable_thinking=False)
|
|
payload = msgs[-1]["content"]
|
|
evidence = json.dumps({"evidence": state}, ensure_ascii=False)[:-1]
|
|
return len(tok.encode(prompt[: prompt.index(payload)] + evidence, add_special_tokens=False)) - 1
|
|
|
|
|
|
lo, hi = 1000, len(text)
|
|
while lo < hi:
|
|
mid = (lo + hi + 1) // 2
|
|
if prefix_tokens(text[:mid]) <= 3900:
|
|
lo = mid
|
|
else:
|
|
hi = mid - 1
|
|
state = text[:lo]
|
|
open(sys.argv[2], "w").write(state)
|
|
print(json.dumps({"chars": len(state), "prefix_tokens": prefix_tokens(state)}))
|