"""Test doubles for SemIf's tokenizer seam (INV-7), shared by test_prefix and test_engine_load. MergeTokenizer reproduces the real failure's shape without the Qwen vocabulary; the real tokenizer is checked on the card, in acceptance.""" import json class MergeTokenizer: """Char-level with three merges, in the shape of the real failure. ')"' and '"}' are single tokens, but ')"},' splits as [')', '"},']. So a state tail ')"}' encodes as [')"', '}'] when nothing follows it (the prefix), and the full prompt's ')"},' encodes as [')', '"},']. Dropping one token from the prefix leaves ')"', which the full prompt does not contain. A tail like '."}' stays ['.', '"}'] either way, which is the ordinary case.""" def apply_chat_template(self, messages, tokenize, add_generation_prompt, enable_thinking): assert tokenize is False and add_generation_prompt is True and enable_thinking is False return "" + "|".join(m["content"] for m in messages) + "" def encode(self, text, add_special_tokens): assert add_special_tokens is False out, i = [], 0 while i < len(text): if text.startswith(')"},', i): out += [')', '"},']; i += 4 elif text.startswith(')"', i) or text.startswith('"}', i): out.append(text[i:i + 2]); i += 2 else: out.append(text[i]); i += 1 return out def messages(row): # the shape of semif_phase1.core.direct_messages: evidence first, then the criterion return [{"role": "user", "content": json.dumps( {"evidence": row["state"], "criterion": row["question"], "options": row["options"]}, ensure_ascii=False)}] def upstream_prefix(tokenizer, state): # what SemIf's _state_prefix does: through the state, minus one token return tokenizer.encode("" + json.dumps({"evidence": state}, ensure_ascii=False)[:-1], add_special_tokens=False)[:-1]