#!/usr/bin/env python3 """P() at the first generated token, with enable_thinking=False. Measures how much probability mass a checkpoint puts on OPENING a think block when the chat template has ALREADY closed one for it. That is the exact event behind the h300 gen-seat leak. CPU-only: no GPU contention, no seat downtime. """ import json, sys, torch from transformers import AutoTokenizer, AutoModelForCausalLM path = sys.argv[1] PROMPT = ("A farmer has 17 sheep. All but 9 run away. He buys twice as many as he " "has left, then sells 4. How many now? Explain.") tok = AutoTokenizer.from_pretrained(path) text = tok.apply_chat_template([{"role": "user", "content": PROMPT}], tokenize=False, add_generation_prompt=True, enable_thinking=False) assert text.rstrip().endswith(""), "template did NOT pre-close the think block:\n" + repr(text[-120:]) ids = tok(text, return_tensors="pt") model = AutoModelForCausalLM.from_pretrained(path, dtype=torch.bfloat16, device_map=None) model.eval() with torch.no_grad(): logits = model(**ids).logits[0, -1].float() probs = torch.softmax(logits, dim=-1) think_id = tok.convert_tokens_to_ids("") p_think = probs[think_id].item() top = torch.topk(probs, 12) out = { "model": path.rstrip("/").split("/")[-1], "p_think": p_think, "think_token_id": think_id, "think_rank": int((probs > p_think).sum().item()) + 1, "top12": [{"tok": tok.decode([i]), "p": round(probs[i].item(), 5)} for i in top.indices.tolist()], } print("RESULT " + json.dumps(out))