#!/usr/bin/env python3
"""P() at the first generated token, with enable_thinking=False.
Measures how much probability mass a checkpoint puts on OPENING a think block
when the chat template has ALREADY closed one for it. That is the exact event
behind the h300 gen-seat leak. CPU-only: no GPU contention, no seat downtime.
"""
import json, sys, torch
from transformers import AutoTokenizer, AutoModelForCausalLM
path = sys.argv[1]
PROMPT = ("A farmer has 17 sheep. All but 9 run away. He buys twice as many as he "
"has left, then sells 4. How many now? Explain.")
tok = AutoTokenizer.from_pretrained(path)
text = tok.apply_chat_template([{"role": "user", "content": PROMPT}],
tokenize=False, add_generation_prompt=True,
enable_thinking=False)
assert text.rstrip().endswith(""), "template did NOT pre-close the think block:\n" + repr(text[-120:])
ids = tok(text, return_tensors="pt")
model = AutoModelForCausalLM.from_pretrained(path, dtype=torch.bfloat16, device_map=None)
model.eval()
with torch.no_grad():
logits = model(**ids).logits[0, -1].float()
probs = torch.softmax(logits, dim=-1)
think_id = tok.convert_tokens_to_ids("")
p_think = probs[think_id].item()
top = torch.topk(probs, 12)
out = {
"model": path.rstrip("/").split("/")[-1],
"p_think": p_think,
"think_token_id": think_id,
"think_rank": int((probs > p_think).sum().item()) + 1,
"top12": [{"tok": tok.decode([i]), "p": round(probs[i].item(), 5)} for i in top.indices.tolist()],
}
print("RESULT " + json.dumps(out))