import json, sys, urllib.request, collections KEY=open('/home/lkraven/.config/litellm/infra-ops-key').read().strip() URL="http://10.250.50.70:4000/v1/chat/completions" PROMPTS = [ "A farmer has 17 sheep. All but 9 run away. He then buys twice as many as he has left, and sells 4. How many does he have? Explain your reasoning.", "Compare the trade-offs of NVMe RAIDZ2 versus mirrored vdevs for a write-heavy database workload.", "Write a short scene: two engineers argue about whether to ship a known-flaky feature.", ] def call(model, prompt, max_tokens): body={"model":model,"messages":[{"role":"user","content":prompt}],"max_tokens":max_tokens} req=urllib.request.Request(URL,data=json.dumps(body).encode(), headers={"Authorization":"Bearer "+KEY,"Content-Type":"application/json"}) d=json.loads(urllib.request.urlopen(req,timeout=300).read().decode(),strict=False) c=d["choices"][0]; m=c["message"] return {"finish":c.get("finish_reason"), "content":m.get("content") or "", "reasoning":m.get("reasoning_content") or "", "ctok":d.get("usage",{}).get("completion_tokens"), "rtok":(d.get("usage",{}).get("completion_tokens_details") or {}).get("reasoning_tokens")} model=sys.argv[1]; maxtok=int(sys.argv[2]); n=int(sys.argv[3]) print(f"=== {model} | max_tokens={maxtok} | n={n} per prompt ===") tally=collections.Counter() for pi,p in enumerate(PROMPTS): for i in range(n): try: r=call(model,p,maxtok) except Exception as e: print(f" p{pi} #{i}: ERROR {e}"); tally["error"]+=1; continue empty = len(r["content"].strip())==0 has_think = "" in r["content"] or "" in r["reasoning"] flag = " <<< EMPTY CONTENT" if empty else "" tally[r["finish"]]+=1 if empty: tally["empty_content"]+=1 if r["reasoning"]: tally["had_reasoning"]+=1 print(f" p{pi} #{i}: finish={r['finish']:<8} ctok={r['ctok']:<5} content={len(r['content']):<5} reasoning={len(r['reasoning']):<6} rtok={r['rtok']} think_tag={has_think}{flag}") print(" TALLY:", dict(tally))