docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced

Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231,
hard 0.613) reproduced exactly; negative control and a 4-restart noise floor
(0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with-
rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one
ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench
rank does not transfer. Raw per-item data kept out of git.
This commit is contained in:
vh
2026-09-30 05:00:45 -07:00
parent 9a6ac59da7
commit 475d6d6bcb
27 changed files with 19068 additions and 0 deletions
@@ -0,0 +1,63 @@
"""A TypeSafe-style POST /v1/systemone server around Intern-Decision's own inference.py
(DecisionEngine.predict, shipped in the model repo), so JevBench's typesafe adapter and the bench
harness reach it the same way they reach jevk5-serve and imajev's server. Nothing about the scoring
is changed: the request goes to predict() as-is (the card's documented request format) and its
response is returned with one added field, timing.server_ms.
python intern_server.py --checkpoint <snapshot dir> --port 18090
"""
import argparse
import json
import sys
import threading
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
ap = argparse.ArgumentParser()
ap.add_argument("--checkpoint", required=True)
ap.add_argument("--host", default="127.0.0.1")
ap.add_argument("--port", type=int, default=18090)
args = ap.parse_args()
sys.path.insert(0, args.checkpoint)
from inference import DecisionEngine # noqa: E402 (the checkpoint's own module)
engine = DecisionEngine(checkpoint=args.checkpoint, device="cuda")
engine.predict({"state": "warm-up", "questions": {"q": {"type": "noul", "instructions": "Is this a warm-up?"}}})
lock = threading.Lock()
class H(BaseHTTPRequestHandler):
def _send(self, code, payload):
data = json.dumps(payload).encode()
self.send_response(code)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(data)))
self.end_headers()
self.wfile.write(data)
def do_GET(self):
self._send(200, {"ok": True, "model": "Intern-Decision"} if self.path.rstrip("/") == "/health"
else {"error": "not found"})
def do_POST(self):
if self.path.rstrip("/") != "/v1/systemone":
return self._send(404, {"error": "not found"})
t = time.perf_counter()
try:
body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0))))
with lock:
out = engine.predict({k: body[k] for k in ("state", "questions", "images") if k in body})
except (ValueError, KeyError, TypeError, ArithmeticError) as e:
return self._send(400, {"error": f"{type(e).__name__}: {e}"})
except Exception as e: # noqa: BLE001 e.g. CUDA OOM under the cap
import torch
torch.cuda.empty_cache()
return self._send(503, {"error": f"{type(e).__name__}: {str(e)[:300]}"})
out.setdefault("timing", {})["server_ms"] = round((time.perf_counter() - t) * 1000, 2)
self._send(200, out)
def log_message(self, *a):
pass
print(f"serving Intern-Decision {args.checkpoint} on {args.host}:{args.port}", flush=True)
ThreadingHTTPServer((args.host, args.port), H).serve_forever()