"""A TypeSafe-style POST /v1/systemone server around Intern-Decision's own inference.py (DecisionEngine.predict, shipped in the model repo), so JevBench's typesafe adapter and the bench harness reach it the same way they reach jevk5-serve and imajev's server. Nothing about the scoring is changed: the request goes to predict() as-is (the card's documented request format) and its response is returned with one added field, timing.server_ms. python intern_server.py --checkpoint --port 18090 """ import argparse import json import sys import threading import time from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer ap = argparse.ArgumentParser() ap.add_argument("--checkpoint", required=True) ap.add_argument("--host", default="127.0.0.1") ap.add_argument("--port", type=int, default=18090) args = ap.parse_args() sys.path.insert(0, args.checkpoint) from inference import DecisionEngine # noqa: E402 (the checkpoint's own module) engine = DecisionEngine(checkpoint=args.checkpoint, device="cuda") engine.predict({"state": "warm-up", "questions": {"q": {"type": "noul", "instructions": "Is this a warm-up?"}}}) lock = threading.Lock() class H(BaseHTTPRequestHandler): def _send(self, code, payload): data = json.dumps(payload).encode() self.send_response(code) self.send_header("Content-Type", "application/json") self.send_header("Content-Length", str(len(data))) self.end_headers() self.wfile.write(data) def do_GET(self): self._send(200, {"ok": True, "model": "Intern-Decision"} if self.path.rstrip("/") == "/health" else {"error": "not found"}) def do_POST(self): if self.path.rstrip("/") != "/v1/systemone": return self._send(404, {"error": "not found"}) t = time.perf_counter() try: body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0)))) with lock: out = engine.predict({k: body[k] for k in ("state", "questions", "images") if k in body}) except (ValueError, KeyError, TypeError, ArithmeticError) as e: return self._send(400, {"error": f"{type(e).__name__}: {e}"}) except Exception as e: # noqa: BLE001 e.g. CUDA OOM under the cap import torch torch.cuda.empty_cache() return self._send(503, {"error": f"{type(e).__name__}: {str(e)[:300]}"}) out.setdefault("timing", {})["server_ms"] = round((time.perf_counter() - t) * 1000, 2) self._send(200, out) def log_message(self, *a): pass print(f"serving Intern-Decision {args.checkpoint} on {args.host}:{args.port}", flush=True) ThreadingHTTPServer((args.host, args.port), H).serve_forever()