Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
64 lines
2.7 KiB
Python
64 lines
2.7 KiB
Python
"""A TypeSafe-style POST /v1/systemone server around Intern-Decision's own inference.py
|
|
(DecisionEngine.predict, shipped in the model repo), so JevBench's typesafe adapter and the bench
|
|
harness reach it the same way they reach jevk5-serve and imajev's server. Nothing about the scoring
|
|
is changed: the request goes to predict() as-is (the card's documented request format) and its
|
|
response is returned with one added field, timing.server_ms.
|
|
python intern_server.py --checkpoint <snapshot dir> --port 18090
|
|
"""
|
|
import argparse
|
|
import json
|
|
import sys
|
|
import threading
|
|
import time
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--checkpoint", required=True)
|
|
ap.add_argument("--host", default="127.0.0.1")
|
|
ap.add_argument("--port", type=int, default=18090)
|
|
args = ap.parse_args()
|
|
sys.path.insert(0, args.checkpoint)
|
|
from inference import DecisionEngine # noqa: E402 (the checkpoint's own module)
|
|
|
|
engine = DecisionEngine(checkpoint=args.checkpoint, device="cuda")
|
|
engine.predict({"state": "warm-up", "questions": {"q": {"type": "noul", "instructions": "Is this a warm-up?"}}})
|
|
lock = threading.Lock()
|
|
|
|
|
|
class H(BaseHTTPRequestHandler):
|
|
def _send(self, code, payload):
|
|
data = json.dumps(payload).encode()
|
|
self.send_response(code)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(data)))
|
|
self.end_headers()
|
|
self.wfile.write(data)
|
|
|
|
def do_GET(self):
|
|
self._send(200, {"ok": True, "model": "Intern-Decision"} if self.path.rstrip("/") == "/health"
|
|
else {"error": "not found"})
|
|
|
|
def do_POST(self):
|
|
if self.path.rstrip("/") != "/v1/systemone":
|
|
return self._send(404, {"error": "not found"})
|
|
t = time.perf_counter()
|
|
try:
|
|
body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0))))
|
|
with lock:
|
|
out = engine.predict({k: body[k] for k in ("state", "questions", "images") if k in body})
|
|
except (ValueError, KeyError, TypeError, ArithmeticError) as e:
|
|
return self._send(400, {"error": f"{type(e).__name__}: {e}"})
|
|
except Exception as e: # noqa: BLE001 e.g. CUDA OOM under the cap
|
|
import torch
|
|
torch.cuda.empty_cache()
|
|
return self._send(503, {"error": f"{type(e).__name__}: {str(e)[:300]}"})
|
|
out.setdefault("timing", {})["server_ms"] = round((time.perf_counter() - t) * 1000, 2)
|
|
self._send(200, out)
|
|
|
|
def log_message(self, *a):
|
|
pass
|
|
|
|
|
|
print(f"serving Intern-Decision {args.checkpoint} on {args.host}:{args.port}", flush=True)
|
|
ThreadingHTTPServer((args.host, args.port), H).serve_forever()
|