docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
This commit is contained in:
@@ -0,0 +1,63 @@
|
||||
"""A TypeSafe-style POST /v1/systemone server around Intern-Decision's own inference.py
|
||||
(DecisionEngine.predict, shipped in the model repo), so JevBench's typesafe adapter and the bench
|
||||
harness reach it the same way they reach jevk5-serve and imajev's server. Nothing about the scoring
|
||||
is changed: the request goes to predict() as-is (the card's documented request format) and its
|
||||
response is returned with one added field, timing.server_ms.
|
||||
python intern_server.py --checkpoint <snapshot dir> --port 18090
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--checkpoint", required=True)
|
||||
ap.add_argument("--host", default="127.0.0.1")
|
||||
ap.add_argument("--port", type=int, default=18090)
|
||||
args = ap.parse_args()
|
||||
sys.path.insert(0, args.checkpoint)
|
||||
from inference import DecisionEngine # noqa: E402 (the checkpoint's own module)
|
||||
|
||||
engine = DecisionEngine(checkpoint=args.checkpoint, device="cuda")
|
||||
engine.predict({"state": "warm-up", "questions": {"q": {"type": "noul", "instructions": "Is this a warm-up?"}}})
|
||||
lock = threading.Lock()
|
||||
|
||||
|
||||
class H(BaseHTTPRequestHandler):
|
||||
def _send(self, code, payload):
|
||||
data = json.dumps(payload).encode()
|
||||
self.send_response(code)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
def do_GET(self):
|
||||
self._send(200, {"ok": True, "model": "Intern-Decision"} if self.path.rstrip("/") == "/health"
|
||||
else {"error": "not found"})
|
||||
|
||||
def do_POST(self):
|
||||
if self.path.rstrip("/") != "/v1/systemone":
|
||||
return self._send(404, {"error": "not found"})
|
||||
t = time.perf_counter()
|
||||
try:
|
||||
body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0))))
|
||||
with lock:
|
||||
out = engine.predict({k: body[k] for k in ("state", "questions", "images") if k in body})
|
||||
except (ValueError, KeyError, TypeError, ArithmeticError) as e:
|
||||
return self._send(400, {"error": f"{type(e).__name__}: {e}"})
|
||||
except Exception as e: # noqa: BLE001 e.g. CUDA OOM under the cap
|
||||
import torch
|
||||
torch.cuda.empty_cache()
|
||||
return self._send(503, {"error": f"{type(e).__name__}: {str(e)[:300]}"})
|
||||
out.setdefault("timing", {})["server_ms"] = round((time.perf_counter() - t) * 1000, 2)
|
||||
self._send(200, out)
|
||||
|
||||
def log_message(self, *a):
|
||||
pass
|
||||
|
||||
|
||||
print(f"serving Intern-Decision {args.checkpoint} on {args.host}:{args.port}", flush=True)
|
||||
ThreadingHTTPServer((args.host, args.port), H).serve_forever()
|
||||
Reference in New Issue
Block a user