feat(intern-decision-serve): Intern-Decision-4B behind semif-serve's HTTP surface
Contract, service and tests (fake engine, no GPU). Scores through the checkpoint's own inference.py (DecisionEngine.predict, sha256-pinned); maps semif decisions onto Jev choice questions, packs /decide/shared into calls of at most 16, runs orderings in waves, and keeps semif's error mapping, admission, body limit and hard VRAM cap. Deltas from semif-serve are listed in the contract.
This commit is contained in:
@@ -0,0 +1,55 @@
|
||||
"""A torch-free stand-in for the real engine: answers Jev requests in the shape
|
||||
DecisionEngine.predict() returns, and records every call it was given."""
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
|
||||
CALIBRATION = {"method": "temperature-scaling", "temperature": 1.99241824}
|
||||
|
||||
|
||||
def softmax(xs: list[float]) -> list[float]:
|
||||
top = max(xs)
|
||||
e = [math.exp(x - top) for x in xs]
|
||||
return [v / sum(e) for v in e]
|
||||
|
||||
|
||||
def score_by_description(field: str, question: dict) -> list[float]:
|
||||
"""Default scorer: the option whose description is longest wins; a small first-position bias."""
|
||||
descs = list(question["criteria"].values())
|
||||
return [len(d) + (0.5 if i == 0 else 0.0) for i, d in enumerate(descs)]
|
||||
|
||||
|
||||
class FakeEngine:
|
||||
def __init__(self, scorer=score_by_description, tokens_per_question: int = 100):
|
||||
self.scorer = scorer
|
||||
self.tokens_per_question = tokens_per_question
|
||||
self.calls: list[dict] = []
|
||||
|
||||
metadata = {"name": "Intern-Decision-4B", "revision": "0" * 40}
|
||||
|
||||
def health(self) -> dict:
|
||||
return {**self.metadata, "reserved_gib": 9.1}
|
||||
|
||||
def predict(self, request: dict) -> tuple[dict, str]:
|
||||
self.calls.append(copy.deepcopy(request))
|
||||
answers = {}
|
||||
for field, question in request["questions"].items():
|
||||
ids = list(question["criteria"])
|
||||
probs = dict(zip(ids, softmax(self.scorer(field, question))))
|
||||
best = min(ids, key=lambda i: (-probs[i], i))
|
||||
answers[field] = {"type": "choice", "probabilities": probs, "confidence": probs[best],
|
||||
"choice": best, "source": "local", "decision": best}
|
||||
response = {"answers": answers,
|
||||
"usage": {"input_tokens": self.tokens_per_question * len(answers),
|
||||
"output_tokens": len(answers), "decision_count": len(answers)},
|
||||
"timing": {"inference_ms": 12.5}, "calibration": dict(CALIBRATION),
|
||||
"model": "Intern-Decision-4B", "backend": "hf"}
|
||||
return response, request_sha(request)
|
||||
|
||||
|
||||
def request_sha(request: dict) -> str:
|
||||
"""Stands in for the real prompt hash: the same call always hashes the same."""
|
||||
return hashlib.sha256(json.dumps(request, sort_keys=True).encode()).hexdigest()
|
||||
Reference in New Issue
Block a user