services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF cache and returns SemIf's result dicts unchanged, with an optional per-workload temperature-calibrated view. Contract: semif-serve.contract.md. Built with a short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid bug-hunt panel (pending). On the card: - torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack; - a hard 12 GiB VRAM cap. Two defects surfaced only on the card, and each fix is covered by a test: - 0.1.1: an OOM raised as a chained exception kept the failed request's tensors alive (11.9 GiB after the 503). It is now raised unchained, after gc. - 0.1.2: a large request left 12.6 GB reserved on the shared card. After each call, reserved memory over the baseline + 512 MiB is now released. Acceptance against SemIf's committed torch predictions (authored144): - 142/144 same top choice; both misses are exact bf16 ties; - 144/144 identical prompt hashes; - deterministic A-vs-A; - negative control 14/144; - shared vs direct 72/72. 21 binary criteria over one state take 159 ms. The shared-mode capacity table under the cap is in stacks/semif/README.md. The Dockerfile installs dependencies from a manifest with the project version blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild, dependency layer CACHED. DNS: semif.fv.internal. Token: vault semif/api-token.
102 lines
4.3 KiB
Python
102 lines
4.3 KiB
Python
"""semif-serve acceptance on the real card. Contract § Acceptance.
|
|
|
|
Needs SemIf's authored144 rows and their committed torch predictions (same model revision,
|
|
torch 2.10.0+cu128) from a SemIf checkout at the pinned commit:
|
|
SEMIF_DIR=<checkout> SEMIF_URL=http://10.251.50.54:8032 SEMIF_TOKEN=... \
|
|
uv run --with httpx python accept.py out.json
|
|
|
|
1 parity our /decide vs their direct-authored144 predictions: top choice, max |Δp|, prompt hash
|
|
2 noise floor the same 144 rows again (A-vs-A)
|
|
3 negative option descriptions rotated one place: agreement with the reference MUST drop
|
|
4 shared one state, many criteria through /decide/shared vs /decide on the same rows
|
|
5 speed 21 binary criteria over one state, shared vs 21 sequential /decide, 3 runs, after warm-up
|
|
"""
|
|
import json
|
|
import os
|
|
import statistics
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
|
|
URL, TOKEN, SEMIF = os.environ["SEMIF_URL"], os.environ["SEMIF_TOKEN"], Path(os.environ["SEMIF_DIR"])
|
|
H = {"Authorization": f"Bearer {TOKEN}"}
|
|
client = httpx.Client(timeout=300)
|
|
|
|
|
|
def row(r):
|
|
return {k: r[k] for k in ("id", "state", "question", "options")}
|
|
|
|
|
|
def decide(r):
|
|
resp = client.post(f"{URL}/decide", json=row(r), headers=H)
|
|
resp.raise_for_status()
|
|
return resp.json()
|
|
|
|
|
|
def argmax(p):
|
|
return max(range(len(p)), key=p.__getitem__)
|
|
|
|
|
|
def compare(ours, ref):
|
|
agree = sum(argmax(ours[i]["probabilities"]) == argmax(ref[i]["probabilities"]) for i in ref)
|
|
gap = max(max(abs(a - b) for a, b in zip(ours[i]["probabilities"], ref[i]["probabilities"])) for i in ref)
|
|
return {"rows": len(ref), "top_choice_agree": agree, "max_abs_prob_gap": gap}
|
|
|
|
|
|
rows = [json.loads(l) for l in (SEMIF / "benchmarks/data/authored144.jsonl").read_text().splitlines() if l.strip()]
|
|
ref = {p["id"]: p for p in map(json.loads, (SEMIF / "results/raw/predictions/direct-authored144.jsonl").read_text().splitlines()) if p}
|
|
report = {"url": URL, "health": client.get(f"{URL}/health").json()}
|
|
|
|
run_a = {r["id"]: decide(r) for r in rows}
|
|
run_b = {r["id"]: decide(r) for r in rows}
|
|
report["1_parity_vs_upstream"] = compare(run_a, ref)
|
|
report["1_prompt_sha256_equal"] = sum(run_a[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ref)
|
|
report["2_noise_floor_a_vs_b"] = compare(run_a, run_b)
|
|
|
|
rotated = []
|
|
for r in rows:
|
|
descs = [o["description"] for o in r["options"]]
|
|
descs = descs[1:] + descs[:1]
|
|
rotated.append({**r, "options": [{**o, "description": d} for o, d in zip(r["options"], descs)]})
|
|
report["3_negative_rotated_options"] = compare({r["id"]: decide(r) for r in rotated}, ref)
|
|
|
|
# 4: group authored144 by identical state; score every multi-row group both ways.
|
|
groups = {}
|
|
for r in rows:
|
|
groups.setdefault(json.dumps(r["state"], sort_keys=True), []).append(r)
|
|
multi = [g for g in groups.values() if len(g) > 1]
|
|
shared_out = {}
|
|
for g in multi:
|
|
resp = client.post(f"{URL}/decide/shared", headers=H, json={
|
|
"state": g[0]["state"], "decisions": [{k: r[k] for k in ("id", "question", "options")} for r in g]})
|
|
resp.raise_for_status()
|
|
shared_out.update({res["id"]: res for res in resp.json()["results"]})
|
|
report["4_shared_vs_direct"] = {"groups": len(multi), **compare(shared_out, {i: run_a[i] for i in shared_out})}
|
|
|
|
# 5: 21 binary criteria over one state
|
|
state = rows[0]["state"]
|
|
crit = [{"id": f"c{i}", "question": f"Does the evidence mention item number {i}?",
|
|
"options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]} for i in range(21)]
|
|
for _ in range(2): # warm-up
|
|
client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit}).raise_for_status()
|
|
shared_s, seq_s, prefix = [], [], None
|
|
for _ in range(3):
|
|
t = time.perf_counter()
|
|
resp = client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit})
|
|
resp.raise_for_status()
|
|
shared_s.append(time.perf_counter() - t)
|
|
prefix = resp.json()["timing"]["prefix_tokens"]
|
|
t = time.perf_counter()
|
|
for c in crit:
|
|
decide({**c, "state": state})
|
|
seq_s.append(time.perf_counter() - t)
|
|
report["5_speed_21_binary"] = {
|
|
"prefix_tokens": prefix,
|
|
"shared_s": {"runs": shared_s, "median": statistics.median(shared_s)},
|
|
"sequential_decide_s": {"runs": seq_s, "median": statistics.median(seq_s)},
|
|
}
|
|
json.dump(report, open(sys.argv[1], "w"), indent=1)
|
|
print(json.dumps({k: v for k, v in report.items() if k != "health"}, indent=1))
|