Files
esh-pfi-infrastructure/services/intern-decision-serve/acceptance/gpu3-2026-09-30/run.sh
T
vh a262477a61 feat(intern-decision): stack, DNS and GPU 3 acceptance for the SemIf replacement
stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob,
healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README.
dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3).
acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows
(pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across
restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and
recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
2026-09-30 09:38:00 -07:00

35 lines
2.2 KiB
Bash
Executable File

#!/bin/bash
# run.sh <label> <cap_gib> <conditions|full>: one process lifetime of intern-decision-serve on GPU 3.
set -u
cd /tmp/intern-accept
label=$1 cap=$2 mode=$3
O=out/$label; mkdir -p "$O"
[ "$(nvidia-smi -i 3 --query-gpu=memory.used --format=csv,noheader,nounits)" -gt 64 ] && { echo "GPU 3 not empty"; exit 10; }
docker run -d --name intern-decision-accept --gpus '"device=3"' -p 18033:8000 --env-file env \
-e INTERN_DECISION_DEVICE=cuda -e INTERN_DECISION_VRAM_CAP_GIB="$cap" \
-v /tank/aimodels/huggingface:/hf:ro intern-decision-serve:0.1.0 > "$O/container.id"
for i in $(seq 1 60); do
[ "$(curl -s -o /dev/null -w '%{http_code}' localhost:18033/health)" = 200 ] && break
[ "$(docker inspect -f '{{.State.Running}}' intern-decision-accept)" != true ] && { docker logs intern-decision-accept > "$O/server.log" 2>&1; echo "died"; exit 11; }
sleep 2
done
./poll.sh intern-decision-accept "$O/vram.csv" & POLL=$!
sleep 3
curl -s localhost:18033/health > "$O/health-rest.json"
echo "$(date +%s.%N) sets:start" >> "$O/events.txt"
if [ "$mode" = full ]; then conds=single,rotations,repeat,reversed,shuffled,negative,multifield; else conds=$mode; fi
python3 bench_sets.py --backend semif --url http://127.0.0.1:18033 --token-file token --label "$label" --out "$O/sets.json" --conditions "$conds" 2>&1 | tee "$O/sets.log"
echo "$(date +%s.%N) sets:done" >> "$O/events.txt"
if [ "$mode" = full ]; then
python3 bench_shape.py --backend semif --url http://127.0.0.1:18033 --token-file token --long-state-file long_state.txt --label "$label" --out "$O/shape.json" 2>&1 | tee "$O/shape.log"
echo "$(date +%s.%N) shape:done" >> "$O/events.txt"
python3 checks.py --url http://127.0.0.1:18033 --token-file token --out "$O/checks.json" --checks auth,chunking,queue,fit,maxreq,oom --long-state-file long_state.txt 2>&1 | tee "$O/checks.log"
echo "$(date +%s.%N) checks:done" >> "$O/events.txt"
fi
curl -s localhost:18033/health > "$O/health-end.json"
kill $POLL; wait $POLL 2>/dev/null
docker logs intern-decision-accept > "$O/server.log" 2>&1
docker rm -f intern-decision-accept > /dev/null
sleep 3
echo "gpu3 after: $(nvidia-smi -i 3 --query-gpu=memory.used --format=csv,noheader)" | tee -a "$O/events.txt"