Image local/mia:0.1.0 built from stacks/mia: MIA v2 @ bbd8b158 (MIT) with its pinned submodules, dread-dev's proven Python lock with the torch family swapped to cu129, and a driver adapted from dread-dev's run_mia.py that seeds every mesh (fix_random + trimesh's module RNG) and writes weights_effective into the npz. Weights stay in fv-ml1's shared HF cache at pinned revisions, mounted read-only. scripts/mia-run mirrors blender-run: --job DIR is shipped to fv-ml1:/tank/mia/jobs, one docker run --rm rigs every mesh, out/ comes back. Acceptance on the four Dread Naught characters: 3.9-4.7 s a mesh (median of 3) plus 12.7 s model load, peak 3,394 MiB; seeded runs bit-identical across rotated mesh order; GPU-vs-CPU distances the same size as sampling noise, with an unseeded GPU run as the positive control.
56 lines
3.7 KiB
Python
56 lines
3.7 KiB
Python
"""MIA GPU acceptance analysis: timings (median of 3 seeded runs), determinism, and GPU-vs-CPU
|
|
agreement measured with dread-dev's noise_floor.py metrics against the CPU run-to-run floor."""
|
|
import itertools, json, statistics as st, sys, numpy as np
|
|
A, C = sys.argv[1], sys.argv[2]
|
|
MESHES = ["goblin", "villager", "captain", "ogre"]
|
|
|
|
def load(p):
|
|
d = np.load(p)
|
|
if "weights_effective" not in d:
|
|
raise SystemExit(f"{p}: no weights_effective")
|
|
return d
|
|
|
|
def metrics(Ar, Br):
|
|
H = float(np.ptp(Ar["verts"][:, 1]))
|
|
dj = np.linalg.norm(Ar["joints_head"] - Br["joints_head"], axis=1) / H
|
|
dt = np.linalg.norm(Ar["joints_tail"] - Br["joints_tail"], axis=1) / H
|
|
WA, WB = Ar["weights_effective"], Br["weights_effective"]
|
|
moved = 0.5 * np.abs(WA - WB).sum(1)
|
|
return dict(jh_med=float(np.median(dj)), jh_max=float(dj.max()), jt_max=float(dt.max()),
|
|
w_moved_mean=float(moved.mean()), w_moved_p95=float(np.percentile(moved, 95)),
|
|
dom_changed=float((WA.argmax(1) != WB.argmax(1)).mean()))
|
|
|
|
def rng(ms, k):
|
|
v = [m[k] for m in ms]
|
|
return f"{min(v):.4f}..{max(v):.4f}"
|
|
|
|
KEYS = ["jh_med", "jh_max", "jt_max", "w_moved_mean", "w_moved_p95", "dom_changed"]
|
|
out = {}
|
|
for n in MESHES:
|
|
g = {r: load(f"{A}/{r}/out/{n}_pred.npz") for r in ("r1", "r2", "r3", "u1")}
|
|
c = [load(f"{C}/{n}_pred.npz"), load(f"{C}/repeats/{n}_rep2_pred.npz"), load(f"{C}/repeats/{n}_rep3_pred.npz")]
|
|
runs = {r: json.load(open(f"{A}/{r}/out/{n}_run.json")) for r in ("r1", "r2", "r3", "u1")}
|
|
same_geom = all(np.array_equal(g["r1"]["verts"], x["verts"]) and np.array_equal(g["r1"]["faces"], x["faces"]) for x in c)
|
|
# determinism across the 3 seeded runs (different mesh orders): exact?
|
|
det = {k: max(float(np.abs(g["r1"][k] - g[r][k]).max()) for r in ("r2", "r3"))
|
|
for k in ("joints_head", "joints_tail", "weights", "weights_effective", "pose_to_rest")}
|
|
cpu_cpu = [metrics(a, b) for a, b in itertools.combinations(c, 2)] # the CPU noise floor (null)
|
|
gpu_cpu = [metrics(g["r1"], x) for x in c] # the claim under test
|
|
unseeded = [metrics(g["u1"], g[r]) for r in ("r1",)] # positive control: must move
|
|
t = [runs[r]["pipeline_total_wall_s"] for r in ("r1", "r2", "r3")]
|
|
tm = [runs[r]["model_total_wall_s"] for r in ("r1", "r2", "r3")]
|
|
ta = [runs[r]["timings"]["vis_blender_apose_hint"]["wall_s"] for r in ("r1", "r2", "r3")]
|
|
pk = [runs[r]["gpu_peak_reserved_mib"] for r in ("r1", "r2", "r3", "u1")]
|
|
cpu_t = [json.load(open(p))["pipeline_total_wall_s"] for p in (f"{C}/{n}_run.json", f"{C}/repeats/{n}_rep2_run.json", f"{C}/repeats/{n}_rep3_run.json")]
|
|
print(f"\n== {n}: verts {runs['r1']['n_verts']}, same geometry as CPU: {same_geom}")
|
|
print(f" GPU pipeline s: {[round(x,2) for x in t]} median {st.median(t):.2f} | model part median {st.median(tm):.2f} | A-pose-hint export median {st.median(ta):.2f} | CPU pipeline median {st.median(cpu_t):.1f}")
|
|
print(f" peak torch reserved MiB: {pk}")
|
|
print(f" seeded r1/r2/r3 max abs diff: " + ", ".join(f"{k} {v:.2e}" for k, v in det.items()))
|
|
for k in KEYS:
|
|
print(f" {k:13s} CPU-vs-CPU {rng(cpu_cpu,k)} | GPU-vs-CPU {rng(gpu_cpu,k)} | unseeded-vs-seeded GPU {rng(unseeded,k)}")
|
|
out[n] = dict(gpu_pipeline_s=t, gpu_model_s=tm, apose_hint_s=ta, cpu_pipeline_s=cpu_t, peak_reserved_mib=pk,
|
|
seeded_maxdiff=det, cpu_cpu=cpu_cpu, gpu_cpu=gpu_cpu, unseeded_vs_seeded=unseeded, same_geometry=same_geom)
|
|
init = [json.load(open(f"{A}/{r}/out/goblin_run.json"))["init_wall_s"] for r in ("r1", "r2", "r3", "u1")]
|
|
print("\nmodel load (init) s per container:", [round(x, 1) for x in init])
|
|
json.dump(out, open(f"{A}/acceptance.json", "w"), indent=1)
|