"""MIA GPU acceptance analysis: timings (median of 3 seeded runs), determinism, and GPU-vs-CPU agreement measured with dread-dev's noise_floor.py metrics against the CPU run-to-run floor.""" import itertools, json, statistics as st, sys, numpy as np A, C = sys.argv[1], sys.argv[2] MESHES = ["goblin", "villager", "captain", "ogre"] def load(p): d = np.load(p) if "weights_effective" not in d: raise SystemExit(f"{p}: no weights_effective") return d def metrics(Ar, Br): H = float(np.ptp(Ar["verts"][:, 1])) dj = np.linalg.norm(Ar["joints_head"] - Br["joints_head"], axis=1) / H dt = np.linalg.norm(Ar["joints_tail"] - Br["joints_tail"], axis=1) / H WA, WB = Ar["weights_effective"], Br["weights_effective"] moved = 0.5 * np.abs(WA - WB).sum(1) return dict(jh_med=float(np.median(dj)), jh_max=float(dj.max()), jt_max=float(dt.max()), w_moved_mean=float(moved.mean()), w_moved_p95=float(np.percentile(moved, 95)), dom_changed=float((WA.argmax(1) != WB.argmax(1)).mean())) def rng(ms, k): v = [m[k] for m in ms] return f"{min(v):.4f}..{max(v):.4f}" KEYS = ["jh_med", "jh_max", "jt_max", "w_moved_mean", "w_moved_p95", "dom_changed"] out = {} for n in MESHES: g = {r: load(f"{A}/{r}/out/{n}_pred.npz") for r in ("r1", "r2", "r3", "u1")} c = [load(f"{C}/{n}_pred.npz"), load(f"{C}/repeats/{n}_rep2_pred.npz"), load(f"{C}/repeats/{n}_rep3_pred.npz")] runs = {r: json.load(open(f"{A}/{r}/out/{n}_run.json")) for r in ("r1", "r2", "r3", "u1")} same_geom = all(np.array_equal(g["r1"]["verts"], x["verts"]) and np.array_equal(g["r1"]["faces"], x["faces"]) for x in c) # determinism across the 3 seeded runs (different mesh orders): exact? det = {k: max(float(np.abs(g["r1"][k] - g[r][k]).max()) for r in ("r2", "r3")) for k in ("joints_head", "joints_tail", "weights", "weights_effective", "pose_to_rest")} cpu_cpu = [metrics(a, b) for a, b in itertools.combinations(c, 2)] # the CPU noise floor (null) gpu_cpu = [metrics(g["r1"], x) for x in c] # the claim under test unseeded = [metrics(g["u1"], g[r]) for r in ("r1",)] # positive control: must move t = [runs[r]["pipeline_total_wall_s"] for r in ("r1", "r2", "r3")] tm = [runs[r]["model_total_wall_s"] for r in ("r1", "r2", "r3")] ta = [runs[r]["timings"]["vis_blender_apose_hint"]["wall_s"] for r in ("r1", "r2", "r3")] pk = [runs[r]["gpu_peak_reserved_mib"] for r in ("r1", "r2", "r3", "u1")] cpu_t = [json.load(open(p))["pipeline_total_wall_s"] for p in (f"{C}/{n}_run.json", f"{C}/repeats/{n}_rep2_run.json", f"{C}/repeats/{n}_rep3_run.json")] print(f"\n== {n}: verts {runs['r1']['n_verts']}, same geometry as CPU: {same_geom}") print(f" GPU pipeline s: {[round(x,2) for x in t]} median {st.median(t):.2f} | model part median {st.median(tm):.2f} | A-pose-hint export median {st.median(ta):.2f} | CPU pipeline median {st.median(cpu_t):.1f}") print(f" peak torch reserved MiB: {pk}") print(f" seeded r1/r2/r3 max abs diff: " + ", ".join(f"{k} {v:.2e}" for k, v in det.items())) for k in KEYS: print(f" {k:13s} CPU-vs-CPU {rng(cpu_cpu,k)} | GPU-vs-CPU {rng(gpu_cpu,k)} | unseeded-vs-seeded GPU {rng(unseeded,k)}") out[n] = dict(gpu_pipeline_s=t, gpu_model_s=tm, apose_hint_s=ta, cpu_pipeline_s=cpu_t, peak_reserved_mib=pk, seeded_maxdiff=det, cpu_cpu=cpu_cpu, gpu_cpu=gpu_cpu, unseeded_vs_seeded=unseeded, same_geometry=same_geom) init = [json.load(open(f"{A}/{r}/out/goblin_run.json"))["init_wall_s"] for r in ("r1", "r2", "r3", "u1")] print("\nmodel load (init) s per container:", [round(x, 1) for x in init]) json.dump(out, open(f"{A}/acceptance.json", "w"), indent=1)