Files
esh-pfi-infrastructure/stacks/mia/mia_driver.py
T
vh f44280e6b2 feat(mia): one-shot Make-It-Animatable v2 auto-rigger on fv-ml1 GPU 3 (scripts/mia-run)
Image local/mia:0.1.0 built from stacks/mia: MIA v2 @ bbd8b158 (MIT) with
its pinned submodules, dread-dev's proven Python lock with the torch family
swapped to cu129, and a driver adapted from dread-dev's run_mia.py that
seeds every mesh (fix_random + trimesh's module RNG) and writes
weights_effective into the npz. Weights stay in fv-ml1's shared HF cache
at pinned revisions, mounted read-only.

scripts/mia-run mirrors blender-run: --job DIR is shipped to
fv-ml1:/tank/mia/jobs, one docker run --rm rigs every mesh, out/ comes back.

Acceptance on the four Dread Naught characters: 3.9-4.7 s a mesh (median
of 3) plus 12.7 s model load, peak 3,394 MiB; seeded runs bit-identical
across rotated mesh order; GPU-vs-CPU distances the same size as sampling
noise, with an unseeded GPU run as the positive control.
2026-10-01 09:45:22 -07:00

263 lines
11 KiB
Python

"""Fleet driver for Make-It-Animatable v2 (app_v2.py): one container run, any number of meshes.
Usage (inside the local/mia image; scripts/mia-run is the caller-facing wrapper):
python mia_driver.py [--seed N | --unseeded] <name>=<input.glb> [<name>=<input.glb> ...]
Inputs are paths relative to the working directory (the job dir). Per mesh it writes to out/:
<name>_pred.npz predictions in the INPUT file's coordinates, plus `weights_effective`
<name>.fbx / .glb MIA's FBX (Blender export) and its FBX2glTF preview
<name>_rest.glb Blender glTF export of the rigged rest pose
<name>_apose-hint.* the same predictions, Blender stage re-run with "Input Rest Pose = A-pose"
<name>_run.json stage timings, peak GPU memory, seed, provenance
Adapted from dread-dev's tools/mia/run_mia.py + finalize.py (dreadnaught repo, 2026-10-01). The
pipeline calls, the npz keys and the artifact set are theirs, unchanged. Differences:
- Paths: repo at /opt/mia/repo; MIA's scratch (the input copy and the work dir it writes beside
it) lives in a temp dir that dies with the container, so only out/ comes back to the caller.
- SEEDED BY DEFAULT. Before EVERY mesh: util.utils.fix_random(seed) (python/numpy/torch seeds,
cudnn deterministic), and trimesh's module RNG reset to default_rng(seed). trimesh >= 4 samples
surface points from trimesh.util._RANDOM_DEFAULT, which fix_random never reaches; unseeded runs
moved joints 0.1-0.27% of height (dread-dev's CPU noise floor). Resetting per mesh also makes a
mesh's result independent of which meshes ran before it in the same process.
--unseeded restores upstream behaviour (fix_random once at model load, trimesh unseeded).
- weights_effective is written straight into the npz (finalize.py's formula: weights > 1e-3 kept,
rows renormalised; what the FBX vertex groups and the GLB WEIGHTS_0 actually carry).
- Each stage is followed by torch.cuda.synchronize() so the per-stage wall times are honest.
"""
import argparse
import json
import os
import re
import shutil
import sys
import tempfile
import time
REPO = "/opt/mia/repo"
JOB = os.getcwd()
OUT_DIR = os.path.join(JOB, "out")
NAME_RE = re.compile(r"^[A-Za-z0-9._-]+$")
def parse_args():
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
g = ap.add_mutually_exclusive_group()
g.add_argument("--seed", type=int, default=0, help="seed for every mesh (default 0)")
g.add_argument("--unseeded", action="store_true", help="upstream behaviour: trimesh sampling unseeded")
ap.add_argument("meshes", nargs="+", metavar="NAME=INPUT.glb")
a = ap.parse_args()
jobs = []
for m in a.meshes:
if "=" not in m:
ap.error(f"{m!r}: expected NAME=INPUT.glb")
name, path = m.split("=", 1)
if not NAME_RE.match(name):
ap.error(f"{name!r}: NAME must be [A-Za-z0-9._-]+")
src = os.path.realpath(os.path.join(JOB, path))
if not src.startswith(JOB + os.sep):
ap.error(f"{path!r}: inputs must be inside the job dir")
if not os.path.isfile(src):
ap.error(f"{path!r}: no such file in the job dir")
jobs.append((name, src))
if len({n for n, _ in jobs}) != len(jobs):
ap.error("NAMEs must be unique")
return a, jobs
args, JOBS = parse_args()
SEED = None if args.unseeded else args.seed
os.environ.setdefault("GRADIO_ANALYTICS_ENABLED", "False")
os.environ["HF_HUB_OFFLINE"] = "1" # all weights are on the read-only /hf mount; never touch the network
sys.path.insert(0, REPO)
os.chdir(REPO)
import numpy as np # noqa: E402
import torch # noqa: E402
import trimesh.util # noqa: E402
t0 = time.perf_counter()
import app_v2 as A # noqa: E402
from util.utils import fix_random # noqa: E402
A.init_models()
A.init_blocks() # only defines the Gradio component globals the stage functions return as dict keys; no server
t_init = time.perf_counter() - t0
CUDA = torch.cuda.is_available()
print(f"[mia] init (imports + 3 models + blocks): {t_init:.1f}s, device {'cuda:' + torch.cuda.get_device_name(0) if CUDA else 'cpu'}, seed {SEED}", flush=True)
STAGES = ["prepare_input", "preprocess", "infer", "vis", "vis_blender", "finish"]
BONE_NAMES = [None] * len(A.BONES_IDX_DICT)
for k, v in A.BONES_IDX_DICT.items():
BONE_NAMES[v] = k
PARENTS = list(A.KINEMATIC_TREE.parent_indices)
PROVENANCE = dict(line.strip().split("=", 1) for line in open("/opt/mia/provenance") if "=" in line)
def sync():
if CUDA:
torch.cuda.synchronize()
def to_np(x):
if isinstance(x, torch.Tensor):
x = x.detach().cpu().numpy()
return np.asarray(x)
def run_one(name: str, src: str, scratch: str):
if SEED is not None:
fix_random(SEED)
trimesh.util._RANDOM_DEFAULT = np.random.default_rng(SEED)
if CUDA:
torch.cuda.reset_peak_memory_stats()
inp = os.path.join(scratch, f"{name}.glb")
shutil.copyfile(src, inp) # MIA writes its work dir next to the input file, so run on a copy
work = os.path.join(scratch, name)
db = A.DB()
gen = A._pipeline(
input_path=inp,
is_gs=False,
opacity_threshold=0.01,
no_fingers=False, # UI default
rest_pose_type="No", # UI default
ignore_pose_parts=[], # UI default
input_normal=True, # UI default (fixed)
bw_fix=True, # UI default: weight post-processing on
bw_vis_bone="LeftArm", # UI default (visualisation only)
restore_global=False, # UI default: outputs in MIA's normalised frame
reset_to_rest=True, # UI default: apply predicted T-pose as the rest pose
animation_file=None, # deviation from UI default ("Standard Run.fbx"): static rig, no animation baked
retarget=True,
inplace=True,
db=db,
)
timings = {}
snap = {}
for stage in STAGES:
w0 = time.perf_counter()
next(gen)
sync()
timings[stage] = {"wall_s": time.perf_counter() - w0}
print(f"[mia] {name}: {stage} {timings[stage]['wall_s']:.2f}s", flush=True)
if stage == "prepare_input":
snap["verts_input"] = to_np(db.verts)[0].copy()
snap["faces"] = np.asarray(db.faces).copy()
elif stage == "infer":
snap["bw_raw"] = to_np(db.bw)[0].copy()
try:
next(gen)
except StopIteration:
pass
# --- predictions in the INPUT file's coordinates ---
T = db.global_transform # pytorch3d Transform3d (row-vector): input -> MIA-normalised frame
Tinv = T.inverse()
T_mat = T.get_matrix().transpose(-1, -2)[0].cpu().numpy() # column-vector 4x4
Tinv_mat = Tinv.get_matrix().transpose(-1, -2)[0].cpu().numpy()
joints_n = np.asarray(db.joints, dtype=np.float32)
tails_n = np.asarray(db.joints_tail, dtype=np.float32)
dev = Tinv.device
joints_in = Tinv.transform_points(torch.from_numpy(joints_n)[None].to(dev))[0].cpu().numpy()
tails_in = Tinv.transform_points(torch.from_numpy(tails_n)[None].to(dev))[0].cpu().numpy()
verts_n = np.asarray(db.verts, dtype=np.float32)
verts_back = Tinv.transform_points(torch.from_numpy(verts_n)[None].to(dev))[0].cpu().numpy()
roundtrip_err = float(np.abs(verts_back - snap["verts_input"]).max())
pose_n = np.asarray(db.pose, dtype=np.float32)
pose_in = np.einsum("ij,kjl,lm->kim", Tinv_mat, pose_n, T_mat).astype(np.float32)
W = np.asarray(db.bw, dtype=np.float32)
We = np.where(W > 1e-3, W, 0).astype(np.float32) # finalize.py: MIA's set_weights threshold
We /= np.maximum(We.sum(1, keepdims=True), 1e-12)
np.savez_compressed(
os.path.join(OUT_DIR, f"{name}_pred.npz"),
verts=snap["verts_input"].astype(np.float32),
faces=snap["faces"].astype(np.int64),
weights=W,
weights_raw=snap["bw_raw"].astype(np.float32),
weights_effective=We,
bone_names=np.array(BONE_NAMES),
parents=np.array(PARENTS, dtype=np.int64),
joints_head=joints_in.astype(np.float32),
joints_tail=tails_in.astype(np.float32),
pose_to_rest=pose_in,
input_to_mia=T_mat.astype(np.float32),
joints_head_mia=joints_n,
joints_tail_mia=tails_n,
verts_mia=verts_n,
)
# --- MIA's own artifacts (default settings) ---
copies = {
f"{name}.fbx": db.anim_path, # MIA's primary output (Blender FBX export)
f"{name}.glb": db.anim_vis_path, # MIA's FBX2glTF conversion of the FBX (the app's "GLB preview")
f"{name}_rest.glb": db.rest_vis_path, # Blender glTF export of the rigged rest-pose model
}
missing = []
for dst, srcp in copies.items():
if srcp and os.path.isfile(srcp):
shutil.copyfile(srcp, os.path.join(OUT_DIR, dst))
else:
missing.append(dst)
# --- variant: same predictions, Blender stage re-run with the UI's "Input Rest Pose = A-pose" hint ---
w0 = time.perf_counter()
db.anim_path = os.path.join(work, f"{name}_apose-hint.fbx")
db.anim_vis_path = os.path.join(work, f"{name}_apose-hint.glb")
db.rest_vis_path = os.path.join(work, f"{name}_apose-hint_rest.glb")
A.vis_blender(
reset_to_rest=True,
remove_fingers=False,
rest_pose_type="A-pose",
ignore_pose_parts=[],
animation_file=None,
retarget=True,
inplace=True,
restore_global=False,
db=db,
)
timings["vis_blender_apose_hint"] = {"wall_s": time.perf_counter() - w0}
for suffix in ("_apose-hint.fbx", "_apose-hint.glb", "_apose-hint_rest.glb"):
srcp = os.path.join(work, f"{name}{suffix}")
if os.path.isfile(srcp):
shutil.copyfile(srcp, os.path.join(OUT_DIR, f"{name}{suffix}"))
else:
missing.append(f"{name}{suffix}")
meta = {
"name": name,
"source": os.path.relpath(src, JOB),
"n_verts": int(snap["verts_input"].shape[0]),
"n_faces": int(snap["faces"].shape[0]),
"seed": SEED,
"device": torch.cuda.get_device_name(0) if CUDA else "cpu",
"timings": timings,
"model_total_wall_s": sum(timings[s]["wall_s"] for s in ("preprocess", "infer")),
"pipeline_total_wall_s": sum(timings[s]["wall_s"] for s in STAGES),
"init_wall_s": t_init,
"gpu_peak_allocated_mib": torch.cuda.max_memory_allocated() / 2**20 if CUDA else None,
"gpu_peak_reserved_mib": torch.cuda.max_memory_reserved() / 2**20 if CUDA else None,
"roundtrip_err_input_to_mia_and_back": roundtrip_err,
"missing_artifacts": missing,
"provenance": PROVENANCE,
}
with open(os.path.join(OUT_DIR, f"{name}_run.json"), "w") as f:
json.dump(meta, f, indent=2)
print(f"[mia] {name}: done, pipeline {meta['pipeline_total_wall_s']:.1f}s, peak reserved "
f"{meta['gpu_peak_reserved_mib'] or 0:.0f} MiB, roundtrip err {roundtrip_err:.2e}"
+ (f", MISSING {missing}" if missing else ""), flush=True)
return meta
if __name__ == "__main__":
os.makedirs(OUT_DIR, exist_ok=True)
failed = 0
with tempfile.TemporaryDirectory(prefix="mia-") as scratch:
for name, src in JOBS:
meta = run_one(name, src, scratch)
failed += bool(meta["missing_artifacts"])
sys.exit(1 if failed else 0)