feat(blender-run): --cpu runs with no GPU attached (modelling/IO jobs stay off GPU 3)

This commit is contained in:
vh
2026-09-28 14:23:12 -07:00
parent 5e6a6f3b26
commit a51188c76a
2 changed files with 16 additions and 4 deletions
+5
View File
@@ -48,6 +48,11 @@ scripts/blender-run -- --python-expr 'import bpy; print(bpy.app.version_string)'
screen datablock's VIEW_3D area. Worked calls for every add-on:
`scripts/blender-probes/extensions_acceptance.py`. Full notes and foot-guns: the stack README,
section "Extensions".
- **`--cpu`** attaches no GPU at all (no nvidia runtime), so the run never touches GPU 3. Use it
for modelling, add-ons and import/export, which is anything that does not render. It is proven
for those (the extension acceptance ran that way) and for Cycles on CPU. EEVEE and Workbench
are expected to fail without the GPU (untested). The container hostname is always
`fv-ml1-blender`.
- **Budget:** each run is capped at 64 GB RAM and 48 CPUs, with up to the whole 96 GB of VRAM.
Keep to about 2 concurrent renders. It is not on irv-ml1, so irv-ml1's working-set budget
does not apply.
+11 -4
View File
@@ -2,7 +2,7 @@
# blender-run — one-shot HEADLESS Blender on fv-ml1 GPU 3, for scripted/CLI callers (draupnir etc.).
# The agent-driven, interactive path is scripts/blender-mcp; this is the batch path.
#
# scripts/blender-run [--job DIR] [--extensions] -- <blender args after -b>
# scripts/blender-run [--job DIR] [--extensions] [--cpu] -- <blender args after -b>
# scripts/blender-run --job /mnt/smithy/draupnir/j42 -- --python render.py -- --out out.png
# scripts/blender-run --extensions -- --python-expr 'import bpy; print(bpy.app.version_string)'
#
@@ -30,6 +30,12 @@
# per distinct DIR (small files, on /tank). Without --job, the only paths Blender sees are under
# /work (= fv-ml1:/tank/blender).
#
# --cpu: no GPU attached at all (no nvidia runtime), so the run never touches GPU 3. For modelling,
# add-on work, import/export and Cycles on CPU (the 2026-09-28 extension acceptance ran this way).
# EEVEE and Workbench render through EGL on the GPU and are expected to fail here (untested). It is
# also the shape of the fallback if GPU 3 goes back to vLLM. Asked for by draupnir, whose design
# stage never renders (2026-09-28).
#
# The container's hostname is fixed as fv-ml1-blender, so socket.gethostname() is stable across
# runs (callers record it as provenance; draupnir, 2026-09-28). Without it, it is the container id.
#
@@ -42,6 +48,7 @@ HOST=${BLENDER_SSH_HOST:-infra-ops@10.251.50.54}
ENV_FILE=/opt/docker/compose/blender/.env
JOB=""
EXT=""
GPU="--runtime nvidia -e NVIDIA_VISIBLE_DEVICES=3 -e NVIDIA_DRIVER_CAPABILITIES=all"
while [ $# -gt 0 ]; do
case "$1" in
--job) JOB=${2:?--job needs a directory}; shift 2 ;;
@@ -51,8 +58,9 @@ while [ $# -gt 0 ]; do
EXT="--mount type=bind,src=/tank/blender-extensions/5.2/system,dst=/blender/5.2/extensions/system,readonly \
--mount type=bind,src=/opt/docker/conf/blender/scripts/startup/fleet_extensions.py,dst=/fleet/fleet_extensions.py,readonly"
shift ;;
--cpu) GPU=""; shift ;;
--) shift; break ;;
-h|--help) sed -n 2,38p "$0"; exit 0 ;;
-h|--help) sed -n 2,43p "$0"; exit 0 ;;
*) echo "blender-run: unknown option $1 (blender args go after --)" >&2; exit 2 ;;
esac
done
@@ -79,8 +87,7 @@ PRE=""
[ -n "$EXT" ] && PRE="--python /fleet/fleet_extensions.py"
set +e
ssh -n -o BatchMode=yes "$HOST" "IMG=\$(grep '^IMAGE=' $ENV_FILE | cut -d= -f2) && \
exec docker run --rm --name blender-run-\$\$ --hostname fv-ml1-blender --runtime nvidia \
-e NVIDIA_VISIBLE_DEVICES=3 -e NVIDIA_DRIVER_CAPABILITIES=all \
exec docker run --rm --name blender-run-\$\$ --hostname fv-ml1-blender $GPU \
--user 1002:1003 -e HOME=/tmp -e USER=infra-ops -e LOGNAME=infra-ops --memory 64g --cpus 48 \
-v /tank/blender:/work $EXT -w $WORKDIR --entrypoint /blender/blender \"\$IMG\" \
-b --factory-startup --python-exit-code 1 $PRE $ARGS"