feat(lora-worker): stand up in-arbo LoRA training worker on irv-ml1 (arbo Phase 1 §4.1)

Host service (runs as llmuser, owns /opt/fluxgym + GPU access) that runs
sd-scripts SDXL LoRA training on demand for arbo — the infra-ops half of the
in-arbo LoRA training Phase 1 ownership split (vh/arbo
docs/contracts/in-arbo-lora-training-phase1.contract.md §4.1/§2).

- Fixed-invocation only (INV-T7): bounded params -> one sd-scripts command
  shape; every param range/allowlist/path-containment checked before spawn;
  bad request = 422, never a silent downgrade. 14 unit tests green.
- Thin supervisor: never imports torch; subprocesses the fluxgym venv's
  accelerate. 1-job-at-a-time (arbo lease is the serializer, 409 is backstop).
  Durable job records + boot reconciliation (§4.3).
- API: POST /train, GET /train/{id}[/log], POST /train/{id}/cancel,
  GET /gpu-status (per-device VRAM + tts_on_3090 co-OOM signal), GET /healthz.
- Wire-shape (§7 resolved with comfy-dev): shared /worktank/arbo/train handoff
  (group arbotrain, setgid 2770); worker binds 0.0.0.0:8203, arbo reaches via
  host.docker.internal:host-gateway (reachability proven on 172.20.0.1:8203);
  device-aware TTS steering via /gpu-status.

Deployed to irv-ml1 via playbooks/deploy-lora-training-worker.yaml (elway,
idempotent); systemd unit active; /healthz + /gpu-status verified live.
This commit is contained in:
vh
2026-07-06 18:28:04 -07:00
parent 5c64d31094
commit 888ba6a714
13 changed files with 1883 additions and 0 deletions
@@ -0,0 +1,131 @@
"""Tests for the fixed-invocation builder — the INV-T7 enforcement surface.
These assert that (a) a valid request produces exactly the proven §4.6 command shape, and
(b) every out-of-bounds / unsafe / mis-fit request is REJECTED before any argv is produced.
Pure functions, no GPU, no subprocess — runnable anywhere.
"""
import pytest
from worker import config
from worker.invocation import InvalidTrainRequest, build_command
def _req(**over):
base = {
"dataset_dir": "/worktank/arbo/train/t123/dataset",
"base_model_path": "/opt/fluxgym/models/sdxl/base.safetensors",
"output_dir": "/worktank/arbo/train/t123/out",
"output_name": "char_t123",
"trigger": "ohwx",
"subject_class": "woman",
"repeats": 10,
"tier": "balanced",
"device_index": 0,
"seed": 42,
}
base.update(over)
return base
def _flag_value(argv, flag):
return argv[argv.index(flag) + 1]
# ---- happy path ------------------------------------------------------------------------
def test_balanced_on_3090_builds_proven_shape():
argv, env, params = build_command(_req(tier="balanced", device_index=0))
assert str(config.ACCELERATE_BIN) == argv[0]
assert "launch" == argv[1]
assert str(config.SDXL_TRAIN_SCRIPT) in argv
# tier table (§4.6): balanced -> 1500 steps, dim 32, res 768
assert _flag_value(argv, "--max_train_steps") == "1500"
assert _flag_value(argv, "--network_dim") == "32"
assert _flag_value(argv, "--resolution") == "768,768"
# the load-bearing lean flags + caption gotcha
assert "--network_train_unet_only" in argv
assert "--gradient_checkpointing" in argv
assert _flag_value(argv, "--caption_extension") == ".txt"
assert _flag_value(argv, "--optimizer_type") == "adamw8bit"
# env: PCI_BUS_ID ordering + the assigned device + allocator
assert env["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID"
assert env["CUDA_VISIBLE_DEVICES"] == "0"
assert env["PYTORCH_CUDA_ALLOC_CONF"] == "expandable_segments:True"
assert params["steps"] == 1500
def test_quality_on_a6000_ok_1024():
argv, env, _ = build_command(_req(tier="quality", device_index=1))
assert _flag_value(argv, "--resolution") == "1024,1024"
assert _flag_value(argv, "--max_train_steps") == "3000"
assert env["CUDA_VISIBLE_DEVICES"] == "1"
def test_paths_and_names_flow_through():
argv, _, _ = build_command(_req(output_name="my_char", trigger="ohwx"))
assert _flag_value(argv, "--output_name") == "my_char"
assert _flag_value(argv, "--train_data_dir") == "/worktank/arbo/train/t123/dataset"
# ---- rejections (INV-T7 / §4.6 tier-device fit) ----------------------------------------
def test_quality_on_3090_rejected():
with pytest.raises(InvalidTrainRequest, match="quality"):
build_command(_req(tier="quality", device_index=0))
def test_unknown_tier_rejected():
with pytest.raises(InvalidTrainRequest, match="tier"):
build_command(_req(tier="ultra"))
def test_bad_device_index_rejected():
with pytest.raises(InvalidTrainRequest, match="device_index"):
build_command(_req(device_index=3))
def test_dataset_dir_outside_handoff_root_rejected():
with pytest.raises(InvalidTrainRequest, match="dataset_dir"):
build_command(_req(dataset_dir="/etc/passwd"))
def test_path_traversal_rejected():
with pytest.raises(InvalidTrainRequest, match="dataset_dir"):
build_command(_req(dataset_dir="/worktank/arbo/train/../../etc/shadow"))
def test_base_model_outside_allowed_roots_rejected():
with pytest.raises(InvalidTrainRequest, match="base_model_path"):
build_command(_req(base_model_path="/home/someone/evil.safetensors"))
def test_unsafe_output_name_rejected():
with pytest.raises(InvalidTrainRequest, match="output_name"):
build_command(_req(output_name="a; rm -rf /"))
def test_unsafe_trigger_rejected():
with pytest.raises(InvalidTrainRequest, match="trigger"):
build_command(_req(trigger="$(whoami)"))
def test_repeats_out_of_range_rejected():
with pytest.raises(InvalidTrainRequest, match="repeats"):
build_command(_req(repeats=0))
with pytest.raises(InvalidTrainRequest, match="repeats"):
build_command(_req(repeats=1000))
def test_relative_dataset_dir_rejected():
with pytest.raises(InvalidTrainRequest, match="absolute"):
build_command(_req(dataset_dir="relative/path"))
def test_alpha_scales_with_dim():
_, _, params = build_command(_req(tier="fast"))
# fast -> dim 16; alpha = dim * ratio (default 1.0) = 16
argv, _, _ = build_command(_req(tier="fast"))
assert _flag_value(argv, "--network_dim") == "16"
assert _flag_value(argv, "--network_alpha") == "16"
assert params["dim"] == 16