#!/usr/bin/env bash # ───────────────────────────────────────────────────────────────────────────── # Reproducible build of the CUSTOM llama.cpp that backs the char-rp-reasoning # seat (Deckard-PKD, ana-ml2:8018). # # = llama.cpp latest master (pinned 6eddde0, 2026-07-13, ~b9990 era) # + UNMERGED PR #25544 (aldehir:reasoning-budget-multi-seq) # # WHY THIS EXISTS (do not "clean this up" without reading README.md): # Stock llama.cpp b8840 (and every released build to date) has a reasoning- # budget sampler that knows only ONE end-tag (). Qwen3.5's tool-call # path terminates reasoning with , which the single-tag sampler # cannot match, so --reasoning-budget forcing NEVER fires on the tool-retry # path → the model loops in reasoning to max_tokens (32768) ≈ 22 min. That is # Worldtree #355's residual. PR #25544 teaches the budget MULTIPLE terminating # sequences ( OR ) — the exact fix — but it is UNMERGED # upstream, so we build it ourselves. # # ⚠️ REMOVE-WHEN-MERGED: once #25544 merges upstream and lands in a release, # retire this custom build and switch the seat back to a stock image # (see README.md → "Retiring this custom build"). # # Run on ana-ml2 (Blackwell RTX PRO 6000, sm_120; docker + buildx; ~120G free). # Produces: llamacpp-charrp:6eddde0-pr25544 (+ :custom-latest) # Runtime ~20-40 min (CUDA compile for sm_120). # ───────────────────────────────────────────────────────────────────────────── set -euo pipefail MASTER_SHA="${MASTER_SHA:-6eddde0}" # llama.cpp master pin PR="${PR:-25544}" # aldehir:reasoning-budget-multi-seq BUILDDIR="${BUILDDIR:-/home/lkraven/llamacpp-build}" IMAGE="${IMAGE:-llamacpp-charrp}" CUDA_VERSION="${CUDA_VERSION:-12.8.1}" # 12.8+ required for Blackwell sm_120 CUDA_ARCH="${CUDA_ARCH:-120}" # RTX PRO 6000 = compute 12.0 echo "== fetch master@${MASTER_SHA} + PR #${PR} ==" rm -rf "$BUILDDIR" git clone https://github.com/ggml-org/llama.cpp "$BUILDDIR" cd "$BUILDDIR" git checkout "$MASTER_SHA" git fetch origin "pull/${PR}/head:pr${PR}" git config user.email infra-ops@phasefinal.com git config user.name infra-ops echo "== merge PR #${PR} (expect 1 conflict in server-common.cpp) ==" git merge --no-commit --no-ff "pr${PR}" || true # Resolve the single conflict: keep the PR's PLURAL reasoning_budget_end_tags # (the whole point of the fix) AND master's per-request body-read of # reasoning_budget_message (a newer master feature the PR's base predates). python3 - <<'PY' p = "tools/server/server-common.cpp" s = open(p).read() i = s.index("<<<<<<< HEAD") j = s.index(">>>>>>> pr25544") + len(">>>>>>> pr25544") new = (' llama_params["reasoning_budget_end_tags"] = chat_params.thinking_end_tags;\n' ' llama_params["reasoning_budget_message"] = json_value(body, "reasoning_budget_message", opt.reasoning_budget_message);') open(p, "w").write(s[:i] + new + s[j:]) assert open(p).read().count("<<<<<<<") == 0, "unresolved conflict markers remain" print("resolved server-common.cpp") PY git add -A git commit -m "merge PR #${PR} (reasoning-budget multi-seq) onto master ${MASTER_SHA} — char-rp-reasoning seat custom build" echo "== docker build (server target, CUDA ${CUDA_VERSION}, sm_${CUDA_ARCH}) ==" docker build -f .devops/cuda.Dockerfile --target server \ --build-arg CUDA_VERSION="${CUDA_VERSION}" \ --build-arg CUDA_DOCKER_ARCH="${CUDA_ARCH}" \ --build-arg APP_VERSION="${MASTER_SHA}-pr${PR}" \ --build-arg APP_REVISION="$(git rev-parse HEAD)" \ -t "${IMAGE}:${MASTER_SHA}-pr${PR}" \ -t "${IMAGE}:custom-latest" \ . echo "== built ${IMAGE}:${MASTER_SHA}-pr${PR} (entrypoint /app/llama-server, drop-in for the seat) =="