From 09d33df1b5c0f97573047da6d616101010ee2b46 Mon Sep 17 00:00:00 2001 From: dvcdsys Date: Mon, 3 Aug 2026 10:32:48 +0100 Subject: [PATCH] chore(server): bump llama.cpp to b10238 / current image digests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Upstream moved well past every pin we carry. This bumps all three of them and fixes the bundle breakage the new layout exposed. Pins: - Dockerfile.cuda 7b3d7834 -> fd68d130 (:server-cuda) - Dockerfile 6bc9134e -> 9e60dd36 (:server, multi-arch) - Makefile LLAMA_VERSION b8914 -> b10238 (+ checksum row) fetch-llama.sh shipped a hand-maintained list of dylibs. b10238 moved each tool's logic into its own libllama--impl.dylib, so the list produced a bundle whose llama-server aborted at dyld load time: Library not loaded: @rpath/libllama-server-impl.dylib Ship the whole dylib set instead — the same lesson the Dockerfiles already encode with `COPY /app/*.so*`. The other tools' impl dylibs cost ~1 MB of a ~52 MB bundle. Standalone binaries are still dropped. A post-copy otool check now fails the fetch when any @rpath dependency of llama-server is missing, so the next layout change surfaces at fetch time instead of on an operator's machine. Also fix the weekly pin-freshness workflow's duplicate detection: the issue title contains "(" and ":", which GitHub's search parser reads as a qualifier, so `--search "$title in:title"` never matched and every run filed a fresh duplicate (#170 and #205 are the same reminder). Match the exact title against the open-issue list instead. Verified: - CPU image builds locally (arm64) and /app/llama-server runs; docker scout: 0 CRITICAL / 0 HIGH / 0 MEDIUM / 0 LOW, 73 MB - macOS arm64 bundle: llama-server --version -> b10238, all @rpath dependencies resolve - CUDA image: /app/llama-server + libllama-server-impl.so present in the new digest and the `COPY /app/*.so*` glob still captures them; a full CUDA build + scan still needs the amd64 builder (host was offline) and must run before the next server release Co-Authored-By: Claude Opus 5 --- .github/workflows/llama-pin-check.yml | 8 +++- server/Dockerfile | 2 +- server/Dockerfile.cuda | 2 +- server/Makefile | 2 +- server/scripts/fetch-llama.sh | 60 +++++++++++++-------------- server/scripts/llama-checksums.txt | 1 + 6 files changed, 40 insertions(+), 35 deletions(-) diff --git a/.github/workflows/llama-pin-check.yml b/.github/workflows/llama-pin-check.yml index 751b75d3..78343568 100644 --- a/.github/workflows/llama-pin-check.yml +++ b/.github/workflows/llama-pin-check.yml @@ -55,7 +55,13 @@ jobs: _Auto-filed by the weekly \`llama.cpp pin freshness\` workflow._ EOF )" - existing="$(gh issue list --state open --search "$title in:title" --json number --jq '.[0].number // empty')" + # Exact-title match against the open issues, NOT `--search`. The + # title contains "(" and ":" — GitHub's search parser reads + # "chore(server/cuda):" as a (bogus) qualifier and matches nothing, + # so every weekly run filed a fresh duplicate instead of commenting + # on the existing reminder (that is how #170 and #205 both existed). + existing="$(gh issue list --state open --limit 200 --json number,title \ + | jq -r --arg t "$title" 'map(select(.title == $t)) | .[0].number // empty')" if [ -n "$existing" ]; then echo "Updating existing issue #$existing" gh issue comment "$existing" --body "$body" diff --git a/server/Dockerfile b/server/Dockerfile index c79e2ece..3d3a26c8 100644 --- a/server/Dockerfile +++ b/server/Dockerfile @@ -127,7 +127,7 @@ RUN mkdir -p /out/data/models # target platform. Bump this digest deliberately and re-run `make scout-cpu`. # Resolve a new digest with: # docker buildx imagetools inspect ghcr.io/ggml-org/llama.cpp:server -FROM ghcr.io/ggml-org/llama.cpp:server@sha256:6bc9134e3278a0ecab23d7ef2f6a46b4595740014fe9bc2f67e8ba7dca8395b4 AS llama-source +FROM ghcr.io/ggml-org/llama.cpp:server@sha256:9e60dd363798776f451bcca97f22f5d75579480e53b0b46303d061a69ceea4df AS llama-source # ── Stage: distroless runtime ────────────────────────────────────────────── # gcr.io/distroless/cc-debian13:nonroot (Debian 13 trixie) — NOT static-debian12. diff --git a/server/Dockerfile.cuda b/server/Dockerfile.cuda index 07de417f..07eb4221 100644 --- a/server/Dockerfile.cuda +++ b/server/Dockerfile.cuda @@ -101,7 +101,7 @@ RUN curl -fsSL \ # missing libllama-server-impl.so). Bump this digest deliberately and re-run # `make scout-cuda` to validate. Resolve a new digest with: # docker buildx imagetools inspect ghcr.io/ggml-org/llama.cpp:server-cuda -FROM ghcr.io/ggml-org/llama.cpp:server-cuda@sha256:7b3d7834fc7307cb54f24f8869b67bfff276404c416452a48d11321bc36a81be AS llama-source +FROM ghcr.io/ggml-org/llama.cpp:server-cuda@sha256:fd68d13013141833e8214ecad6e1fbefb532db6a00b980cdecfe33603dbf2675 AS llama-source # ── Stage 3: extract CUDA shared libraries ───────────────────────────────── # Install the CUDA libs here, then COPY individual .so files into the diff --git a/server/Makefile b/server/Makefile index 4c4bda24..3edae8f8 100644 --- a/server/Makefile +++ b/server/Makefile @@ -5,7 +5,7 @@ # commit — the new SHA256 must be appended to scripts/llama-checksums.txt and # the parity gate re-run before shipping. -LLAMA_VERSION ?= b8914 +LLAMA_VERSION ?= b10238 LLAMA_REPO ?= ggml-org/llama.cpp # `make bundle OS=... ARCH=...` supports only darwin-arm64 in Phase 3. diff --git a/server/scripts/fetch-llama.sh b/server/scripts/fetch-llama.sh index daf45d56..eae1ea03 100755 --- a/server/scripts/fetch-llama.sh +++ b/server/scripts/fetch-llama.sh @@ -87,37 +87,19 @@ mkdir -p "$DEST_DIR" # Clean out any previous fetch — stale dylibs could get picked up by DYLD. rm -f "$DEST_DIR"/* 2>/dev/null || true -# Files we ship. llama-server is the only binary we need; dylibs are its -# runtime deps. We deliberately drop llama-cli, llama-bench, llama-quantize, -# rpc-server, llama-server's *-debug variants, mtmd-*, etc. to keep the -# bundle lean. -SHIP=( - "llama-server" - "libllama.dylib" - "libllama-common.dylib" - "libmtmd.dylib" - "libggml.dylib" - "libggml-base.dylib" - "libggml-cpu.dylib" - "libggml-metal.dylib" - "libggml-blas.dylib" - "libggml-rpc.dylib" -) -# Versioned dylib aliases — dyld resolves these via symlink/rpath. Include -# everything that matches the base names so @rpath lookups do not break. -for base in "${SHIP[@]}"; do - # Copy the bare file if present. - if [[ -e "$INNER_DIR/$base" ]]; then - cp -p "$INNER_DIR/$base" "$DEST_DIR/" - fi - # Copy any versioned variants (libfoo.0.dylib, libfoo.0.0.1234.dylib, ...) - # that begin with the same stem. Loose glob: for each dylib name stem we - # look for ".*.dylib". - stem="${base%.dylib}" - for match in "$INNER_DIR/$stem".*.dylib; do - [[ -e "$match" ]] || continue - cp -p "$match" "$DEST_DIR/" - done +# Files we ship: llama-server plus the WHOLE dylib set, not a hand-maintained +# list. Upstream periodically refactors the shared-library layout — b10238 +# moved each tool's logic into its own libllama--impl.dylib, so the old +# fixed list produced a bundle whose llama-server died at dyld load time with +# "Library not loaded: @rpath/libllama-server-impl.dylib". The same lesson is +# already encoded in Dockerfile/Dockerfile.cuda (`COPY /app/*.so*`). The other +# tools' impl dylibs cost ~1 MB out of a ~42 MB bundle — far cheaper than a +# broken bundle. Standalone binaries (llama-cli, llama-bench, llama-quantize, +# ggml-rpc-server, mtmd-*, …) are still dropped. +cp -p "$INNER_DIR/llama-server" "$DEST_DIR/" +for match in "$INNER_DIR"/*.dylib; do + [[ -e "$match" ]] || continue + cp -p "$match" "$DEST_DIR/" done # Sanity: llama-server must be present and executable. @@ -126,6 +108,22 @@ if [[ ! -x "$DEST_DIR/llama-server" ]]; then exit 1 fi +# Sanity: every @rpath dependency of llama-server must have landed in +# DEST_DIR. Without this the breakage only surfaces at runtime, on the +# operator's machine, as a dyld abort — exactly how the b10238 layout change +# was found. Fail the fetch instead. +missing=() +while read -r dep; do + [[ -n "$dep" ]] || continue + [[ -e "$DEST_DIR/$dep" ]] || missing+=("$dep") +done < <(otool -L "$DEST_DIR/llama-server" | awk '/@rpath\//{sub(/^.*@rpath\//, "", $1); print $1}') +if [[ ${#missing[@]} -gt 0 ]]; then + echo "fetch-llama: llama-server has unresolved @rpath dependencies:" >&2 + printf ' %s\n' "${missing[@]}" >&2 + echo "fetch-llama: upstream $LLAMA_VERSION likely changed its library layout — inspect the release archive." >&2 + exit 1 +fi + # macOS Gatekeeper quarantine can apply to downloaded binaries even via curl. # Strip the attribute so end users do not hit a silent kill on first run. if command -v xattr >/dev/null 2>&1; then diff --git a/server/scripts/llama-checksums.txt b/server/scripts/llama-checksums.txt index 15684e56..354d12bd 100644 --- a/server/scripts/llama-checksums.txt +++ b/server/scripts/llama-checksums.txt @@ -3,3 +3,4 @@ # Bumping LLAMA_VERSION in the Makefile MUST be accompanied by a commit that # records the new row here. CI fails hard on SHA mismatch. cd70e17321b822820f757a250c2f8128c69290f43048f6e7bc79ec8822a1a5c5 llama-b8914-bin-macos-arm64.tar.gz +1bc2c461b1c64f8f71c99deb55a21d5e43e2deeea1d398f8bb0a1969eb1306aa llama-b10238-bin-macos-arm64.tar.gz