Compare commits

1 Commits
Author SHA1 Message Date
julianmb 765c2391d0 receipts: merged engine 256k validation 2026-09-01 16:30:52 +10:00
4 changed files with 75 additions and 61 deletions
+22 -51
View File
@@ -1,48 +1,21 @@
# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv) # haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
# builds the nathanw1014 strix-halo-vulkan engine and serves with the # builds the nathanw1014 strix-halo-vulkan engine and serves with the
# recommended flags. models are mounted, not baked in. # recommended flags. models are mounted, not baked in.
#
# notes on the build:
# - the old `ppa:kisak/kisak` repo does not exist (the PPA is `kisak-mesa`),
# which failed `add-apt-repository`. noble-updates already ships
# mesa 25.2.x (radv supports gfx1151), so no PPA is needed.
# - noble's glslc (shaderc 2023.8) predates GL_NV_cooperative_matrix2, so
# the fork's coopmat2 kernels silently degrade. we pin a recent
# conda-forge shaderc stack (self-contained, runs on noble libs).
# - LLAMA_CURL is a deprecated no-op option in this fork; dropped.
# - the fork builds many shared libs (libmtmd, *-impl libs, backends);
# copy all of bin/*.so*, not just libggml*/libllama*.
# ---- stage 1: build ---- # ---- stage 1: build ----
FROM ubuntu:24.04 AS build FROM ubuntu:24.04 AS build
ENV DEBIAN_FRONTEND=noninteractive ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y \
build-essential cmake ninja-build git ccache pkg-config \ build-essential cmake ninja-build git ccache \
ca-certificates curl unzip zstd \ libvulkan-dev glslc vulkan-tools \
libvulkan-dev spirv-headers libssl-dev \ libcurl4-openssl-dev \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
# modern glslc (shaderc v2026.3) + its libs, pinned conda-forge builds.
# glslc uses rpath $ORIGIN/../lib, so /opt/glslc/{bin,lib} is self-contained.
RUN set -eux; \
mkdir -p /opt/glslc /tmp/gc; cd /tmp/gc; \
for u in \
https://conda.anaconda.org/conda-forge/linux-64/shaderc-2026.3-hcebf71c_1.conda \
https://conda.anaconda.org/conda-forge/linux-64/glslang-16.5.0-h980caa0_2.conda \
https://conda.anaconda.org/conda-forge/linux-64/spirv-tools-2026.3-h7148c6a_1.conda; do \
curl -fL --retry 3 -o pkg.conda "$u"; \
unzip -o -q pkg.conda "pkg-*.tar.zst"; \
tar --zstd -xf pkg-*.tar.zst -C /opt/glslc; \
rm -f pkg.conda pkg-*.tar.zst; \
done; \
rm -rf /tmp/gc; \
/opt/glslc/bin/glslc --version | head -1
RUN git clone --depth 1 -b strix-halo-vulkan \ RUN git clone --depth 1 -b strix-halo-vulkan \
https://github.com/Nathanw1014/llama.cpp /src/engine https://github.com/Nathanw1014/llama.cpp /src/engine
RUN cmake -B /src/engine/build -S /src/engine -G Ninja \ RUN cmake -B /src/engine/build -S /src/engine \
-DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \ -DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
-DVulkan_GLSLC_EXECUTABLE=/opt/glslc/bin/glslc \ -DLLAMA_CURL=ON \
&& cmake --build /src/engine/build --parallel $(nproc) \ && cmake --build /src/engine/build --parallel $(nproc) \
--target llama-server llama-cli llama-bench --target llama-server llama-cli llama-bench
@@ -50,24 +23,22 @@ RUN cmake -B /src/engine/build -S /src/engine -G Ninja \
FROM ubuntu:24.04 FROM ubuntu:24.04
ENV DEBIAN_FRONTEND=noninteractive ENV DEBIAN_FRONTEND=noninteractive
# stock noble mesa (25.2.x in noble-updates) — radv gfx1151 supported, # add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x)
# no PPA needed RUN apt-get update && apt-get install -y software-properties-common gpg-agent \
RUN apt-get update && apt-get install -y --no-install-recommends \ && add-apt-repository -y ppa:kisak/kisak \
&& apt-get update && apt-get install -y \
mesa-vulkan-drivers vulkan-tools libvulkan1 \ mesa-vulkan-drivers vulkan-tools libvulkan1 \
libssl3t64 libgomp1 libstdc++6 ca-certificates \ libcurl4 \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
COPY --from=build /src/engine/build/bin/llama-server /app/llama-server COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
COPY --from=build /src/engine/build/bin/ /tmp/engine-bin/ COPY --from=build /src/engine/build/bin/libggml*.so* /app/
RUN set -eux; \ COPY --from=build /src/engine/build/bin/libllama*.so* /app/
for f in /tmp/engine-bin/*.so*; do \
[ -e "$f" ] || continue; \ RUN ldconfig /app 2>/dev/null; true
if [ -L "$f" ]; then cp -a "$f" /app/; else cp -aL "$f" /app/; fi; \ ENV LD_LIBRARY_PATH=/app
done; \
rm -rf /tmp/engine-bin; \
echo /app > /etc/ld.so.conf.d/app.conf; ldconfig
# models volume # models volume
VOLUME /models VOLUME /models
@@ -75,9 +46,9 @@ VOLUME /models
WORKDIR /app WORKDIR /app
EXPOSE 8080 EXPOSE 8080
# bare binary — pass all args (model, ctx, spec decode, ...) from the CMD ["/app/llama-server", \
# container runtime / llamaswap. bind defaults kept env-configurable: "-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \
# override with -e LLAMA_ARG_HOST=... / -e LLAMA_ARG_PORT=... "-ngl", "999", "-fa", "on", \
ENV LLAMA_ARG_HOST=0.0.0.0 \ "-ctk", "q8_0", "-ctv", "q8_0", \
LLAMA_ARG_PORT=8080 "-c", "32768", "-ub", "2048", "-t", "4", \
CMD ["/app/llama-server"] "--jinja", "--host", "0.0.0.0", "--port", "8080"]
+4 -10
View File
@@ -16,14 +16,8 @@ services:
- "8080:8080" - "8080:8080"
environment: environment:
- LD_LIBRARY_PATH=/app - LD_LIBRARY_PATH=/app
# the image CMD is bare (/app/llama-server) — pass the model + flags: # override CMD to change context, quant, or add mtp:
command: >- # docker compose run qwen38-flash-next \
-m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf # /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \
-ngl 999 -fa on -ctk q8_0 -ctv q8_0 # -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
-c 32768 -ub 2048 -t 4 --jinja
# for mtp speculative decoding:
# command: >-
# -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf
# -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf
# --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75 # --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
# -ngl 999 -fa on -ctk q8_0 -ctv q8_0 -c 32768 -ub 2048 -t 4 --jinja
Binary file not shown.
+49
View File
@@ -0,0 +1,49 @@
# Merged engine 256k validation receipt — 2026-08-31
Run 2026-09-01 AEST on 128 GiB Strix Halo. Engine: `build-hq38/bin/llama-cli`, build `b10888-081edc343` (binary size `1.4M`); target: 91 GiB `Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf`; MTP: 3.9 GiB Q8_0 sidecar. Every launch used `timeout -k 30 2400`, a preflight `free -h` plus exact-name process check, `oom_score_adj=500`, and 30-second `VmRSS`/`VmSwap` sampling. The build exposes the requested lazy option as `--lazy-mode auto` rather than `--tensor-read-lazy auto`.
## 256k MTP generation
| context / prompt | result | generation grep | lazy grep | peak process RSS / swap |
|---|---|---|---|---:|
| 257,024 / 255,718 tokens | **ABORTED, no timing** | no `Generation: ... t/s` match | no match in non-verbose run log | 847,512 / 110,788 KiB |
The run held `VmSwap: 0 kB` for 32 samples through 15:15:22, then sampled 29,152 and 110,788 KiB at 15:15:52/15:16:22. The mandatory two-sample guard killed PID 89526 and recorded `requires lazy PLE (swap >8GiB)` / exit 137. Therefore the missing 256k MTP number is **not validated**. Lazy PLE itself is proven active by the verbose oracle grep: `tensor per_layer_token_embd.weight (size = 27465 MiB) lazy read enabled` in both oracle logs.
## Greedy identity oracle (`--reasoning off`, `-n 2048`)
| path | generation | graphs reused | identity |
|---|---:|---:|---|
| plain | 26.9 t/s | 290 | **FAIL** |
| MTP n=6 | 40.2 t/s | 20 | **FAIL — 28 unified-diff lines** |
The archived `results/hq38-oracle.diff` shows a real greedy divergence: MTP omits the docstring's `Raises` section and moves the negative-input check below memo initialization. The archive also contains the extracted `hq38-oracle-{plain,mtp-n6}.txt` outputs; this does not satisfy the expected zero diff.
## PLE depth sweep
Cells are prompt / generation tokens per second, from the exact grep lines in `results/hq38-depth<depth>-<mode>.log`.
| depth | bounded context | plain pp / tg | MTP n=6 pp / tg |
|---:|---:|---:|---:|
| 0 | 8,192 | 94.6 / 29.0 | 70.2 / 45.7 |
| 8k | 16,384 | 496.9 / 21.7 | 464.8 / 27.6 |
| 32k | 40,960 | 400.5 / 17.7 | 382.5 / 23.5 |
| 128k | 139,264 | 266.8 / 9.0 | 257.2 / 12.9 |
| 256k | 257,024 | 202.9 / 6.2 | **aborted; no timing** |
## Memory receipt
| leg | peak VmRSS KiB | peak VmSwap KiB | exit |
|---|---:|---:|---:|
| oracle plain / MTP | 405,892 / 671,072 | 0 / 0 | 0 / 0 |
| 0 plain / MTP | 35,280 / 666,392 | 0 / 0 | 0 / 0 |
| 8k plain / MTP | 289,416 / 431,820 | 0 / 0 | 0 / 0 |
| 32k plain / MTP | 317,404 / 710,288 | 0 / 0 | 0 / 0 |
| 128k plain / MTP | 490,400 / 1,616,636 | 0 / 0 | 0 / 0 |
| 256k plain / MTP | 628,088 / 847,512 | 0 / 110,788 | 0 / 137 (guard) |
Raw evidence is committed as `2026-08-31-merged-engine-256k-logs.tar.gz`
(`sha256:32605f52753545fdc258f7ffe79ff10e0764286e445ab897341464b2e6985951`):
all `hq38-*.log`/memory traces, both oracle texts and their diff, the guard log,
and matrix status. No two GPU jobs overlapped; every preflight recorded
`llama-cli=0 llama-server=0` before launch.