From a7c7aac004fffac3c5d1ad0b0d3a1914aca2ad1b Mon Sep 17 00:00:00 2001 From: Julian Beltran Date: Mon, 31 Aug 2026 23:50:55 +1000 Subject: [PATCH] =?UTF-8?q?haloq38flash=20=E2=80=94=20qwen3.8-flash-next?= =?UTF-8?q?=20on=20strix=20halo:=20converter=20fix,=2091g=20provenance-ver?= =?UTF-8?q?ified=20quant,=20engine=20a/b,=20depth=20tables=20through=20256?= =?UTF-8?q?k?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .gitignore | 6 + Dockerfile | 54 ++++++++ README.md | 165 +++++++++++++++++++++++ docker-compose.yml | 23 ++++ docs/engine-cherry-pick-plan.md | 123 +++++++++++++++++ scripts/convert-flash-next-rocmfpx.sh | 56 ++++++++ scripts/depth-bench-strix-halo-vulkan.sh | 47 +++++++ scripts/gguf-header-peek.py | 133 ++++++++++++++++++ scripts/mtp-test-strix-halo-vulkan.sh | 65 +++++++++ 9 files changed, 672 insertions(+) create mode 100644 .gitignore create mode 100644 Dockerfile create mode 100644 README.md create mode 100644 docker-compose.yml create mode 100644 docs/engine-cherry-pick-plan.md create mode 100755 scripts/convert-flash-next-rocmfpx.sh create mode 100755 scripts/depth-bench-strix-halo-vulkan.sh create mode 100755 scripts/gguf-header-peek.py create mode 100755 scripts/mtp-test-strix-halo-vulkan.sh diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..4312eda --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +models/ +weights +.codegraph +.omo/ +wiki-*.parquet +corpus/ diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..914e2dd --- /dev/null +++ b/Dockerfile @@ -0,0 +1,54 @@ +# haloq38flash β€” qwen3.8-flash-next on strix halo (vulkan/radv) +# builds the nathanw1014 strix-halo-vulkan engine and serves with the +# recommended flags. models are mounted, not baked in. + +# ---- stage 1: build ---- +FROM ubuntu:24.04 AS build +ENV DEBIAN_FRONTEND=noninteractive +RUN apt-get update && apt-get install -y \ + build-essential cmake ninja-build git ccache \ + libvulkan-dev glslc vulkan-tools \ + libcurl4-openssl-dev \ + && rm -rf /var/lib/apt/lists/* + +RUN git clone --depth 1 -b strix-halo-vulkan \ + https://github.com/Nathanw1014/llama.cpp /src/engine +RUN cmake -B /src/engine/build -S /src/engine \ + -DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \ + -DLLAMA_CURL=ON \ + && cmake --build /src/engine/build --parallel $(nproc) \ + --target llama-server llama-cli llama-bench + +# ---- stage 2: runtime ---- +FROM ubuntu:24.04 +ENV DEBIAN_FRONTEND=noninteractive + +# add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x) +RUN apt-get update && apt-get install -y software-properties-common gpg-agent \ + && add-apt-repository -y ppa:kisak/kisak \ + && apt-get update && apt-get install -y \ + mesa-vulkan-drivers vulkan-tools libvulkan1 \ + libcurl4 \ + && rm -rf /var/lib/apt/lists/* + +COPY --from=build /src/engine/build/bin/llama-server /app/llama-server +COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli +COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench +COPY --from=build /src/engine/build/bin/libggml*.so* /app/ +COPY --from=build /src/engine/build/bin/libllama*.so* /app/ + +RUN ldconfig /app 2>/dev/null; true +ENV LD_LIBRARY_PATH=/app + +# models volume +VOLUME /models + +WORKDIR /app +EXPOSE 8080 + +CMD ["/app/llama-server", \ + "-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \ + "-ngl", "999", "-fa", "on", \ + "-ctk", "q8_0", "-ctv", "q8_0", \ + "-c", "32768", "-ub", "2048", "-t", "4", \ + "--jinja", "--host", "0.0.0.0", "--port", "8080"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..c661f65 --- /dev/null +++ b/README.md @@ -0,0 +1,165 @@ +
+ +# haloq38flash + +**qwen3.8-flash-next on amd strix halo β€” 91g quant, 56 tok/s, 262k context** + +[![HF Model](https://img.shields.io/badge/πŸ€—_Model-IQ4_XS_PLE-ffD21E)](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF) +[![Engine](https://img.shields.io/badge/Engine-nathanw1014__vulkan-blue)](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) +[![License](https://img.shields.io/badge/License-Qwen_Community-orange)](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE) + +[![Speed](https://img.shields.io/badge/MTP_@8k-56.4_t%2Fs-brightgreen)](#results) +[![Depth](https://img.shields.io/badge/Verified-0_β†’_256k-blueviolet)](#results) +[![Provenance](https://img.shields.io/badge/Provenance-byte--verified-success)](#the-converter-bug) + +*every published quant byte-traced back to the official checkpoint* + +
+ +--- + +## results + +engine: [nathanw1014/llama.cpp `strix-halo-vulkan`](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) Β· vulkan/radv Β· q8_0 kv Β· `-ub 2048` Β· temp 0 + +| depth | plain pp/tg | mtp pp/tg | +|------:|:-----------:|:---------:| +| 0 | 92.5 / 29.9 | 87.0 / **53.1** | +| 8k | 480 / 24.1 | 458 / **56.4** | +| 32k | 397 / 20.1 | 379 / **30.2** | +| 128k | 222 / 11.0 | 214 / 18.6 | +| 256k | 139 / 6.2 | β€” | + +> [!NOTE] +> no collapse through 32k. the 128k+ falloff is context-mechanics +> (sparse-attention indexer), not quant size β€” see the reversal below. + +
+the 128k reversal β€” the PLE quant loses under MTP at depth + +at ≀32k the PLE quant wins everywhere. at 128k under mtp it *loses* to the +static 116g (18.6 vs 26.9 t/s). plausible mechanism: iq4_nl noise in the +n-gram table compounds over deep history and lowers draft acceptance. +single runs, n=1 caveat. pick your file by use case β€” see the table above. + +
+ +--- + +## πŸ“¦ published quants + +[huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF) + +| file | size | pick it when | +|------|------|:------------:| +| `...-IQ4_XS-`**`PLE`**`.gguf` | 91 giB | ctx ≀ 32k β€” wins everywhere, mtp to 56 t/s | +| `...-IQ4_XS.gguf` | 116 giB | ctx β‰₯ 128k β€” faster mtp at depth, wider fork compat | +| `mtp-...-Q8_0.gguf` | 3.9 giB | mtp sidecar for nathanw1014-lineage engines | + +
+the PLE cut β€” why the 51b n-gram table tolerates 4-bit + +the PLE table is gathered 16 random rows per token via hash lookup β€” there is +no matmul on the table itself, and no two consecutive tokens hit the same rows. +the rows tolerate iq4_nl (4.25 bpw) with no measurable degradation across the +depth sweep. the cut: `--tensor-type "per_layer_token_embd=IQ4_XS"` on our +quantizer β†’ 54g β†’ 27g. + +**fork caveat:** engines that feed gathered PLE rows straight into mul_mat as +quantized B operands assert (apepojken-class, ggml-vulkan.cpp:7794). verified +working on nathanw1014 strix-halo-vulkan and rocmfpx. + +
+ +### pick your setup + +| your use case | quant | engine | ctx | expect | +|---|---|---|---|---| +| coding agents, chat | 91g PLE | [nathanw1014 vulkan](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) + mtp | ≀ 32k | 56 t/s | +| long documents | 116g static | same engine + mtp | 128k | 27 t/s | +| full rag / research | 116g static | [rocm 10 container](https://github.com/MorezMartin/engramhalo-rocm10) + ssd streaming | 262k | 14 t/s | + +--- + +## πŸ› the converter bug + +our first quant printed deterministic garbage at temp 0. bisect to root cause: + +- experts, gdn reorder, ple scale, metadata: all innocent +- **97 of 388 f32 tensors differed by exactly 1.0** β€” every hyper-connection + norm shipped raw where the runtime expects `raw + 1` +- cause: the checkpoint nests hyper-connections under + `attn_hyper_connection` / `mlp_hyper_connection` / `hyper_connection_mixer`, + and those names hit early-return branches in the converter that bypass the + generic `norm.weight β†’ +1` rule + +> [!WARNING] +> **any fork rolling its own qwen4exp converter must fold `(1 + w)` into the +> hyper-connection gammas.** upstream runtime documents the contract at +> `qwen4exp.cpp:231`. miss it and every layer normalizes wrong β€” garbage +> from layer 0, all shapes correct, all shape-only tests pass. + +fix + regression test: rocmfpx `port-qwen4exp` commit `61b6a3b48` +([pr charlie12345/ROCmFPX#98](https://github.com/charlie12345/ROCmFPX/pull/98)) + +--- + +## 🐳 docker + +```bash +git clone https://github.com/julianmb/haloq38flash && cd haloq38flash +docker compose up --build +# serve on :8080 β€” vulkan/radv, no rocm install needed +``` + +add the mtp sidecar for 56 t/s: + +```bash +docker compose run qwen38-flash-next /app/llama-server \ + -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \ + --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75 +``` + +--- + +## πŸ”¬ engine merge (paused) + +we merged nathanw1014's branch (175 commits: vulkan perf stack, qwen4exp +runtime past the squash, lazy ple, spec-decode fixes) into ggml-org master β€” +one engine with lazy ple (262k on 66g resident) + depth fixes + master's +general improvements. paused at 53% build: the branches diverged semantically +in shared enum/base-class files. + +- [`docs/engine-cherry-pick-plan.md`](docs/engine-cherry-pick-plan.md) β€” full + 175-commit classification +- [`docs/engine-merge-status.md`](docs/engine-merge-status.md) β€” resume point + +--- + +## πŸ“ layout + +| path | what | +|------|------| +| `models/` | symlink farm to `/mnt/ssd2/models/` (never in git) | +| `results/` | dated receipts, one per investigation | +| `scripts/` | conversion pipeline, oracle, depth bench, resume helpers | +| `reddit/` | archived community threads that drove the investigation | +| `docs/` | engine merge plan, cherry-pick classification | + +--- + +## ⚠️ operational gotchas (128g strix halo) + +- always `-c 8192`-bounded ctx + `timeout` + `/usr/bin/time -v` β€” the gguf + default 262144 + full offload hard-hung this box once +- `vm.dirty_ratio=15 / dirty_background_ratio=5` β€” the 191g ple conversion + memmap wedges `balance_dirty_pages` for hours at kernel defaults +- conversion peak: ple scratch (191g) + f16 output (354g) coexist β€” budget + ~560g free +- `pkill -x llama-cli`, never `-f` (matches your own wrapper shell) +- gpu memory is shared with everything else on the apu β€” two engines cannot + hold ~90g+ models simultaneously without an oom cascade + +--- + +license: [qwen community license 1.0](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE) Β· base model: [Qwen/Qwen3.8-Flash-Next](https://huggingface.co/Qwen/Qwen3.8-Flash-Next) diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..e426906 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,23 @@ +services: + qwen38-flash-next: + build: . + image: haloq38flash:latest + container_name: qwen38-flash-next + devices: + - /dev/dri + group_add: + - video + - render + security_opt: + - seccomp=unconfined + volumes: + - /mnt/ssd2/models/qwen38-flash-next:/models:ro + ports: + - "8080:8080" + environment: + - LD_LIBRARY_PATH=/app + # override CMD to change context, quant, or add mtp: + # docker compose run qwen38-flash-next \ + # /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \ + # -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \ + # --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75 diff --git a/docs/engine-cherry-pick-plan.md b/docs/engine-cherry-pick-plan.md new file mode 100644 index 0000000..1505809 --- /dev/null +++ b/docs/engine-cherry-pick-plan.md @@ -0,0 +1,123 @@ +# engine cherry-pick plan β€” nathan/strix-halo-vulkan β†’ ggml-org master + +date: 2026-08-31. base for counting: merge-base `9f0d017ef` (#27235 era). +nathan branch tip: `ad914eb65`. #27742 landed in master as squash `6c84c7d5d`. +nathan's branch = the #27742 development history + his strix-halo patch stack + +three master merges he already did + the qwen4exp runtime continued past the +squash point. + +## counts + +- 175 commits on `6c84c7d5d..nathan/strix-halo-vulkan` (reverse order in + `docs/nathan-175-commits.txt`) +- ~40 of them are the #27742 development history β€” **skip**, master's squash + `6c84c7d5d` already carries that content +- ~8 are master commits that reached his branch via his three master merges + (muse glimmer #26841/#26879, motif-3, dspark #27508, kv-cell #27762) β€” + **skip**, master has them +- 4 are CI/toolbox release plumbing β€” **skip** (fork-specific) +- ~10 are merge commits β€” **skip** (resolved by the one big merge below) +- **~115 genuine candidates**, grouped below + +## recommendation: one merge, not 115 cherry-picks + +the qwen4exp runtime commits and the vulkan shader stack interleave (the +sparse-FA shaders are prerequisites for the qwen4exp QSA gather path; the FACP +refactor renames classes the later commits use). piecemeal cherry-picking +breaks the build between commits. instead: + +``` +git checkout -b haloq38flash-engine ggml-org/master # or origin/master +git merge nathan/strix-halo-vulkan +# resolve conflicts once: ggml-vulkan mostly takes THEIRS (the perf stack), +# src/llama*.cpp mixed, everything else master +cmake -B build -DGGML_VULKAN=ON && cmake --build build -j 24 +``` + +nathan already merged master into his branch three times +(`aaf4fba83`, `b7b85da9c`, `f94fad0e8`/`add19980d`) β€” the reverse merge is the +same operation he proved works, and conflicts concentrate in the files he owns. + +## group A β€” vulkan fa/mmq perf stack (~45, oldest first) + +the coopmat1 FA rework, dequant-once scratch, contiguized KV, mul_mat_id tile +probes, f16-B path, q5_K/q4_K scale caches, wave32, LDS pad tuning, the six +env-gated perf flags now default-on. cherry-pick as a block, oldest first; +`acd14737e FACP` and `892924042 single source of truth` are the load-bearing +refactors the later ones sit on. skip `681675530` (marked NEGATIVE result). + +## group B β€” dsv4 lightning indexer + sparse fa gather (~25) + +`890550c0a` indexer kernels + indexed sparse FA, `5dfc01ff6` gather-to-compact +decode, the sparse prefill split/tile/cache cluster, quantised K/V inside the +gathers (`8b66f91c6`, `7b63cbd6b`, `6b2cade31`), small-batch union +(`8115df4c7`..`31202f9df`). written for deepseek v4, powers qwen4exp's QSA the +same way. NOTE: `b65c360c7` fixes multi-sequence β€” keep. + +## group c β€” fused hyper-connection ops + command buffers (~5) + +`2041049a4` fused HC pre/comb/post (the 3550β†’2800 dispatch win), +`e709b949e` command buffers bounded by memory traffic, +`18239a695` perf-logger flush, `0f80b884d`/`8a8fee776` UMA copy path. + +## group d β€” hip/"ggml-cuda" rdna3.5 tuning (~12) + +`64e5c14f1` kernel tuning, `f074165ae` quantized-KV FA, MMQ tile tuning +(`71ac6c1d9`, `f70839f9a`, `86e3f34fc`), Q8_1 activation cache (`a649f1634`), +WMMA indexer (`8209c8954`), tiled FA (`e88b92eff`), GDN tune (`910f0f25d`), +NaN fix (`4ea44eef2`), tests (`b1282d2af`, `6e7b355cb`). named ggml-cuda +because the hip backend rides the cuda code paths. + +## group e β€” qwen4exp runtime past the squash point (~25) + +what master's squash does NOT have: +- `be71d63c9` quantized KV cache in the QSA attention path (the q8_0 kv fix + our rocmfpx build lacks) +- `631b9ffb1` decode-graph reuse + host-side PLE gather (graphs reused 68 vs 0) +- `354390810` + `39817c476` NextN/MTP draft: sidecar AND in-file loading +- `f32aca1c1`/`79c2d2cad`/`3849d54b8`/`d763facad`/`fdf96fcea` indexer cache in + llama_memory_hybrid_idx, slots, names +- `87f31259a`/`c04b3ff4b`/`8f58c2f0a` PLE history per context + iterator fix +- `05f6575ab`/`25a796300`/`cdd2e47ae` indexer cache save/restore + slots +- `c1d5b2d0e`/`bd92a90c4`/`671203688` random-access mmap advice for the + gather table (the ple-ssd-streaming primitive) +- `024b7ad93` QSA bias per block, `7073ae357` hparams shrink, + `1486f6b88` non-unified KV in QSA, `a80d678ad` image placeholder hash, + `562cb00bc`/`42d976771` tensor-split segments, `d6f65ff28` graph budget +- quantizer: `7a4d5960d` PLE streaming (independently written β€” same fix as + our banding), `9e2d2eb84` --tensor-type names the PLE (the flag the 91g + quant used), `5beb9965b` f16 fallback for odd ncols, + `5096585d6` exact output buffer +- tests: `171ddb8df`/`086457e7b`/`77953f1e1` + +## group f β€” speculative decoding fixes (~10) + +the 7-bug stack behind "spec decode works end to end": +`53fd8b48c` GDN state graph order, `9c5d899ff`/`f25eefeaf`/`a17e8432b` +MTP rollback full checkpoints (applyβ†’revertβ†’reapply), `08a325524` +checkpoints on device, `0eb528051` draft trimming for mtmd, +`64e2b680a` dflash cache alignment, `397ef7c72` no_vocab special tokens. + +## group g β€” optional, other archs (~10, default skip) + +dspark bailingmoe3 (`2586f6edd`), dflash2 (`015f09c8a`/`0b0f35d0e`), +motif-3 (`4c7f96093`/`be54e2891`/`a359e55c9`) β€” only if wanted; they ride +along in the merge anyway. + +## verification after the merge + +1. build vulkan, zero errors +2. our depth bench on the 91g quant: 0/8k/32k/128k β€” expect >= the nathanw1014 + numbers (29.9/24.1/20.1 plain, 53.1/56.4/18.6 mtp) plus master's 771 commits +3. greedy oracle: with `39817c476` spec decode + the rollback fixes, the + 6-line divergence on our current engine should close (his fork is the one + the 7-bug fix stack was written for) +4. lazy ple: `--tensor-read-lazy auto` must log "lazy read enabled" for + per_layer_token_embd β€” the load_mode=none hardcode does not exist on + master's path +5. then 256k mtp: no thrash expected (66g resident with lazy ple) + +## candidates for upstreaming after validation + +radix/sparse top-k fa, fused hc epilogs, gdn concat fix, the q8_0-kv-in-qsa +fix, the lazy ple plumbing, the converter trap note (hc norms). diff --git a/scripts/convert-flash-next-rocmfpx.sh b/scripts/convert-flash-next-rocmfpx.sh new file mode 100755 index 0000000..fb21bd3 --- /dev/null +++ b/scripts/convert-flash-next-rocmfpx.sh @@ -0,0 +1,56 @@ +#!/usr/bin/env bash +# convert-flash-next-rocmfpx.sh β€” Qwen3.8-Flash-Next FP8 safetensors -> ROCmFP4_FAST GGUF +# +# Lives in haloq38flash; the engine (converter + llama-quantize) is ~/source/ROCmFPX +# (branch port-qwen4exp, PR #98). +# +# Pipeline: +# 1. convert_hf_to_gguf.py -> F16 GGUF (PLE fp8 scale captured + applied) +# 2. llama-quantize -> Q4_0_ROCMFP4_FAST (per_layer_token_embd protected to Q8_0; +# banded/streaming quantizer keeps RAM bounded) +# 3. smoke: llama-completion (NOT llama-cli) with bounded -c and a timer +# +set -eo pipefail + +SRC_DIR="${SRC_DIR:-/mnt/ssd2/models/qwen38-flash-next/meta-fp8}" +WORK_DIR="${WORK_DIR:-/mnt/ssd2/models/qwen38-flash-next}" +ENGINE="${ENGINE:-/home/user/source/ROCmFPX}" +OUT_F16="${WORK_DIR}/Qwen3.8-Flash-Next-F16.gguf" +OUT_QUANT="${WORK_DIR}/Qwen3.8-Flash-Next-ROCmFP4_FAST.gguf" +SMOKE_CTX="${SMOKE_CTX:-8192}" + +mkdir -p "${WORK_DIR}" + +if [ ! -f "${SRC_DIR}/config.json" ]; then + echo "ERROR: ${SRC_DIR}/config.json missing - download not complete?" >&2 + exit 1 +fi + +echo "=== [1/3] convert FP8 safetensors -> F16 GGUF ===" +if [ ! -f "${OUT_F16}" ]; then + cd "${ENGINE}" + setsid nohup python3 convert_hf_to_gguf.py "${SRC_DIR}" \ + --outfile "${OUT_F16}" \ + --outtype f16 +else + echo "F16 GGUF exists, skipping: ${OUT_F16}" +fi + +echo "=== [2/3] quantize -> Q4_0_ROCMFP4_FAST ===" +if [ ! -f "${OUT_QUANT}" ]; then + /usr/bin/time -v "${ENGINE}/build-strix-rocmfp4/bin/llama-quantize" \ + "${OUT_F16}" \ + "${OUT_QUANT}" \ + Q4_0_ROCMFP4_FAST +else + echo "quant exists, skipping: ${OUT_QUANT}" +fi + +echo "=== [3/3] smoke: llama-completion, bounded context, timer ===" +/usr/bin/time -v timeout 900 "${ENGINE}/build-strix-rocmfp4/bin/llama-completion" \ + -m "${OUT_QUANT}" \ + -dev Vulkan0 -ngl 47 -c "${SMOKE_CTX}" -fa on -ub 2048 \ + -p "The capital of France is" -n 64 --temp 0 -no-cnv --simple-io 2>&1 | tail -40 + +ls -lah "${OUT_F16}" "${OUT_QUANT}" +echo "DONE" diff --git a/scripts/depth-bench-strix-halo-vulkan.sh b/scripts/depth-bench-strix-halo-vulkan.sh new file mode 100755 index 0000000..cc090ec --- /dev/null +++ b/scripts/depth-bench-strix-halo-vulkan.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +# Decode + prefill vs context depth, with and without MTP. +# +# Uses filler prompts of ~8k and ~32k tokens (results/filler/) and reports the +# [ Prompt: X t/s | Generation: Y t/s ] line llama-cli prints. --reasoning off +# keeps the 128-token generation budget from being eaten by a thinking block. +# q8_0 KV keeps the cache small at depth. +# +# usage: scripts/depth-bench-strix-halo-vulkan.sh [depths] # e.g. "8k 32k" +set -u + +BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin +TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf} +DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf} +OUT=${OUT:-/home/user/source/haloq38flash/results} +FILLER=$OUT/filler +SHORT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example." +DEPTHS=${1:-0 8k 32k} + +declare -A PROMPT_CTX=( [0]=8192 [8k]=16384 [32k]=40960 [128k]=139264 [256k]=257024 ) + +run() { + local depth=$1 mode=$2 + local tag="depth$depth-$mode" + local log=$OUT/${TAGPREFIX:-}shvd-$tag.log + local args=() + if [ "$depth" = "0" ]; then + args+=(-p "$SHORT") + else + args+=(-f "$FILLER/filler-$depth.txt") + fi + [ "$mode" = "mtp" ] && args+=(-md "$DRAFT" --spec-type draft-mtp \ + --spec-draft-n-max 6 --spec-draft-p-min 0.75) + + timeout 2400 "$BIN/llama-cli" -m "$TARGET" "${args[@]}" \ + -dev Vulkan0 -ngl 999 -c "${PROMPT_CTX[$depth]}" -fa on -ub 2048 \ + -ctk q8_0 -ctv q8_0 \ + -n 128 --temp 0 --reasoning off -no-cnv -st --simple-io > "$log" 2>&1 + printf '%-18s exit=%-3s %s %s\n' "$tag" "$?" \ + "$(grep -oE 'Prompt: [0-9.]+ t/s' "$log" | tail -1)" \ + "$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)" +} + +for d in $DEPTHS; do + run "$d" plain + run "$d" mtp +done diff --git a/scripts/gguf-header-peek.py b/scripts/gguf-header-peek.py new file mode 100755 index 0000000..ca14ae1 --- /dev/null +++ b/scripts/gguf-header-peek.py @@ -0,0 +1,133 @@ +#!/usr/bin/env python3 +"""Inspect a remote GGUF's metadata + tensor names without downloading the weights. + +The interesting part of a GGUF (architecture, block_count, nextn_predict_layers, +tensor names) all lives in the header, which is a few MB even for a 100 GB file. +This range-fetches the first chunk and parses the header directly. + +usage: + scripts/gguf-header-peek.py [filename-substring] + scripts/gguf-header-peek.py /path/to/local.gguf + +examples: + scripts/gguf-header-peek.py EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF + scripts/gguf-header-peek.py unsloth/Qwen3.8-Flash-Next-GGUF Q3_K_XL +""" +import struct +import subprocess +import sys +import tempfile + +CHUNK = 96 * 1024 * 1024 # header is dominated by the tokenizer strings + +SCALAR_BYTES = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8} +SCALAR_FMT = {0: " 32: + for _ in range(count): + self.value(sub) + return f"<{count} x {TYPE_NAMES.get(sub, sub)}>" + return [self.value(sub) for _ in range(count)] + if vtype in SCALAR_BYTES: + raw = self.fh.read(SCALAR_BYTES[vtype]) + if len(raw) < SCALAR_BYTES[vtype]: + raise ValueError("truncated header: increase CHUNK") + if vtype == 7: + return bool(raw[0]) + return struct.unpack(SCALAR_FMT[vtype], raw)[0] + raise ValueError(f"unknown gguf value type {vtype}") + + +def parse(path): + with open(path, "rb") as fh: + r = Reader(fh) + assert fh.read(4) == b"GGUF", "not a GGUF file" + version, n_tensors, n_kv = r.u32(), r.u64(), r.u64() + kv = {r.string(): r.value(r.u32()) for _ in range(n_kv)} + names = [] + for _ in range(n_tensors): + names.append(r.string()) + r.fh.read(8 * r.u32()) # dims + r.fh.read(12) # type + offset + return version, kv, names + + +def fetch(repo, want): + listing = subprocess.run( + ["curl", "-s", f"https://huggingface.co/api/models/{repo}"], + capture_output=True, text=True, check=True).stdout + import json + files = [s["rfilename"] for s in json.loads(listing)["siblings"] + if s["rfilename"].endswith(".gguf")] + matches = [f for f in files if want in f] if want else files + if not matches: + raise SystemExit(f"no .gguf matching {want!r} in {repo}") + name = sorted(matches)[0] + print(f"repo: {repo}\nfile: {name} (of {len(files)} gguf files)") + url = f"https://huggingface.co/{repo}/resolve/main/{name}" + tmp = tempfile.NamedTemporaryFile(suffix=".gguf", delete=False) + subprocess.run(["curl", "-sL", "-r", f"0-{CHUNK}", "-o", tmp.name, url], check=True) + return tmp.name + + +def main(): + if len(sys.argv) < 2: + raise SystemExit(__doc__) + target = sys.argv[1] + if target.startswith(("http", "/")) and target.endswith(".gguf"): + path = target if target.startswith("/") else fetch(target, "") + else: + path = fetch(target, sys.argv[2] if len(sys.argv) > 2 else "") + + version, kv, names = parse(path) + arch = kv.get("general.architecture", "?") + print(f"gguf v{version} arch={arch} tensors={len(names)} kv={len(kv)}") + print("\n-- key metadata --") + for k in sorted(kv): + if k.startswith("tokenizer.") or k.startswith("general."): + continue + v = kv[k] + if isinstance(v, list) and len(v) > 12: + v = f"{v[:12]} ... (len {len(v)})" + print(f" {k:<46} {v}") + + blocks = sorted({n.split(".")[1] for n in names + if n.startswith("blk.") and n.split(".")[1].isdigit()}) + print(f"\n-- block indices: {blocks[:12]}{' ...' if len(blocks) > 12 else ''} " + f"({len(blocks)} total)") + suffixes = sorted({n.split(".", 2)[2] for n in names + if n.startswith("blk.") and len(n.split(".")) > 2}) + print(f"-- per-block suffixes ({len(suffixes)}):") + for s in suffixes: + print(f" {s}") + other = sorted(n for n in names if not n.startswith("blk.")) + if other: + print("-- non-block tensors:") + for n in other: + print(f" {n}") + + +if __name__ == "__main__": + main() diff --git a/scripts/mtp-test-strix-halo-vulkan.sh b/scripts/mtp-test-strix-halo-vulkan.sh new file mode 100755 index 0000000..daecc22 --- /dev/null +++ b/scripts/mtp-test-strix-halo-vulkan.sh @@ -0,0 +1,65 @@ +#!/usr/bin/env bash +# A/B speculative decoding with MTP on the Nathanw1014 strix-halo-vulkan build. +# +# Runs the same greedy prompt with and without the MTP draft and diffs the text. +# At temp 0 the two must be identical (a "greedy identity oracle"): any +# divergence means the draft path is corrupting state. +# +# Uses the ORIGINAL sidecar (block_count = 49, blk.48), not the renumbered +# -blk0 one -- the runtime selects the trailing block itself. +# See results/2026-08-29-post-reboot-validation.md Β§6b. +# +# Tool notes: llama-cli, not llama-completion (the latter's parser rejects -md); +# -st is required or cli sits in an interactive loop printing "> " forever. +# +# usage: scripts/mtp-test-strix-halo-vulkan.sh [plain|mtp|both] +# NMAX=2,4,6 scripts/mtp-test-strix-halo-vulkan.sh mtp # sweep depths +set -u + +BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin +TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf} +DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf} +OUT=${OUT:-/home/user/source/haloq38flash/results} +# long-form prompt: the model emits EOS early on short factual ones, which makes +# the tok/s figure meaningless +PROMPT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example. Then explain how the memoization cache works." +CTX=8192 +NPRED=512 +NMAX_LIST=${NMAX:-6} + +run() { + local tag=$1; shift + local log=$OUT/${TAGPREFIX:-}shv-$tag.log + /usr/bin/time -v timeout 1800 "$BIN/llama-cli" \ + -m "$TARGET" "$@" \ + -dev Vulkan0 -ngl 999 -c "$CTX" -fa on -ub 2048 \ + -p "$PROMPT" -n "$NPRED" --temp 0 -no-cnv -st --simple-io > "$log" 2>&1 + local rc=$? + # generated text sits between the "> " echo and the timing line + awk '/^> /{f=1} /^\[ Prompt:/{f=0} f' "$log" > "$OUT/${TAGPREFIX:-}shv-$tag.txt" + printf '%-16s exit=%s %s (%s bytes of output)\n' "$tag" "$rc" \ + "$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)" \ + "$(wc -c < "$OUT/shv-$tag.txt")" +} + +case ${1:-both} in + plain) run plain ;; + mtp) + for n in ${NMAX_LIST//,/ }; do + run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \ + --spec-draft-n-max "$n" --spec-draft-p-min 0.75 + done ;; + both) + run plain + for n in ${NMAX_LIST//,/ }; do + run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \ + --spec-draft-n-max "$n" --spec-draft-p-min 0.75 + if diff -q "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" > /dev/null; then + echo " greedy identity n=$n: PASS" + else + echo " greedy identity n=$n: FAIL" + diff "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" | head -10 + fi + done ;; + *) echo "usage: $0 [plain|mtp|both]" ; exit 1 ;; +esac