haloq38flash — qwen3.8-flash-next on strix halo: converter fix, 91g provenance-verified quant, engine a/b, depth tables through 256k
This commit is contained in:
@@ -0,0 +1,6 @@
|
||||
models/
|
||||
weights
|
||||
.codegraph
|
||||
.omo/
|
||||
wiki-*.parquet
|
||||
corpus/
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
|
||||
# builds the nathanw1014 strix-halo-vulkan engine and serves with the
|
||||
# recommended flags. models are mounted, not baked in.
|
||||
|
||||
# ---- stage 1: build ----
|
||||
FROM ubuntu:24.04 AS build
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
RUN apt-get update && apt-get install -y \
|
||||
build-essential cmake ninja-build git ccache \
|
||||
libvulkan-dev glslc vulkan-tools \
|
||||
libcurl4-openssl-dev \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone --depth 1 -b strix-halo-vulkan \
|
||||
https://github.com/Nathanw1014/llama.cpp /src/engine
|
||||
RUN cmake -B /src/engine/build -S /src/engine \
|
||||
-DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
|
||||
-DLLAMA_CURL=ON \
|
||||
&& cmake --build /src/engine/build --parallel $(nproc) \
|
||||
--target llama-server llama-cli llama-bench
|
||||
|
||||
# ---- stage 2: runtime ----
|
||||
FROM ubuntu:24.04
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x)
|
||||
RUN apt-get update && apt-get install -y software-properties-common gpg-agent \
|
||||
&& add-apt-repository -y ppa:kisak/kisak \
|
||||
&& apt-get update && apt-get install -y \
|
||||
mesa-vulkan-drivers vulkan-tools libvulkan1 \
|
||||
libcurl4 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
|
||||
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
|
||||
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
|
||||
COPY --from=build /src/engine/build/bin/libggml*.so* /app/
|
||||
COPY --from=build /src/engine/build/bin/libllama*.so* /app/
|
||||
|
||||
RUN ldconfig /app 2>/dev/null; true
|
||||
ENV LD_LIBRARY_PATH=/app
|
||||
|
||||
# models volume
|
||||
VOLUME /models
|
||||
|
||||
WORKDIR /app
|
||||
EXPOSE 8080
|
||||
|
||||
CMD ["/app/llama-server", \
|
||||
"-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \
|
||||
"-ngl", "999", "-fa", "on", \
|
||||
"-ctk", "q8_0", "-ctv", "q8_0", \
|
||||
"-c", "32768", "-ub", "2048", "-t", "4", \
|
||||
"--jinja", "--host", "0.0.0.0", "--port", "8080"]
|
||||
@@ -0,0 +1,165 @@
|
||||
<div align="center">
|
||||
|
||||
# haloq38flash
|
||||
|
||||
**qwen3.8-flash-next on amd strix halo — 91g quant, 56 tok/s, 262k context**
|
||||
|
||||
[](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF)
|
||||
[](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan)
|
||||
[](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE)
|
||||
|
||||
[](#results)
|
||||
[](#results)
|
||||
[](#the-converter-bug)
|
||||
|
||||
*every published quant byte-traced back to the official checkpoint*
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
## results
|
||||
|
||||
engine: [nathanw1014/llama.cpp `strix-halo-vulkan`](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) · vulkan/radv · q8_0 kv · `-ub 2048` · temp 0
|
||||
|
||||
| depth | plain pp/tg | mtp pp/tg |
|
||||
|------:|:-----------:|:---------:|
|
||||
| 0 | 92.5 / 29.9 | 87.0 / **53.1** |
|
||||
| 8k | 480 / 24.1 | 458 / **56.4** |
|
||||
| 32k | 397 / 20.1 | 379 / **30.2** |
|
||||
| 128k | 222 / 11.0 | 214 / 18.6 |
|
||||
| 256k | 139 / 6.2 | — |
|
||||
|
||||
> [!NOTE]
|
||||
> no collapse through 32k. the 128k+ falloff is context-mechanics
|
||||
> (sparse-attention indexer), not quant size — see the reversal below.
|
||||
|
||||
<details>
|
||||
<summary><b>the 128k reversal — the PLE quant loses under MTP at depth</b></summary>
|
||||
|
||||
at ≤32k the PLE quant wins everywhere. at 128k under mtp it *loses* to the
|
||||
static 116g (18.6 vs 26.9 t/s). plausible mechanism: iq4_nl noise in the
|
||||
n-gram table compounds over deep history and lowers draft acceptance.
|
||||
single runs, n=1 caveat. pick your file by use case — see the table above.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
|
||||
## 📦 published quants
|
||||
|
||||
[huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF)
|
||||
|
||||
| file | size | pick it when |
|
||||
|------|------|:------------:|
|
||||
| `...-IQ4_XS-`**`PLE`**`.gguf` | 91 giB | ctx ≤ 32k — wins everywhere, mtp to 56 t/s |
|
||||
| `...-IQ4_XS.gguf` | 116 giB | ctx ≥ 128k — faster mtp at depth, wider fork compat |
|
||||
| `mtp-...-Q8_0.gguf` | 3.9 giB | mtp sidecar for nathanw1014-lineage engines |
|
||||
|
||||
<details>
|
||||
<summary><b>the PLE cut — why the 51b n-gram table tolerates 4-bit</b></summary>
|
||||
|
||||
the PLE table is gathered 16 random rows per token via hash lookup — there is
|
||||
no matmul on the table itself, and no two consecutive tokens hit the same rows.
|
||||
the rows tolerate iq4_nl (4.25 bpw) with no measurable degradation across the
|
||||
depth sweep. the cut: `--tensor-type "per_layer_token_embd=IQ4_XS"` on our
|
||||
quantizer → 54g → 27g.
|
||||
|
||||
**fork caveat:** engines that feed gathered PLE rows straight into mul_mat as
|
||||
quantized B operands assert (apepojken-class, ggml-vulkan.cpp:7794). verified
|
||||
working on nathanw1014 strix-halo-vulkan and rocmfpx.
|
||||
|
||||
</details>
|
||||
|
||||
### pick your setup
|
||||
|
||||
| your use case | quant | engine | ctx | expect |
|
||||
|---|---|---|---|---|
|
||||
| coding agents, chat | 91g PLE | [nathanw1014 vulkan](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) + mtp | ≤ 32k | 56 t/s |
|
||||
| long documents | 116g static | same engine + mtp | 128k | 27 t/s |
|
||||
| full rag / research | 116g static | [rocm 10 container](https://github.com/MorezMartin/engramhalo-rocm10) + ssd streaming | 262k | 14 t/s |
|
||||
|
||||
---
|
||||
|
||||
## 🐛 the converter bug
|
||||
|
||||
our first quant printed deterministic garbage at temp 0. bisect to root cause:
|
||||
|
||||
- experts, gdn reorder, ple scale, metadata: all innocent
|
||||
- **97 of 388 f32 tensors differed by exactly 1.0** — every hyper-connection
|
||||
norm shipped raw where the runtime expects `raw + 1`
|
||||
- cause: the checkpoint nests hyper-connections under
|
||||
`attn_hyper_connection` / `mlp_hyper_connection` / `hyper_connection_mixer`,
|
||||
and those names hit early-return branches in the converter that bypass the
|
||||
generic `norm.weight → +1` rule
|
||||
|
||||
> [!WARNING]
|
||||
> **any fork rolling its own qwen4exp converter must fold `(1 + w)` into the
|
||||
> hyper-connection gammas.** upstream runtime documents the contract at
|
||||
> `qwen4exp.cpp:231`. miss it and every layer normalizes wrong — garbage
|
||||
> from layer 0, all shapes correct, all shape-only tests pass.
|
||||
|
||||
fix + regression test: rocmfpx `port-qwen4exp` commit `61b6a3b48`
|
||||
([pr charlie12345/ROCmFPX#98](https://github.com/charlie12345/ROCmFPX/pull/98))
|
||||
|
||||
---
|
||||
|
||||
## 🐳 docker
|
||||
|
||||
```bash
|
||||
git clone https://github.com/julianmb/haloq38flash && cd haloq38flash
|
||||
docker compose up --build
|
||||
# serve on :8080 — vulkan/radv, no rocm install needed
|
||||
```
|
||||
|
||||
add the mtp sidecar for 56 t/s:
|
||||
|
||||
```bash
|
||||
docker compose run qwen38-flash-next /app/llama-server \
|
||||
-md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
|
||||
--spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔬 engine merge (paused)
|
||||
|
||||
we merged nathanw1014's branch (175 commits: vulkan perf stack, qwen4exp
|
||||
runtime past the squash, lazy ple, spec-decode fixes) into ggml-org master —
|
||||
one engine with lazy ple (262k on 66g resident) + depth fixes + master's
|
||||
general improvements. paused at 53% build: the branches diverged semantically
|
||||
in shared enum/base-class files.
|
||||
|
||||
- [`docs/engine-cherry-pick-plan.md`](docs/engine-cherry-pick-plan.md) — full
|
||||
175-commit classification
|
||||
- [`docs/engine-merge-status.md`](docs/engine-merge-status.md) — resume point
|
||||
|
||||
---
|
||||
|
||||
## 📁 layout
|
||||
|
||||
| path | what |
|
||||
|------|------|
|
||||
| `models/` | symlink farm to `/mnt/ssd2/models/` (never in git) |
|
||||
| `results/` | dated receipts, one per investigation |
|
||||
| `scripts/` | conversion pipeline, oracle, depth bench, resume helpers |
|
||||
| `reddit/` | archived community threads that drove the investigation |
|
||||
| `docs/` | engine merge plan, cherry-pick classification |
|
||||
|
||||
---
|
||||
|
||||
## ⚠️ operational gotchas (128g strix halo)
|
||||
|
||||
- always `-c 8192`-bounded ctx + `timeout` + `/usr/bin/time -v` — the gguf
|
||||
default 262144 + full offload hard-hung this box once
|
||||
- `vm.dirty_ratio=15 / dirty_background_ratio=5` — the 191g ple conversion
|
||||
memmap wedges `balance_dirty_pages` for hours at kernel defaults
|
||||
- conversion peak: ple scratch (191g) + f16 output (354g) coexist — budget
|
||||
~560g free
|
||||
- `pkill -x llama-cli`, never `-f` (matches your own wrapper shell)
|
||||
- gpu memory is shared with everything else on the apu — two engines cannot
|
||||
hold ~90g+ models simultaneously without an oom cascade
|
||||
|
||||
---
|
||||
|
||||
license: [qwen community license 1.0](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE) · base model: [Qwen/Qwen3.8-Flash-Next](https://huggingface.co/Qwen/Qwen3.8-Flash-Next)
|
||||
@@ -0,0 +1,23 @@
|
||||
services:
|
||||
qwen38-flash-next:
|
||||
build: .
|
||||
image: haloq38flash:latest
|
||||
container_name: qwen38-flash-next
|
||||
devices:
|
||||
- /dev/dri
|
||||
group_add:
|
||||
- video
|
||||
- render
|
||||
security_opt:
|
||||
- seccomp=unconfined
|
||||
volumes:
|
||||
- /mnt/ssd2/models/qwen38-flash-next:/models:ro
|
||||
ports:
|
||||
- "8080:8080"
|
||||
environment:
|
||||
- LD_LIBRARY_PATH=/app
|
||||
# override CMD to change context, quant, or add mtp:
|
||||
# docker compose run qwen38-flash-next \
|
||||
# /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \
|
||||
# -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
|
||||
# --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
||||
@@ -0,0 +1,123 @@
|
||||
# engine cherry-pick plan — nathan/strix-halo-vulkan → ggml-org master
|
||||
|
||||
date: 2026-08-31. base for counting: merge-base `9f0d017ef` (#27235 era).
|
||||
nathan branch tip: `ad914eb65`. #27742 landed in master as squash `6c84c7d5d`.
|
||||
nathan's branch = the #27742 development history + his strix-halo patch stack +
|
||||
three master merges he already did + the qwen4exp runtime continued past the
|
||||
squash point.
|
||||
|
||||
## counts
|
||||
|
||||
- 175 commits on `6c84c7d5d..nathan/strix-halo-vulkan` (reverse order in
|
||||
`docs/nathan-175-commits.txt`)
|
||||
- ~40 of them are the #27742 development history — **skip**, master's squash
|
||||
`6c84c7d5d` already carries that content
|
||||
- ~8 are master commits that reached his branch via his three master merges
|
||||
(muse glimmer #26841/#26879, motif-3, dspark #27508, kv-cell #27762) —
|
||||
**skip**, master has them
|
||||
- 4 are CI/toolbox release plumbing — **skip** (fork-specific)
|
||||
- ~10 are merge commits — **skip** (resolved by the one big merge below)
|
||||
- **~115 genuine candidates**, grouped below
|
||||
|
||||
## recommendation: one merge, not 115 cherry-picks
|
||||
|
||||
the qwen4exp runtime commits and the vulkan shader stack interleave (the
|
||||
sparse-FA shaders are prerequisites for the qwen4exp QSA gather path; the FACP
|
||||
refactor renames classes the later commits use). piecemeal cherry-picking
|
||||
breaks the build between commits. instead:
|
||||
|
||||
```
|
||||
git checkout -b haloq38flash-engine ggml-org/master # or origin/master
|
||||
git merge nathan/strix-halo-vulkan
|
||||
# resolve conflicts once: ggml-vulkan mostly takes THEIRS (the perf stack),
|
||||
# src/llama*.cpp mixed, everything else master
|
||||
cmake -B build -DGGML_VULKAN=ON && cmake --build build -j 24
|
||||
```
|
||||
|
||||
nathan already merged master into his branch three times
|
||||
(`aaf4fba83`, `b7b85da9c`, `f94fad0e8`/`add19980d`) — the reverse merge is the
|
||||
same operation he proved works, and conflicts concentrate in the files he owns.
|
||||
|
||||
## group A — vulkan fa/mmq perf stack (~45, oldest first)
|
||||
|
||||
the coopmat1 FA rework, dequant-once scratch, contiguized KV, mul_mat_id tile
|
||||
probes, f16-B path, q5_K/q4_K scale caches, wave32, LDS pad tuning, the six
|
||||
env-gated perf flags now default-on. cherry-pick as a block, oldest first;
|
||||
`acd14737e FACP` and `892924042 single source of truth` are the load-bearing
|
||||
refactors the later ones sit on. skip `681675530` (marked NEGATIVE result).
|
||||
|
||||
## group B — dsv4 lightning indexer + sparse fa gather (~25)
|
||||
|
||||
`890550c0a` indexer kernels + indexed sparse FA, `5dfc01ff6` gather-to-compact
|
||||
decode, the sparse prefill split/tile/cache cluster, quantised K/V inside the
|
||||
gathers (`8b66f91c6`, `7b63cbd6b`, `6b2cade31`), small-batch union
|
||||
(`8115df4c7`..`31202f9df`). written for deepseek v4, powers qwen4exp's QSA the
|
||||
same way. NOTE: `b65c360c7` fixes multi-sequence — keep.
|
||||
|
||||
## group c — fused hyper-connection ops + command buffers (~5)
|
||||
|
||||
`2041049a4` fused HC pre/comb/post (the 3550→2800 dispatch win),
|
||||
`e709b949e` command buffers bounded by memory traffic,
|
||||
`18239a695` perf-logger flush, `0f80b884d`/`8a8fee776` UMA copy path.
|
||||
|
||||
## group d — hip/"ggml-cuda" rdna3.5 tuning (~12)
|
||||
|
||||
`64e5c14f1` kernel tuning, `f074165ae` quantized-KV FA, MMQ tile tuning
|
||||
(`71ac6c1d9`, `f70839f9a`, `86e3f34fc`), Q8_1 activation cache (`a649f1634`),
|
||||
WMMA indexer (`8209c8954`), tiled FA (`e88b92eff`), GDN tune (`910f0f25d`),
|
||||
NaN fix (`4ea44eef2`), tests (`b1282d2af`, `6e7b355cb`). named ggml-cuda
|
||||
because the hip backend rides the cuda code paths.
|
||||
|
||||
## group e — qwen4exp runtime past the squash point (~25)
|
||||
|
||||
what master's squash does NOT have:
|
||||
- `be71d63c9` quantized KV cache in the QSA attention path (the q8_0 kv fix
|
||||
our rocmfpx build lacks)
|
||||
- `631b9ffb1` decode-graph reuse + host-side PLE gather (graphs reused 68 vs 0)
|
||||
- `354390810` + `39817c476` NextN/MTP draft: sidecar AND in-file loading
|
||||
- `f32aca1c1`/`79c2d2cad`/`3849d54b8`/`d763facad`/`fdf96fcea` indexer cache in
|
||||
llama_memory_hybrid_idx, slots, names
|
||||
- `87f31259a`/`c04b3ff4b`/`8f58c2f0a` PLE history per context + iterator fix
|
||||
- `05f6575ab`/`25a796300`/`cdd2e47ae` indexer cache save/restore + slots
|
||||
- `c1d5b2d0e`/`bd92a90c4`/`671203688` random-access mmap advice for the
|
||||
gather table (the ple-ssd-streaming primitive)
|
||||
- `024b7ad93` QSA bias per block, `7073ae357` hparams shrink,
|
||||
`1486f6b88` non-unified KV in QSA, `a80d678ad` image placeholder hash,
|
||||
`562cb00bc`/`42d976771` tensor-split segments, `d6f65ff28` graph budget
|
||||
- quantizer: `7a4d5960d` PLE streaming (independently written — same fix as
|
||||
our banding), `9e2d2eb84` --tensor-type names the PLE (the flag the 91g
|
||||
quant used), `5beb9965b` f16 fallback for odd ncols,
|
||||
`5096585d6` exact output buffer
|
||||
- tests: `171ddb8df`/`086457e7b`/`77953f1e1`
|
||||
|
||||
## group f — speculative decoding fixes (~10)
|
||||
|
||||
the 7-bug stack behind "spec decode works end to end":
|
||||
`53fd8b48c` GDN state graph order, `9c5d899ff`/`f25eefeaf`/`a17e8432b`
|
||||
MTP rollback full checkpoints (apply→revert→reapply), `08a325524`
|
||||
checkpoints on device, `0eb528051` draft trimming for mtmd,
|
||||
`64e2b680a` dflash cache alignment, `397ef7c72` no_vocab special tokens.
|
||||
|
||||
## group g — optional, other archs (~10, default skip)
|
||||
|
||||
dspark bailingmoe3 (`2586f6edd`), dflash2 (`015f09c8a`/`0b0f35d0e`),
|
||||
motif-3 (`4c7f96093`/`be54e2891`/`a359e55c9`) — only if wanted; they ride
|
||||
along in the merge anyway.
|
||||
|
||||
## verification after the merge
|
||||
|
||||
1. build vulkan, zero errors
|
||||
2. our depth bench on the 91g quant: 0/8k/32k/128k — expect >= the nathanw1014
|
||||
numbers (29.9/24.1/20.1 plain, 53.1/56.4/18.6 mtp) plus master's 771 commits
|
||||
3. greedy oracle: with `39817c476` spec decode + the rollback fixes, the
|
||||
6-line divergence on our current engine should close (his fork is the one
|
||||
the 7-bug fix stack was written for)
|
||||
4. lazy ple: `--tensor-read-lazy auto` must log "lazy read enabled" for
|
||||
per_layer_token_embd — the load_mode=none hardcode does not exist on
|
||||
master's path
|
||||
5. then 256k mtp: no thrash expected (66g resident with lazy ple)
|
||||
|
||||
## candidates for upstreaming after validation
|
||||
|
||||
radix/sparse top-k fa, fused hc epilogs, gdn concat fix, the q8_0-kv-in-qsa
|
||||
fix, the lazy ple plumbing, the converter trap note (hc norms).
|
||||
Executable
+56
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env bash
|
||||
# convert-flash-next-rocmfpx.sh — Qwen3.8-Flash-Next FP8 safetensors -> ROCmFP4_FAST GGUF
|
||||
#
|
||||
# Lives in haloq38flash; the engine (converter + llama-quantize) is ~/source/ROCmFPX
|
||||
# (branch port-qwen4exp, PR #98).
|
||||
#
|
||||
# Pipeline:
|
||||
# 1. convert_hf_to_gguf.py -> F16 GGUF (PLE fp8 scale captured + applied)
|
||||
# 2. llama-quantize -> Q4_0_ROCMFP4_FAST (per_layer_token_embd protected to Q8_0;
|
||||
# banded/streaming quantizer keeps RAM bounded)
|
||||
# 3. smoke: llama-completion (NOT llama-cli) with bounded -c and a timer
|
||||
#
|
||||
set -eo pipefail
|
||||
|
||||
SRC_DIR="${SRC_DIR:-/mnt/ssd2/models/qwen38-flash-next/meta-fp8}"
|
||||
WORK_DIR="${WORK_DIR:-/mnt/ssd2/models/qwen38-flash-next}"
|
||||
ENGINE="${ENGINE:-/home/user/source/ROCmFPX}"
|
||||
OUT_F16="${WORK_DIR}/Qwen3.8-Flash-Next-F16.gguf"
|
||||
OUT_QUANT="${WORK_DIR}/Qwen3.8-Flash-Next-ROCmFP4_FAST.gguf"
|
||||
SMOKE_CTX="${SMOKE_CTX:-8192}"
|
||||
|
||||
mkdir -p "${WORK_DIR}"
|
||||
|
||||
if [ ! -f "${SRC_DIR}/config.json" ]; then
|
||||
echo "ERROR: ${SRC_DIR}/config.json missing - download not complete?" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "=== [1/3] convert FP8 safetensors -> F16 GGUF ==="
|
||||
if [ ! -f "${OUT_F16}" ]; then
|
||||
cd "${ENGINE}"
|
||||
setsid nohup python3 convert_hf_to_gguf.py "${SRC_DIR}" \
|
||||
--outfile "${OUT_F16}" \
|
||||
--outtype f16
|
||||
else
|
||||
echo "F16 GGUF exists, skipping: ${OUT_F16}"
|
||||
fi
|
||||
|
||||
echo "=== [2/3] quantize -> Q4_0_ROCMFP4_FAST ==="
|
||||
if [ ! -f "${OUT_QUANT}" ]; then
|
||||
/usr/bin/time -v "${ENGINE}/build-strix-rocmfp4/bin/llama-quantize" \
|
||||
"${OUT_F16}" \
|
||||
"${OUT_QUANT}" \
|
||||
Q4_0_ROCMFP4_FAST
|
||||
else
|
||||
echo "quant exists, skipping: ${OUT_QUANT}"
|
||||
fi
|
||||
|
||||
echo "=== [3/3] smoke: llama-completion, bounded context, timer ==="
|
||||
/usr/bin/time -v timeout 900 "${ENGINE}/build-strix-rocmfp4/bin/llama-completion" \
|
||||
-m "${OUT_QUANT}" \
|
||||
-dev Vulkan0 -ngl 47 -c "${SMOKE_CTX}" -fa on -ub 2048 \
|
||||
-p "The capital of France is" -n 64 --temp 0 -no-cnv --simple-io 2>&1 | tail -40
|
||||
|
||||
ls -lah "${OUT_F16}" "${OUT_QUANT}"
|
||||
echo "DONE"
|
||||
Executable
+47
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env bash
|
||||
# Decode + prefill vs context depth, with and without MTP.
|
||||
#
|
||||
# Uses filler prompts of ~8k and ~32k tokens (results/filler/) and reports the
|
||||
# [ Prompt: X t/s | Generation: Y t/s ] line llama-cli prints. --reasoning off
|
||||
# keeps the 128-token generation budget from being eaten by a thinking block.
|
||||
# q8_0 KV keeps the cache small at depth.
|
||||
#
|
||||
# usage: scripts/depth-bench-strix-halo-vulkan.sh [depths] # e.g. "8k 32k"
|
||||
set -u
|
||||
|
||||
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||
FILLER=$OUT/filler
|
||||
SHORT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example."
|
||||
DEPTHS=${1:-0 8k 32k}
|
||||
|
||||
declare -A PROMPT_CTX=( [0]=8192 [8k]=16384 [32k]=40960 [128k]=139264 [256k]=257024 )
|
||||
|
||||
run() {
|
||||
local depth=$1 mode=$2
|
||||
local tag="depth$depth-$mode"
|
||||
local log=$OUT/${TAGPREFIX:-}shvd-$tag.log
|
||||
local args=()
|
||||
if [ "$depth" = "0" ]; then
|
||||
args+=(-p "$SHORT")
|
||||
else
|
||||
args+=(-f "$FILLER/filler-$depth.txt")
|
||||
fi
|
||||
[ "$mode" = "mtp" ] && args+=(-md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max 6 --spec-draft-p-min 0.75)
|
||||
|
||||
timeout 2400 "$BIN/llama-cli" -m "$TARGET" "${args[@]}" \
|
||||
-dev Vulkan0 -ngl 999 -c "${PROMPT_CTX[$depth]}" -fa on -ub 2048 \
|
||||
-ctk q8_0 -ctv q8_0 \
|
||||
-n 128 --temp 0 --reasoning off -no-cnv -st --simple-io > "$log" 2>&1
|
||||
printf '%-18s exit=%-3s %s %s\n' "$tag" "$?" \
|
||||
"$(grep -oE 'Prompt: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)"
|
||||
}
|
||||
|
||||
for d in $DEPTHS; do
|
||||
run "$d" plain
|
||||
run "$d" mtp
|
||||
done
|
||||
Executable
+133
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect a remote GGUF's metadata + tensor names without downloading the weights.
|
||||
|
||||
The interesting part of a GGUF (architecture, block_count, nextn_predict_layers,
|
||||
tensor names) all lives in the header, which is a few MB even for a 100 GB file.
|
||||
This range-fetches the first chunk and parses the header directly.
|
||||
|
||||
usage:
|
||||
scripts/gguf-header-peek.py <hf-repo-id> [filename-substring]
|
||||
scripts/gguf-header-peek.py /path/to/local.gguf
|
||||
|
||||
examples:
|
||||
scripts/gguf-header-peek.py EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF
|
||||
scripts/gguf-header-peek.py unsloth/Qwen3.8-Flash-Next-GGUF Q3_K_XL
|
||||
"""
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
CHUNK = 96 * 1024 * 1024 # header is dominated by the tokenizer strings
|
||||
|
||||
SCALAR_BYTES = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
|
||||
SCALAR_FMT = {0: "<B", 1: "<b", 2: "<H", 3: "<h", 4: "<I", 5: "<i",
|
||||
6: "<f", 7: "<B", 10: "<Q", 11: "<q", 12: "<d"}
|
||||
TYPE_NAMES = {0: "u8", 1: "i8", 2: "u16", 3: "i16", 4: "u32", 5: "i32", 6: "f32",
|
||||
7: "bool", 8: "str", 9: "arr", 10: "u64", 11: "i64", 12: "f64"}
|
||||
|
||||
|
||||
class Reader:
|
||||
def __init__(self, fh):
|
||||
self.fh = fh
|
||||
|
||||
def u32(self):
|
||||
return struct.unpack("<I", self.fh.read(4))[0]
|
||||
|
||||
def u64(self):
|
||||
return struct.unpack("<Q", self.fh.read(8))[0]
|
||||
|
||||
def string(self):
|
||||
return self.fh.read(self.u64()).decode("utf-8", "replace")
|
||||
|
||||
def value(self, vtype):
|
||||
if vtype == 8:
|
||||
return self.string()
|
||||
if vtype == 9:
|
||||
sub, count = self.u32(), self.u64()
|
||||
if count > 32:
|
||||
for _ in range(count):
|
||||
self.value(sub)
|
||||
return f"<{count} x {TYPE_NAMES.get(sub, sub)}>"
|
||||
return [self.value(sub) for _ in range(count)]
|
||||
if vtype in SCALAR_BYTES:
|
||||
raw = self.fh.read(SCALAR_BYTES[vtype])
|
||||
if len(raw) < SCALAR_BYTES[vtype]:
|
||||
raise ValueError("truncated header: increase CHUNK")
|
||||
if vtype == 7:
|
||||
return bool(raw[0])
|
||||
return struct.unpack(SCALAR_FMT[vtype], raw)[0]
|
||||
raise ValueError(f"unknown gguf value type {vtype}")
|
||||
|
||||
|
||||
def parse(path):
|
||||
with open(path, "rb") as fh:
|
||||
r = Reader(fh)
|
||||
assert fh.read(4) == b"GGUF", "not a GGUF file"
|
||||
version, n_tensors, n_kv = r.u32(), r.u64(), r.u64()
|
||||
kv = {r.string(): r.value(r.u32()) for _ in range(n_kv)}
|
||||
names = []
|
||||
for _ in range(n_tensors):
|
||||
names.append(r.string())
|
||||
r.fh.read(8 * r.u32()) # dims
|
||||
r.fh.read(12) # type + offset
|
||||
return version, kv, names
|
||||
|
||||
|
||||
def fetch(repo, want):
|
||||
listing = subprocess.run(
|
||||
["curl", "-s", f"https://huggingface.co/api/models/{repo}"],
|
||||
capture_output=True, text=True, check=True).stdout
|
||||
import json
|
||||
files = [s["rfilename"] for s in json.loads(listing)["siblings"]
|
||||
if s["rfilename"].endswith(".gguf")]
|
||||
matches = [f for f in files if want in f] if want else files
|
||||
if not matches:
|
||||
raise SystemExit(f"no .gguf matching {want!r} in {repo}")
|
||||
name = sorted(matches)[0]
|
||||
print(f"repo: {repo}\nfile: {name} (of {len(files)} gguf files)")
|
||||
url = f"https://huggingface.co/{repo}/resolve/main/{name}"
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".gguf", delete=False)
|
||||
subprocess.run(["curl", "-sL", "-r", f"0-{CHUNK}", "-o", tmp.name, url], check=True)
|
||||
return tmp.name
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
raise SystemExit(__doc__)
|
||||
target = sys.argv[1]
|
||||
if target.startswith(("http", "/")) and target.endswith(".gguf"):
|
||||
path = target if target.startswith("/") else fetch(target, "")
|
||||
else:
|
||||
path = fetch(target, sys.argv[2] if len(sys.argv) > 2 else "")
|
||||
|
||||
version, kv, names = parse(path)
|
||||
arch = kv.get("general.architecture", "?")
|
||||
print(f"gguf v{version} arch={arch} tensors={len(names)} kv={len(kv)}")
|
||||
print("\n-- key metadata --")
|
||||
for k in sorted(kv):
|
||||
if k.startswith("tokenizer.") or k.startswith("general."):
|
||||
continue
|
||||
v = kv[k]
|
||||
if isinstance(v, list) and len(v) > 12:
|
||||
v = f"{v[:12]} ... (len {len(v)})"
|
||||
print(f" {k:<46} {v}")
|
||||
|
||||
blocks = sorted({n.split(".")[1] for n in names
|
||||
if n.startswith("blk.") and n.split(".")[1].isdigit()})
|
||||
print(f"\n-- block indices: {blocks[:12]}{' ...' if len(blocks) > 12 else ''} "
|
||||
f"({len(blocks)} total)")
|
||||
suffixes = sorted({n.split(".", 2)[2] for n in names
|
||||
if n.startswith("blk.") and len(n.split(".")) > 2})
|
||||
print(f"-- per-block suffixes ({len(suffixes)}):")
|
||||
for s in suffixes:
|
||||
print(f" {s}")
|
||||
other = sorted(n for n in names if not n.startswith("blk."))
|
||||
if other:
|
||||
print("-- non-block tensors:")
|
||||
for n in other:
|
||||
print(f" {n}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+65
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
# A/B speculative decoding with MTP on the Nathanw1014 strix-halo-vulkan build.
|
||||
#
|
||||
# Runs the same greedy prompt with and without the MTP draft and diffs the text.
|
||||
# At temp 0 the two must be identical (a "greedy identity oracle"): any
|
||||
# divergence means the draft path is corrupting state.
|
||||
#
|
||||
# Uses the ORIGINAL sidecar (block_count = 49, blk.48), not the renumbered
|
||||
# -blk0 one -- the runtime selects the trailing block itself.
|
||||
# See results/2026-08-29-post-reboot-validation.md §6b.
|
||||
#
|
||||
# Tool notes: llama-cli, not llama-completion (the latter's parser rejects -md);
|
||||
# -st is required or cli sits in an interactive loop printing "> " forever.
|
||||
#
|
||||
# usage: scripts/mtp-test-strix-halo-vulkan.sh [plain|mtp|both]
|
||||
# NMAX=2,4,6 scripts/mtp-test-strix-halo-vulkan.sh mtp # sweep depths
|
||||
set -u
|
||||
|
||||
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||
# long-form prompt: the model emits EOS early on short factual ones, which makes
|
||||
# the tok/s figure meaningless
|
||||
PROMPT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example. Then explain how the memoization cache works."
|
||||
CTX=8192
|
||||
NPRED=512
|
||||
NMAX_LIST=${NMAX:-6}
|
||||
|
||||
run() {
|
||||
local tag=$1; shift
|
||||
local log=$OUT/${TAGPREFIX:-}shv-$tag.log
|
||||
/usr/bin/time -v timeout 1800 "$BIN/llama-cli" \
|
||||
-m "$TARGET" "$@" \
|
||||
-dev Vulkan0 -ngl 999 -c "$CTX" -fa on -ub 2048 \
|
||||
-p "$PROMPT" -n "$NPRED" --temp 0 -no-cnv -st --simple-io > "$log" 2>&1
|
||||
local rc=$?
|
||||
# generated text sits between the "> " echo and the timing line
|
||||
awk '/^> /{f=1} /^\[ Prompt:/{f=0} f' "$log" > "$OUT/${TAGPREFIX:-}shv-$tag.txt"
|
||||
printf '%-16s exit=%s %s (%s bytes of output)\n' "$tag" "$rc" \
|
||||
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||
"$(wc -c < "$OUT/shv-$tag.txt")"
|
||||
}
|
||||
|
||||
case ${1:-both} in
|
||||
plain) run plain ;;
|
||||
mtp)
|
||||
for n in ${NMAX_LIST//,/ }; do
|
||||
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||
done ;;
|
||||
both)
|
||||
run plain
|
||||
for n in ${NMAX_LIST//,/ }; do
|
||||
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||
if diff -q "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" > /dev/null; then
|
||||
echo " greedy identity n=$n: PASS"
|
||||
else
|
||||
echo " greedy identity n=$n: FAIL"
|
||||
diff "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" | head -10
|
||||
fi
|
||||
done ;;
|
||||
*) echo "usage: $0 [plain|mtp|both]" ; exit 1 ;;
|
||||
esac
|
||||
Reference in New Issue
Block a user