haloq38flash — qwen3.8-flash-next on strix halo: converter fix, 91g provenance-verified quant, engine a/b, depth tables through 256k
This commit is contained in:
@@ -0,0 +1,6 @@
|
|||||||
|
models/
|
||||||
|
weights
|
||||||
|
.codegraph
|
||||||
|
.omo/
|
||||||
|
wiki-*.parquet
|
||||||
|
corpus/
|
||||||
+54
@@ -0,0 +1,54 @@
|
|||||||
|
# haloq38flash — qwen3.8-flash-next on strix halo (vulkan/radv)
|
||||||
|
# builds the nathanw1014 strix-halo-vulkan engine and serves with the
|
||||||
|
# recommended flags. models are mounted, not baked in.
|
||||||
|
|
||||||
|
# ---- stage 1: build ----
|
||||||
|
FROM ubuntu:24.04 AS build
|
||||||
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
|
RUN apt-get update && apt-get install -y \
|
||||||
|
build-essential cmake ninja-build git ccache \
|
||||||
|
libvulkan-dev glslc vulkan-tools \
|
||||||
|
libcurl4-openssl-dev \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
RUN git clone --depth 1 -b strix-halo-vulkan \
|
||||||
|
https://github.com/Nathanw1014/llama.cpp /src/engine
|
||||||
|
RUN cmake -B /src/engine/build -S /src/engine \
|
||||||
|
-DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON \
|
||||||
|
-DLLAMA_CURL=ON \
|
||||||
|
&& cmake --build /src/engine/build --parallel $(nproc) \
|
||||||
|
--target llama-server llama-cli llama-bench
|
||||||
|
|
||||||
|
# ---- stage 2: runtime ----
|
||||||
|
FROM ubuntu:24.04
|
||||||
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
|
|
||||||
|
# add kisak ppa for recent mesa/radv (gfx1151 needs >= 24.x)
|
||||||
|
RUN apt-get update && apt-get install -y software-properties-common gpg-agent \
|
||||||
|
&& add-apt-repository -y ppa:kisak/kisak \
|
||||||
|
&& apt-get update && apt-get install -y \
|
||||||
|
mesa-vulkan-drivers vulkan-tools libvulkan1 \
|
||||||
|
libcurl4 \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
COPY --from=build /src/engine/build/bin/llama-server /app/llama-server
|
||||||
|
COPY --from=build /src/engine/build/bin/llama-cli /app/llama-cli
|
||||||
|
COPY --from=build /src/engine/build/bin/llama-bench /app/llama-bench
|
||||||
|
COPY --from=build /src/engine/build/bin/libggml*.so* /app/
|
||||||
|
COPY --from=build /src/engine/build/bin/libllama*.so* /app/
|
||||||
|
|
||||||
|
RUN ldconfig /app 2>/dev/null; true
|
||||||
|
ENV LD_LIBRARY_PATH=/app
|
||||||
|
|
||||||
|
# models volume
|
||||||
|
VOLUME /models
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
EXPOSE 8080
|
||||||
|
|
||||||
|
CMD ["/app/llama-server", \
|
||||||
|
"-m", "/models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf", \
|
||||||
|
"-ngl", "999", "-fa", "on", \
|
||||||
|
"-ctk", "q8_0", "-ctv", "q8_0", \
|
||||||
|
"-c", "32768", "-ub", "2048", "-t", "4", \
|
||||||
|
"--jinja", "--host", "0.0.0.0", "--port", "8080"]
|
||||||
@@ -0,0 +1,165 @@
|
|||||||
|
<div align="center">
|
||||||
|
|
||||||
|
# haloq38flash
|
||||||
|
|
||||||
|
**qwen3.8-flash-next on amd strix halo — 91g quant, 56 tok/s, 262k context**
|
||||||
|
|
||||||
|
[](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF)
|
||||||
|
[](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan)
|
||||||
|
[](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE)
|
||||||
|
|
||||||
|
[](#results)
|
||||||
|
[](#results)
|
||||||
|
[](#the-converter-bug)
|
||||||
|
|
||||||
|
*every published quant byte-traced back to the official checkpoint*
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## results
|
||||||
|
|
||||||
|
engine: [nathanw1014/llama.cpp `strix-halo-vulkan`](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) · vulkan/radv · q8_0 kv · `-ub 2048` · temp 0
|
||||||
|
|
||||||
|
| depth | plain pp/tg | mtp pp/tg |
|
||||||
|
|------:|:-----------:|:---------:|
|
||||||
|
| 0 | 92.5 / 29.9 | 87.0 / **53.1** |
|
||||||
|
| 8k | 480 / 24.1 | 458 / **56.4** |
|
||||||
|
| 32k | 397 / 20.1 | 379 / **30.2** |
|
||||||
|
| 128k | 222 / 11.0 | 214 / 18.6 |
|
||||||
|
| 256k | 139 / 6.2 | — |
|
||||||
|
|
||||||
|
> [!NOTE]
|
||||||
|
> no collapse through 32k. the 128k+ falloff is context-mechanics
|
||||||
|
> (sparse-attention indexer), not quant size — see the reversal below.
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary><b>the 128k reversal — the PLE quant loses under MTP at depth</b></summary>
|
||||||
|
|
||||||
|
at ≤32k the PLE quant wins everywhere. at 128k under mtp it *loses* to the
|
||||||
|
static 116g (18.6 vs 26.9 t/s). plausible mechanism: iq4_nl noise in the
|
||||||
|
n-gram table compounds over deep history and lowers draft acceptance.
|
||||||
|
single runs, n=1 caveat. pick your file by use case — see the table above.
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📦 published quants
|
||||||
|
|
||||||
|
[huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF](https://huggingface.co/julianmb/Qwen3.8-Flash-Next-IQ4_XS-GGUF)
|
||||||
|
|
||||||
|
| file | size | pick it when |
|
||||||
|
|------|------|:------------:|
|
||||||
|
| `...-IQ4_XS-`**`PLE`**`.gguf` | 91 giB | ctx ≤ 32k — wins everywhere, mtp to 56 t/s |
|
||||||
|
| `...-IQ4_XS.gguf` | 116 giB | ctx ≥ 128k — faster mtp at depth, wider fork compat |
|
||||||
|
| `mtp-...-Q8_0.gguf` | 3.9 giB | mtp sidecar for nathanw1014-lineage engines |
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary><b>the PLE cut — why the 51b n-gram table tolerates 4-bit</b></summary>
|
||||||
|
|
||||||
|
the PLE table is gathered 16 random rows per token via hash lookup — there is
|
||||||
|
no matmul on the table itself, and no two consecutive tokens hit the same rows.
|
||||||
|
the rows tolerate iq4_nl (4.25 bpw) with no measurable degradation across the
|
||||||
|
depth sweep. the cut: `--tensor-type "per_layer_token_embd=IQ4_XS"` on our
|
||||||
|
quantizer → 54g → 27g.
|
||||||
|
|
||||||
|
**fork caveat:** engines that feed gathered PLE rows straight into mul_mat as
|
||||||
|
quantized B operands assert (apepojken-class, ggml-vulkan.cpp:7794). verified
|
||||||
|
working on nathanw1014 strix-halo-vulkan and rocmfpx.
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
### pick your setup
|
||||||
|
|
||||||
|
| your use case | quant | engine | ctx | expect |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| coding agents, chat | 91g PLE | [nathanw1014 vulkan](https://github.com/Nathanw1014/llama.cpp/tree/strix-halo-vulkan) + mtp | ≤ 32k | 56 t/s |
|
||||||
|
| long documents | 116g static | same engine + mtp | 128k | 27 t/s |
|
||||||
|
| full rag / research | 116g static | [rocm 10 container](https://github.com/MorezMartin/engramhalo-rocm10) + ssd streaming | 262k | 14 t/s |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🐛 the converter bug
|
||||||
|
|
||||||
|
our first quant printed deterministic garbage at temp 0. bisect to root cause:
|
||||||
|
|
||||||
|
- experts, gdn reorder, ple scale, metadata: all innocent
|
||||||
|
- **97 of 388 f32 tensors differed by exactly 1.0** — every hyper-connection
|
||||||
|
norm shipped raw where the runtime expects `raw + 1`
|
||||||
|
- cause: the checkpoint nests hyper-connections under
|
||||||
|
`attn_hyper_connection` / `mlp_hyper_connection` / `hyper_connection_mixer`,
|
||||||
|
and those names hit early-return branches in the converter that bypass the
|
||||||
|
generic `norm.weight → +1` rule
|
||||||
|
|
||||||
|
> [!WARNING]
|
||||||
|
> **any fork rolling its own qwen4exp converter must fold `(1 + w)` into the
|
||||||
|
> hyper-connection gammas.** upstream runtime documents the contract at
|
||||||
|
> `qwen4exp.cpp:231`. miss it and every layer normalizes wrong — garbage
|
||||||
|
> from layer 0, all shapes correct, all shape-only tests pass.
|
||||||
|
|
||||||
|
fix + regression test: rocmfpx `port-qwen4exp` commit `61b6a3b48`
|
||||||
|
([pr charlie12345/ROCmFPX#98](https://github.com/charlie12345/ROCmFPX/pull/98))
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🐳 docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git clone https://github.com/julianmb/haloq38flash && cd haloq38flash
|
||||||
|
docker compose up --build
|
||||||
|
# serve on :8080 — vulkan/radv, no rocm install needed
|
||||||
|
```
|
||||||
|
|
||||||
|
add the mtp sidecar for 56 t/s:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker compose run qwen38-flash-next /app/llama-server \
|
||||||
|
-md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
|
||||||
|
--spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🔬 engine merge (paused)
|
||||||
|
|
||||||
|
we merged nathanw1014's branch (175 commits: vulkan perf stack, qwen4exp
|
||||||
|
runtime past the squash, lazy ple, spec-decode fixes) into ggml-org master —
|
||||||
|
one engine with lazy ple (262k on 66g resident) + depth fixes + master's
|
||||||
|
general improvements. paused at 53% build: the branches diverged semantically
|
||||||
|
in shared enum/base-class files.
|
||||||
|
|
||||||
|
- [`docs/engine-cherry-pick-plan.md`](docs/engine-cherry-pick-plan.md) — full
|
||||||
|
175-commit classification
|
||||||
|
- [`docs/engine-merge-status.md`](docs/engine-merge-status.md) — resume point
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📁 layout
|
||||||
|
|
||||||
|
| path | what |
|
||||||
|
|------|------|
|
||||||
|
| `models/` | symlink farm to `/mnt/ssd2/models/` (never in git) |
|
||||||
|
| `results/` | dated receipts, one per investigation |
|
||||||
|
| `scripts/` | conversion pipeline, oracle, depth bench, resume helpers |
|
||||||
|
| `reddit/` | archived community threads that drove the investigation |
|
||||||
|
| `docs/` | engine merge plan, cherry-pick classification |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## ⚠️ operational gotchas (128g strix halo)
|
||||||
|
|
||||||
|
- always `-c 8192`-bounded ctx + `timeout` + `/usr/bin/time -v` — the gguf
|
||||||
|
default 262144 + full offload hard-hung this box once
|
||||||
|
- `vm.dirty_ratio=15 / dirty_background_ratio=5` — the 191g ple conversion
|
||||||
|
memmap wedges `balance_dirty_pages` for hours at kernel defaults
|
||||||
|
- conversion peak: ple scratch (191g) + f16 output (354g) coexist — budget
|
||||||
|
~560g free
|
||||||
|
- `pkill -x llama-cli`, never `-f` (matches your own wrapper shell)
|
||||||
|
- gpu memory is shared with everything else on the apu — two engines cannot
|
||||||
|
hold ~90g+ models simultaneously without an oom cascade
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
license: [qwen community license 1.0](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/LICENSE) · base model: [Qwen/Qwen3.8-Flash-Next](https://huggingface.co/Qwen/Qwen3.8-Flash-Next)
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
services:
|
||||||
|
qwen38-flash-next:
|
||||||
|
build: .
|
||||||
|
image: haloq38flash:latest
|
||||||
|
container_name: qwen38-flash-next
|
||||||
|
devices:
|
||||||
|
- /dev/dri
|
||||||
|
group_add:
|
||||||
|
- video
|
||||||
|
- render
|
||||||
|
security_opt:
|
||||||
|
- seccomp=unconfined
|
||||||
|
volumes:
|
||||||
|
- /mnt/ssd2/models/qwen38-flash-next:/models:ro
|
||||||
|
ports:
|
||||||
|
- "8080:8080"
|
||||||
|
environment:
|
||||||
|
- LD_LIBRARY_PATH=/app
|
||||||
|
# override CMD to change context, quant, or add mtp:
|
||||||
|
# docker compose run qwen38-flash-next \
|
||||||
|
# /app/llama-server -m /models/Qwen3.8-Flash-Next-IQ4_XS-PLE.gguf \
|
||||||
|
# -c 131072 -md /models/mtp-Qwen3.8-Flash-Next-Q8_0.gguf \
|
||||||
|
# --spec-type draft-mtp --spec-draft-n-max 6 --spec-draft-p-min 0.75
|
||||||
@@ -0,0 +1,123 @@
|
|||||||
|
# engine cherry-pick plan — nathan/strix-halo-vulkan → ggml-org master
|
||||||
|
|
||||||
|
date: 2026-08-31. base for counting: merge-base `9f0d017ef` (#27235 era).
|
||||||
|
nathan branch tip: `ad914eb65`. #27742 landed in master as squash `6c84c7d5d`.
|
||||||
|
nathan's branch = the #27742 development history + his strix-halo patch stack +
|
||||||
|
three master merges he already did + the qwen4exp runtime continued past the
|
||||||
|
squash point.
|
||||||
|
|
||||||
|
## counts
|
||||||
|
|
||||||
|
- 175 commits on `6c84c7d5d..nathan/strix-halo-vulkan` (reverse order in
|
||||||
|
`docs/nathan-175-commits.txt`)
|
||||||
|
- ~40 of them are the #27742 development history — **skip**, master's squash
|
||||||
|
`6c84c7d5d` already carries that content
|
||||||
|
- ~8 are master commits that reached his branch via his three master merges
|
||||||
|
(muse glimmer #26841/#26879, motif-3, dspark #27508, kv-cell #27762) —
|
||||||
|
**skip**, master has them
|
||||||
|
- 4 are CI/toolbox release plumbing — **skip** (fork-specific)
|
||||||
|
- ~10 are merge commits — **skip** (resolved by the one big merge below)
|
||||||
|
- **~115 genuine candidates**, grouped below
|
||||||
|
|
||||||
|
## recommendation: one merge, not 115 cherry-picks
|
||||||
|
|
||||||
|
the qwen4exp runtime commits and the vulkan shader stack interleave (the
|
||||||
|
sparse-FA shaders are prerequisites for the qwen4exp QSA gather path; the FACP
|
||||||
|
refactor renames classes the later commits use). piecemeal cherry-picking
|
||||||
|
breaks the build between commits. instead:
|
||||||
|
|
||||||
|
```
|
||||||
|
git checkout -b haloq38flash-engine ggml-org/master # or origin/master
|
||||||
|
git merge nathan/strix-halo-vulkan
|
||||||
|
# resolve conflicts once: ggml-vulkan mostly takes THEIRS (the perf stack),
|
||||||
|
# src/llama*.cpp mixed, everything else master
|
||||||
|
cmake -B build -DGGML_VULKAN=ON && cmake --build build -j 24
|
||||||
|
```
|
||||||
|
|
||||||
|
nathan already merged master into his branch three times
|
||||||
|
(`aaf4fba83`, `b7b85da9c`, `f94fad0e8`/`add19980d`) — the reverse merge is the
|
||||||
|
same operation he proved works, and conflicts concentrate in the files he owns.
|
||||||
|
|
||||||
|
## group A — vulkan fa/mmq perf stack (~45, oldest first)
|
||||||
|
|
||||||
|
the coopmat1 FA rework, dequant-once scratch, contiguized KV, mul_mat_id tile
|
||||||
|
probes, f16-B path, q5_K/q4_K scale caches, wave32, LDS pad tuning, the six
|
||||||
|
env-gated perf flags now default-on. cherry-pick as a block, oldest first;
|
||||||
|
`acd14737e FACP` and `892924042 single source of truth` are the load-bearing
|
||||||
|
refactors the later ones sit on. skip `681675530` (marked NEGATIVE result).
|
||||||
|
|
||||||
|
## group B — dsv4 lightning indexer + sparse fa gather (~25)
|
||||||
|
|
||||||
|
`890550c0a` indexer kernels + indexed sparse FA, `5dfc01ff6` gather-to-compact
|
||||||
|
decode, the sparse prefill split/tile/cache cluster, quantised K/V inside the
|
||||||
|
gathers (`8b66f91c6`, `7b63cbd6b`, `6b2cade31`), small-batch union
|
||||||
|
(`8115df4c7`..`31202f9df`). written for deepseek v4, powers qwen4exp's QSA the
|
||||||
|
same way. NOTE: `b65c360c7` fixes multi-sequence — keep.
|
||||||
|
|
||||||
|
## group c — fused hyper-connection ops + command buffers (~5)
|
||||||
|
|
||||||
|
`2041049a4` fused HC pre/comb/post (the 3550→2800 dispatch win),
|
||||||
|
`e709b949e` command buffers bounded by memory traffic,
|
||||||
|
`18239a695` perf-logger flush, `0f80b884d`/`8a8fee776` UMA copy path.
|
||||||
|
|
||||||
|
## group d — hip/"ggml-cuda" rdna3.5 tuning (~12)
|
||||||
|
|
||||||
|
`64e5c14f1` kernel tuning, `f074165ae` quantized-KV FA, MMQ tile tuning
|
||||||
|
(`71ac6c1d9`, `f70839f9a`, `86e3f34fc`), Q8_1 activation cache (`a649f1634`),
|
||||||
|
WMMA indexer (`8209c8954`), tiled FA (`e88b92eff`), GDN tune (`910f0f25d`),
|
||||||
|
NaN fix (`4ea44eef2`), tests (`b1282d2af`, `6e7b355cb`). named ggml-cuda
|
||||||
|
because the hip backend rides the cuda code paths.
|
||||||
|
|
||||||
|
## group e — qwen4exp runtime past the squash point (~25)
|
||||||
|
|
||||||
|
what master's squash does NOT have:
|
||||||
|
- `be71d63c9` quantized KV cache in the QSA attention path (the q8_0 kv fix
|
||||||
|
our rocmfpx build lacks)
|
||||||
|
- `631b9ffb1` decode-graph reuse + host-side PLE gather (graphs reused 68 vs 0)
|
||||||
|
- `354390810` + `39817c476` NextN/MTP draft: sidecar AND in-file loading
|
||||||
|
- `f32aca1c1`/`79c2d2cad`/`3849d54b8`/`d763facad`/`fdf96fcea` indexer cache in
|
||||||
|
llama_memory_hybrid_idx, slots, names
|
||||||
|
- `87f31259a`/`c04b3ff4b`/`8f58c2f0a` PLE history per context + iterator fix
|
||||||
|
- `05f6575ab`/`25a796300`/`cdd2e47ae` indexer cache save/restore + slots
|
||||||
|
- `c1d5b2d0e`/`bd92a90c4`/`671203688` random-access mmap advice for the
|
||||||
|
gather table (the ple-ssd-streaming primitive)
|
||||||
|
- `024b7ad93` QSA bias per block, `7073ae357` hparams shrink,
|
||||||
|
`1486f6b88` non-unified KV in QSA, `a80d678ad` image placeholder hash,
|
||||||
|
`562cb00bc`/`42d976771` tensor-split segments, `d6f65ff28` graph budget
|
||||||
|
- quantizer: `7a4d5960d` PLE streaming (independently written — same fix as
|
||||||
|
our banding), `9e2d2eb84` --tensor-type names the PLE (the flag the 91g
|
||||||
|
quant used), `5beb9965b` f16 fallback for odd ncols,
|
||||||
|
`5096585d6` exact output buffer
|
||||||
|
- tests: `171ddb8df`/`086457e7b`/`77953f1e1`
|
||||||
|
|
||||||
|
## group f — speculative decoding fixes (~10)
|
||||||
|
|
||||||
|
the 7-bug stack behind "spec decode works end to end":
|
||||||
|
`53fd8b48c` GDN state graph order, `9c5d899ff`/`f25eefeaf`/`a17e8432b`
|
||||||
|
MTP rollback full checkpoints (apply→revert→reapply), `08a325524`
|
||||||
|
checkpoints on device, `0eb528051` draft trimming for mtmd,
|
||||||
|
`64e2b680a` dflash cache alignment, `397ef7c72` no_vocab special tokens.
|
||||||
|
|
||||||
|
## group g — optional, other archs (~10, default skip)
|
||||||
|
|
||||||
|
dspark bailingmoe3 (`2586f6edd`), dflash2 (`015f09c8a`/`0b0f35d0e`),
|
||||||
|
motif-3 (`4c7f96093`/`be54e2891`/`a359e55c9`) — only if wanted; they ride
|
||||||
|
along in the merge anyway.
|
||||||
|
|
||||||
|
## verification after the merge
|
||||||
|
|
||||||
|
1. build vulkan, zero errors
|
||||||
|
2. our depth bench on the 91g quant: 0/8k/32k/128k — expect >= the nathanw1014
|
||||||
|
numbers (29.9/24.1/20.1 plain, 53.1/56.4/18.6 mtp) plus master's 771 commits
|
||||||
|
3. greedy oracle: with `39817c476` spec decode + the rollback fixes, the
|
||||||
|
6-line divergence on our current engine should close (his fork is the one
|
||||||
|
the 7-bug fix stack was written for)
|
||||||
|
4. lazy ple: `--tensor-read-lazy auto` must log "lazy read enabled" for
|
||||||
|
per_layer_token_embd — the load_mode=none hardcode does not exist on
|
||||||
|
master's path
|
||||||
|
5. then 256k mtp: no thrash expected (66g resident with lazy ple)
|
||||||
|
|
||||||
|
## candidates for upstreaming after validation
|
||||||
|
|
||||||
|
radix/sparse top-k fa, fused hc epilogs, gdn concat fix, the q8_0-kv-in-qsa
|
||||||
|
fix, the lazy ple plumbing, the converter trap note (hc norms).
|
||||||
Executable
+56
@@ -0,0 +1,56 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# convert-flash-next-rocmfpx.sh — Qwen3.8-Flash-Next FP8 safetensors -> ROCmFP4_FAST GGUF
|
||||||
|
#
|
||||||
|
# Lives in haloq38flash; the engine (converter + llama-quantize) is ~/source/ROCmFPX
|
||||||
|
# (branch port-qwen4exp, PR #98).
|
||||||
|
#
|
||||||
|
# Pipeline:
|
||||||
|
# 1. convert_hf_to_gguf.py -> F16 GGUF (PLE fp8 scale captured + applied)
|
||||||
|
# 2. llama-quantize -> Q4_0_ROCMFP4_FAST (per_layer_token_embd protected to Q8_0;
|
||||||
|
# banded/streaming quantizer keeps RAM bounded)
|
||||||
|
# 3. smoke: llama-completion (NOT llama-cli) with bounded -c and a timer
|
||||||
|
#
|
||||||
|
set -eo pipefail
|
||||||
|
|
||||||
|
SRC_DIR="${SRC_DIR:-/mnt/ssd2/models/qwen38-flash-next/meta-fp8}"
|
||||||
|
WORK_DIR="${WORK_DIR:-/mnt/ssd2/models/qwen38-flash-next}"
|
||||||
|
ENGINE="${ENGINE:-/home/user/source/ROCmFPX}"
|
||||||
|
OUT_F16="${WORK_DIR}/Qwen3.8-Flash-Next-F16.gguf"
|
||||||
|
OUT_QUANT="${WORK_DIR}/Qwen3.8-Flash-Next-ROCmFP4_FAST.gguf"
|
||||||
|
SMOKE_CTX="${SMOKE_CTX:-8192}"
|
||||||
|
|
||||||
|
mkdir -p "${WORK_DIR}"
|
||||||
|
|
||||||
|
if [ ! -f "${SRC_DIR}/config.json" ]; then
|
||||||
|
echo "ERROR: ${SRC_DIR}/config.json missing - download not complete?" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "=== [1/3] convert FP8 safetensors -> F16 GGUF ==="
|
||||||
|
if [ ! -f "${OUT_F16}" ]; then
|
||||||
|
cd "${ENGINE}"
|
||||||
|
setsid nohup python3 convert_hf_to_gguf.py "${SRC_DIR}" \
|
||||||
|
--outfile "${OUT_F16}" \
|
||||||
|
--outtype f16
|
||||||
|
else
|
||||||
|
echo "F16 GGUF exists, skipping: ${OUT_F16}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "=== [2/3] quantize -> Q4_0_ROCMFP4_FAST ==="
|
||||||
|
if [ ! -f "${OUT_QUANT}" ]; then
|
||||||
|
/usr/bin/time -v "${ENGINE}/build-strix-rocmfp4/bin/llama-quantize" \
|
||||||
|
"${OUT_F16}" \
|
||||||
|
"${OUT_QUANT}" \
|
||||||
|
Q4_0_ROCMFP4_FAST
|
||||||
|
else
|
||||||
|
echo "quant exists, skipping: ${OUT_QUANT}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "=== [3/3] smoke: llama-completion, bounded context, timer ==="
|
||||||
|
/usr/bin/time -v timeout 900 "${ENGINE}/build-strix-rocmfp4/bin/llama-completion" \
|
||||||
|
-m "${OUT_QUANT}" \
|
||||||
|
-dev Vulkan0 -ngl 47 -c "${SMOKE_CTX}" -fa on -ub 2048 \
|
||||||
|
-p "The capital of France is" -n 64 --temp 0 -no-cnv --simple-io 2>&1 | tail -40
|
||||||
|
|
||||||
|
ls -lah "${OUT_F16}" "${OUT_QUANT}"
|
||||||
|
echo "DONE"
|
||||||
Executable
+47
@@ -0,0 +1,47 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Decode + prefill vs context depth, with and without MTP.
|
||||||
|
#
|
||||||
|
# Uses filler prompts of ~8k and ~32k tokens (results/filler/) and reports the
|
||||||
|
# [ Prompt: X t/s | Generation: Y t/s ] line llama-cli prints. --reasoning off
|
||||||
|
# keeps the 128-token generation budget from being eaten by a thinking block.
|
||||||
|
# q8_0 KV keeps the cache small at depth.
|
||||||
|
#
|
||||||
|
# usage: scripts/depth-bench-strix-halo-vulkan.sh [depths] # e.g. "8k 32k"
|
||||||
|
set -u
|
||||||
|
|
||||||
|
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||||
|
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||||
|
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||||
|
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||||
|
FILLER=$OUT/filler
|
||||||
|
SHORT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example."
|
||||||
|
DEPTHS=${1:-0 8k 32k}
|
||||||
|
|
||||||
|
declare -A PROMPT_CTX=( [0]=8192 [8k]=16384 [32k]=40960 [128k]=139264 [256k]=257024 )
|
||||||
|
|
||||||
|
run() {
|
||||||
|
local depth=$1 mode=$2
|
||||||
|
local tag="depth$depth-$mode"
|
||||||
|
local log=$OUT/${TAGPREFIX:-}shvd-$tag.log
|
||||||
|
local args=()
|
||||||
|
if [ "$depth" = "0" ]; then
|
||||||
|
args+=(-p "$SHORT")
|
||||||
|
else
|
||||||
|
args+=(-f "$FILLER/filler-$depth.txt")
|
||||||
|
fi
|
||||||
|
[ "$mode" = "mtp" ] && args+=(-md "$DRAFT" --spec-type draft-mtp \
|
||||||
|
--spec-draft-n-max 6 --spec-draft-p-min 0.75)
|
||||||
|
|
||||||
|
timeout 2400 "$BIN/llama-cli" -m "$TARGET" "${args[@]}" \
|
||||||
|
-dev Vulkan0 -ngl 999 -c "${PROMPT_CTX[$depth]}" -fa on -ub 2048 \
|
||||||
|
-ctk q8_0 -ctv q8_0 \
|
||||||
|
-n 128 --temp 0 --reasoning off -no-cnv -st --simple-io > "$log" 2>&1
|
||||||
|
printf '%-18s exit=%-3s %s %s\n' "$tag" "$?" \
|
||||||
|
"$(grep -oE 'Prompt: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||||
|
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)"
|
||||||
|
}
|
||||||
|
|
||||||
|
for d in $DEPTHS; do
|
||||||
|
run "$d" plain
|
||||||
|
run "$d" mtp
|
||||||
|
done
|
||||||
Executable
+133
@@ -0,0 +1,133 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Inspect a remote GGUF's metadata + tensor names without downloading the weights.
|
||||||
|
|
||||||
|
The interesting part of a GGUF (architecture, block_count, nextn_predict_layers,
|
||||||
|
tensor names) all lives in the header, which is a few MB even for a 100 GB file.
|
||||||
|
This range-fetches the first chunk and parses the header directly.
|
||||||
|
|
||||||
|
usage:
|
||||||
|
scripts/gguf-header-peek.py <hf-repo-id> [filename-substring]
|
||||||
|
scripts/gguf-header-peek.py /path/to/local.gguf
|
||||||
|
|
||||||
|
examples:
|
||||||
|
scripts/gguf-header-peek.py EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF
|
||||||
|
scripts/gguf-header-peek.py unsloth/Qwen3.8-Flash-Next-GGUF Q3_K_XL
|
||||||
|
"""
|
||||||
|
import struct
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
CHUNK = 96 * 1024 * 1024 # header is dominated by the tokenizer strings
|
||||||
|
|
||||||
|
SCALAR_BYTES = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
|
||||||
|
SCALAR_FMT = {0: "<B", 1: "<b", 2: "<H", 3: "<h", 4: "<I", 5: "<i",
|
||||||
|
6: "<f", 7: "<B", 10: "<Q", 11: "<q", 12: "<d"}
|
||||||
|
TYPE_NAMES = {0: "u8", 1: "i8", 2: "u16", 3: "i16", 4: "u32", 5: "i32", 6: "f32",
|
||||||
|
7: "bool", 8: "str", 9: "arr", 10: "u64", 11: "i64", 12: "f64"}
|
||||||
|
|
||||||
|
|
||||||
|
class Reader:
|
||||||
|
def __init__(self, fh):
|
||||||
|
self.fh = fh
|
||||||
|
|
||||||
|
def u32(self):
|
||||||
|
return struct.unpack("<I", self.fh.read(4))[0]
|
||||||
|
|
||||||
|
def u64(self):
|
||||||
|
return struct.unpack("<Q", self.fh.read(8))[0]
|
||||||
|
|
||||||
|
def string(self):
|
||||||
|
return self.fh.read(self.u64()).decode("utf-8", "replace")
|
||||||
|
|
||||||
|
def value(self, vtype):
|
||||||
|
if vtype == 8:
|
||||||
|
return self.string()
|
||||||
|
if vtype == 9:
|
||||||
|
sub, count = self.u32(), self.u64()
|
||||||
|
if count > 32:
|
||||||
|
for _ in range(count):
|
||||||
|
self.value(sub)
|
||||||
|
return f"<{count} x {TYPE_NAMES.get(sub, sub)}>"
|
||||||
|
return [self.value(sub) for _ in range(count)]
|
||||||
|
if vtype in SCALAR_BYTES:
|
||||||
|
raw = self.fh.read(SCALAR_BYTES[vtype])
|
||||||
|
if len(raw) < SCALAR_BYTES[vtype]:
|
||||||
|
raise ValueError("truncated header: increase CHUNK")
|
||||||
|
if vtype == 7:
|
||||||
|
return bool(raw[0])
|
||||||
|
return struct.unpack(SCALAR_FMT[vtype], raw)[0]
|
||||||
|
raise ValueError(f"unknown gguf value type {vtype}")
|
||||||
|
|
||||||
|
|
||||||
|
def parse(path):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
r = Reader(fh)
|
||||||
|
assert fh.read(4) == b"GGUF", "not a GGUF file"
|
||||||
|
version, n_tensors, n_kv = r.u32(), r.u64(), r.u64()
|
||||||
|
kv = {r.string(): r.value(r.u32()) for _ in range(n_kv)}
|
||||||
|
names = []
|
||||||
|
for _ in range(n_tensors):
|
||||||
|
names.append(r.string())
|
||||||
|
r.fh.read(8 * r.u32()) # dims
|
||||||
|
r.fh.read(12) # type + offset
|
||||||
|
return version, kv, names
|
||||||
|
|
||||||
|
|
||||||
|
def fetch(repo, want):
|
||||||
|
listing = subprocess.run(
|
||||||
|
["curl", "-s", f"https://huggingface.co/api/models/{repo}"],
|
||||||
|
capture_output=True, text=True, check=True).stdout
|
||||||
|
import json
|
||||||
|
files = [s["rfilename"] for s in json.loads(listing)["siblings"]
|
||||||
|
if s["rfilename"].endswith(".gguf")]
|
||||||
|
matches = [f for f in files if want in f] if want else files
|
||||||
|
if not matches:
|
||||||
|
raise SystemExit(f"no .gguf matching {want!r} in {repo}")
|
||||||
|
name = sorted(matches)[0]
|
||||||
|
print(f"repo: {repo}\nfile: {name} (of {len(files)} gguf files)")
|
||||||
|
url = f"https://huggingface.co/{repo}/resolve/main/{name}"
|
||||||
|
tmp = tempfile.NamedTemporaryFile(suffix=".gguf", delete=False)
|
||||||
|
subprocess.run(["curl", "-sL", "-r", f"0-{CHUNK}", "-o", tmp.name, url], check=True)
|
||||||
|
return tmp.name
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
if len(sys.argv) < 2:
|
||||||
|
raise SystemExit(__doc__)
|
||||||
|
target = sys.argv[1]
|
||||||
|
if target.startswith(("http", "/")) and target.endswith(".gguf"):
|
||||||
|
path = target if target.startswith("/") else fetch(target, "")
|
||||||
|
else:
|
||||||
|
path = fetch(target, sys.argv[2] if len(sys.argv) > 2 else "")
|
||||||
|
|
||||||
|
version, kv, names = parse(path)
|
||||||
|
arch = kv.get("general.architecture", "?")
|
||||||
|
print(f"gguf v{version} arch={arch} tensors={len(names)} kv={len(kv)}")
|
||||||
|
print("\n-- key metadata --")
|
||||||
|
for k in sorted(kv):
|
||||||
|
if k.startswith("tokenizer.") or k.startswith("general."):
|
||||||
|
continue
|
||||||
|
v = kv[k]
|
||||||
|
if isinstance(v, list) and len(v) > 12:
|
||||||
|
v = f"{v[:12]} ... (len {len(v)})"
|
||||||
|
print(f" {k:<46} {v}")
|
||||||
|
|
||||||
|
blocks = sorted({n.split(".")[1] for n in names
|
||||||
|
if n.startswith("blk.") and n.split(".")[1].isdigit()})
|
||||||
|
print(f"\n-- block indices: {blocks[:12]}{' ...' if len(blocks) > 12 else ''} "
|
||||||
|
f"({len(blocks)} total)")
|
||||||
|
suffixes = sorted({n.split(".", 2)[2] for n in names
|
||||||
|
if n.startswith("blk.") and len(n.split(".")) > 2})
|
||||||
|
print(f"-- per-block suffixes ({len(suffixes)}):")
|
||||||
|
for s in suffixes:
|
||||||
|
print(f" {s}")
|
||||||
|
other = sorted(n for n in names if not n.startswith("blk."))
|
||||||
|
if other:
|
||||||
|
print("-- non-block tensors:")
|
||||||
|
for n in other:
|
||||||
|
print(f" {n}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Executable
+65
@@ -0,0 +1,65 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# A/B speculative decoding with MTP on the Nathanw1014 strix-halo-vulkan build.
|
||||||
|
#
|
||||||
|
# Runs the same greedy prompt with and without the MTP draft and diffs the text.
|
||||||
|
# At temp 0 the two must be identical (a "greedy identity oracle"): any
|
||||||
|
# divergence means the draft path is corrupting state.
|
||||||
|
#
|
||||||
|
# Uses the ORIGINAL sidecar (block_count = 49, blk.48), not the renumbered
|
||||||
|
# -blk0 one -- the runtime selects the trailing block itself.
|
||||||
|
# See results/2026-08-29-post-reboot-validation.md §6b.
|
||||||
|
#
|
||||||
|
# Tool notes: llama-cli, not llama-completion (the latter's parser rejects -md);
|
||||||
|
# -st is required or cli sits in an interactive loop printing "> " forever.
|
||||||
|
#
|
||||||
|
# usage: scripts/mtp-test-strix-halo-vulkan.sh [plain|mtp|both]
|
||||||
|
# NMAX=2,4,6 scripts/mtp-test-strix-halo-vulkan.sh mtp # sweep depths
|
||||||
|
set -u
|
||||||
|
|
||||||
|
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||||
|
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||||
|
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||||
|
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||||
|
# long-form prompt: the model emits EOS early on short factual ones, which makes
|
||||||
|
# the tok/s figure meaningless
|
||||||
|
PROMPT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example. Then explain how the memoization cache works."
|
||||||
|
CTX=8192
|
||||||
|
NPRED=512
|
||||||
|
NMAX_LIST=${NMAX:-6}
|
||||||
|
|
||||||
|
run() {
|
||||||
|
local tag=$1; shift
|
||||||
|
local log=$OUT/${TAGPREFIX:-}shv-$tag.log
|
||||||
|
/usr/bin/time -v timeout 1800 "$BIN/llama-cli" \
|
||||||
|
-m "$TARGET" "$@" \
|
||||||
|
-dev Vulkan0 -ngl 999 -c "$CTX" -fa on -ub 2048 \
|
||||||
|
-p "$PROMPT" -n "$NPRED" --temp 0 -no-cnv -st --simple-io > "$log" 2>&1
|
||||||
|
local rc=$?
|
||||||
|
# generated text sits between the "> " echo and the timing line
|
||||||
|
awk '/^> /{f=1} /^\[ Prompt:/{f=0} f' "$log" > "$OUT/${TAGPREFIX:-}shv-$tag.txt"
|
||||||
|
printf '%-16s exit=%s %s (%s bytes of output)\n' "$tag" "$rc" \
|
||||||
|
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||||
|
"$(wc -c < "$OUT/shv-$tag.txt")"
|
||||||
|
}
|
||||||
|
|
||||||
|
case ${1:-both} in
|
||||||
|
plain) run plain ;;
|
||||||
|
mtp)
|
||||||
|
for n in ${NMAX_LIST//,/ }; do
|
||||||
|
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||||
|
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||||
|
done ;;
|
||||||
|
both)
|
||||||
|
run plain
|
||||||
|
for n in ${NMAX_LIST//,/ }; do
|
||||||
|
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||||
|
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||||
|
if diff -q "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" > /dev/null; then
|
||||||
|
echo " greedy identity n=$n: PASS"
|
||||||
|
else
|
||||||
|
echo " greedy identity n=$n: FAIL"
|
||||||
|
diff "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" | head -10
|
||||||
|
fi
|
||||||
|
done ;;
|
||||||
|
*) echo "usage: $0 [plain|mtp|both]" ; exit 1 ;;
|
||||||
|
esac
|
||||||
Reference in New Issue
Block a user