haloq38flash — qwen3.8-flash-next on strix halo: converter fix, 91g provenance-verified quant, engine a/b, depth tables through 256k
This commit is contained in:
Executable
+56
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env bash
|
||||
# convert-flash-next-rocmfpx.sh — Qwen3.8-Flash-Next FP8 safetensors -> ROCmFP4_FAST GGUF
|
||||
#
|
||||
# Lives in haloq38flash; the engine (converter + llama-quantize) is ~/source/ROCmFPX
|
||||
# (branch port-qwen4exp, PR #98).
|
||||
#
|
||||
# Pipeline:
|
||||
# 1. convert_hf_to_gguf.py -> F16 GGUF (PLE fp8 scale captured + applied)
|
||||
# 2. llama-quantize -> Q4_0_ROCMFP4_FAST (per_layer_token_embd protected to Q8_0;
|
||||
# banded/streaming quantizer keeps RAM bounded)
|
||||
# 3. smoke: llama-completion (NOT llama-cli) with bounded -c and a timer
|
||||
#
|
||||
set -eo pipefail
|
||||
|
||||
SRC_DIR="${SRC_DIR:-/mnt/ssd2/models/qwen38-flash-next/meta-fp8}"
|
||||
WORK_DIR="${WORK_DIR:-/mnt/ssd2/models/qwen38-flash-next}"
|
||||
ENGINE="${ENGINE:-/home/user/source/ROCmFPX}"
|
||||
OUT_F16="${WORK_DIR}/Qwen3.8-Flash-Next-F16.gguf"
|
||||
OUT_QUANT="${WORK_DIR}/Qwen3.8-Flash-Next-ROCmFP4_FAST.gguf"
|
||||
SMOKE_CTX="${SMOKE_CTX:-8192}"
|
||||
|
||||
mkdir -p "${WORK_DIR}"
|
||||
|
||||
if [ ! -f "${SRC_DIR}/config.json" ]; then
|
||||
echo "ERROR: ${SRC_DIR}/config.json missing - download not complete?" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "=== [1/3] convert FP8 safetensors -> F16 GGUF ==="
|
||||
if [ ! -f "${OUT_F16}" ]; then
|
||||
cd "${ENGINE}"
|
||||
setsid nohup python3 convert_hf_to_gguf.py "${SRC_DIR}" \
|
||||
--outfile "${OUT_F16}" \
|
||||
--outtype f16
|
||||
else
|
||||
echo "F16 GGUF exists, skipping: ${OUT_F16}"
|
||||
fi
|
||||
|
||||
echo "=== [2/3] quantize -> Q4_0_ROCMFP4_FAST ==="
|
||||
if [ ! -f "${OUT_QUANT}" ]; then
|
||||
/usr/bin/time -v "${ENGINE}/build-strix-rocmfp4/bin/llama-quantize" \
|
||||
"${OUT_F16}" \
|
||||
"${OUT_QUANT}" \
|
||||
Q4_0_ROCMFP4_FAST
|
||||
else
|
||||
echo "quant exists, skipping: ${OUT_QUANT}"
|
||||
fi
|
||||
|
||||
echo "=== [3/3] smoke: llama-completion, bounded context, timer ==="
|
||||
/usr/bin/time -v timeout 900 "${ENGINE}/build-strix-rocmfp4/bin/llama-completion" \
|
||||
-m "${OUT_QUANT}" \
|
||||
-dev Vulkan0 -ngl 47 -c "${SMOKE_CTX}" -fa on -ub 2048 \
|
||||
-p "The capital of France is" -n 64 --temp 0 -no-cnv --simple-io 2>&1 | tail -40
|
||||
|
||||
ls -lah "${OUT_F16}" "${OUT_QUANT}"
|
||||
echo "DONE"
|
||||
Executable
+47
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env bash
|
||||
# Decode + prefill vs context depth, with and without MTP.
|
||||
#
|
||||
# Uses filler prompts of ~8k and ~32k tokens (results/filler/) and reports the
|
||||
# [ Prompt: X t/s | Generation: Y t/s ] line llama-cli prints. --reasoning off
|
||||
# keeps the 128-token generation budget from being eaten by a thinking block.
|
||||
# q8_0 KV keeps the cache small at depth.
|
||||
#
|
||||
# usage: scripts/depth-bench-strix-halo-vulkan.sh [depths] # e.g. "8k 32k"
|
||||
set -u
|
||||
|
||||
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||
FILLER=$OUT/filler
|
||||
SHORT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example."
|
||||
DEPTHS=${1:-0 8k 32k}
|
||||
|
||||
declare -A PROMPT_CTX=( [0]=8192 [8k]=16384 [32k]=40960 [128k]=139264 [256k]=257024 )
|
||||
|
||||
run() {
|
||||
local depth=$1 mode=$2
|
||||
local tag="depth$depth-$mode"
|
||||
local log=$OUT/${TAGPREFIX:-}shvd-$tag.log
|
||||
local args=()
|
||||
if [ "$depth" = "0" ]; then
|
||||
args+=(-p "$SHORT")
|
||||
else
|
||||
args+=(-f "$FILLER/filler-$depth.txt")
|
||||
fi
|
||||
[ "$mode" = "mtp" ] && args+=(-md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max 6 --spec-draft-p-min 0.75)
|
||||
|
||||
timeout 2400 "$BIN/llama-cli" -m "$TARGET" "${args[@]}" \
|
||||
-dev Vulkan0 -ngl 999 -c "${PROMPT_CTX[$depth]}" -fa on -ub 2048 \
|
||||
-ctk q8_0 -ctv q8_0 \
|
||||
-n 128 --temp 0 --reasoning off -no-cnv -st --simple-io > "$log" 2>&1
|
||||
printf '%-18s exit=%-3s %s %s\n' "$tag" "$?" \
|
||||
"$(grep -oE 'Prompt: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)"
|
||||
}
|
||||
|
||||
for d in $DEPTHS; do
|
||||
run "$d" plain
|
||||
run "$d" mtp
|
||||
done
|
||||
Executable
+133
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect a remote GGUF's metadata + tensor names without downloading the weights.
|
||||
|
||||
The interesting part of a GGUF (architecture, block_count, nextn_predict_layers,
|
||||
tensor names) all lives in the header, which is a few MB even for a 100 GB file.
|
||||
This range-fetches the first chunk and parses the header directly.
|
||||
|
||||
usage:
|
||||
scripts/gguf-header-peek.py <hf-repo-id> [filename-substring]
|
||||
scripts/gguf-header-peek.py /path/to/local.gguf
|
||||
|
||||
examples:
|
||||
scripts/gguf-header-peek.py EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF
|
||||
scripts/gguf-header-peek.py unsloth/Qwen3.8-Flash-Next-GGUF Q3_K_XL
|
||||
"""
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
CHUNK = 96 * 1024 * 1024 # header is dominated by the tokenizer strings
|
||||
|
||||
SCALAR_BYTES = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
|
||||
SCALAR_FMT = {0: "<B", 1: "<b", 2: "<H", 3: "<h", 4: "<I", 5: "<i",
|
||||
6: "<f", 7: "<B", 10: "<Q", 11: "<q", 12: "<d"}
|
||||
TYPE_NAMES = {0: "u8", 1: "i8", 2: "u16", 3: "i16", 4: "u32", 5: "i32", 6: "f32",
|
||||
7: "bool", 8: "str", 9: "arr", 10: "u64", 11: "i64", 12: "f64"}
|
||||
|
||||
|
||||
class Reader:
|
||||
def __init__(self, fh):
|
||||
self.fh = fh
|
||||
|
||||
def u32(self):
|
||||
return struct.unpack("<I", self.fh.read(4))[0]
|
||||
|
||||
def u64(self):
|
||||
return struct.unpack("<Q", self.fh.read(8))[0]
|
||||
|
||||
def string(self):
|
||||
return self.fh.read(self.u64()).decode("utf-8", "replace")
|
||||
|
||||
def value(self, vtype):
|
||||
if vtype == 8:
|
||||
return self.string()
|
||||
if vtype == 9:
|
||||
sub, count = self.u32(), self.u64()
|
||||
if count > 32:
|
||||
for _ in range(count):
|
||||
self.value(sub)
|
||||
return f"<{count} x {TYPE_NAMES.get(sub, sub)}>"
|
||||
return [self.value(sub) for _ in range(count)]
|
||||
if vtype in SCALAR_BYTES:
|
||||
raw = self.fh.read(SCALAR_BYTES[vtype])
|
||||
if len(raw) < SCALAR_BYTES[vtype]:
|
||||
raise ValueError("truncated header: increase CHUNK")
|
||||
if vtype == 7:
|
||||
return bool(raw[0])
|
||||
return struct.unpack(SCALAR_FMT[vtype], raw)[0]
|
||||
raise ValueError(f"unknown gguf value type {vtype}")
|
||||
|
||||
|
||||
def parse(path):
|
||||
with open(path, "rb") as fh:
|
||||
r = Reader(fh)
|
||||
assert fh.read(4) == b"GGUF", "not a GGUF file"
|
||||
version, n_tensors, n_kv = r.u32(), r.u64(), r.u64()
|
||||
kv = {r.string(): r.value(r.u32()) for _ in range(n_kv)}
|
||||
names = []
|
||||
for _ in range(n_tensors):
|
||||
names.append(r.string())
|
||||
r.fh.read(8 * r.u32()) # dims
|
||||
r.fh.read(12) # type + offset
|
||||
return version, kv, names
|
||||
|
||||
|
||||
def fetch(repo, want):
|
||||
listing = subprocess.run(
|
||||
["curl", "-s", f"https://huggingface.co/api/models/{repo}"],
|
||||
capture_output=True, text=True, check=True).stdout
|
||||
import json
|
||||
files = [s["rfilename"] for s in json.loads(listing)["siblings"]
|
||||
if s["rfilename"].endswith(".gguf")]
|
||||
matches = [f for f in files if want in f] if want else files
|
||||
if not matches:
|
||||
raise SystemExit(f"no .gguf matching {want!r} in {repo}")
|
||||
name = sorted(matches)[0]
|
||||
print(f"repo: {repo}\nfile: {name} (of {len(files)} gguf files)")
|
||||
url = f"https://huggingface.co/{repo}/resolve/main/{name}"
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".gguf", delete=False)
|
||||
subprocess.run(["curl", "-sL", "-r", f"0-{CHUNK}", "-o", tmp.name, url], check=True)
|
||||
return tmp.name
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
raise SystemExit(__doc__)
|
||||
target = sys.argv[1]
|
||||
if target.startswith(("http", "/")) and target.endswith(".gguf"):
|
||||
path = target if target.startswith("/") else fetch(target, "")
|
||||
else:
|
||||
path = fetch(target, sys.argv[2] if len(sys.argv) > 2 else "")
|
||||
|
||||
version, kv, names = parse(path)
|
||||
arch = kv.get("general.architecture", "?")
|
||||
print(f"gguf v{version} arch={arch} tensors={len(names)} kv={len(kv)}")
|
||||
print("\n-- key metadata --")
|
||||
for k in sorted(kv):
|
||||
if k.startswith("tokenizer.") or k.startswith("general."):
|
||||
continue
|
||||
v = kv[k]
|
||||
if isinstance(v, list) and len(v) > 12:
|
||||
v = f"{v[:12]} ... (len {len(v)})"
|
||||
print(f" {k:<46} {v}")
|
||||
|
||||
blocks = sorted({n.split(".")[1] for n in names
|
||||
if n.startswith("blk.") and n.split(".")[1].isdigit()})
|
||||
print(f"\n-- block indices: {blocks[:12]}{' ...' if len(blocks) > 12 else ''} "
|
||||
f"({len(blocks)} total)")
|
||||
suffixes = sorted({n.split(".", 2)[2] for n in names
|
||||
if n.startswith("blk.") and len(n.split(".")) > 2})
|
||||
print(f"-- per-block suffixes ({len(suffixes)}):")
|
||||
for s in suffixes:
|
||||
print(f" {s}")
|
||||
other = sorted(n for n in names if not n.startswith("blk."))
|
||||
if other:
|
||||
print("-- non-block tensors:")
|
||||
for n in other:
|
||||
print(f" {n}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+65
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
# A/B speculative decoding with MTP on the Nathanw1014 strix-halo-vulkan build.
|
||||
#
|
||||
# Runs the same greedy prompt with and without the MTP draft and diffs the text.
|
||||
# At temp 0 the two must be identical (a "greedy identity oracle"): any
|
||||
# divergence means the draft path is corrupting state.
|
||||
#
|
||||
# Uses the ORIGINAL sidecar (block_count = 49, blk.48), not the renumbered
|
||||
# -blk0 one -- the runtime selects the trailing block itself.
|
||||
# See results/2026-08-29-post-reboot-validation.md §6b.
|
||||
#
|
||||
# Tool notes: llama-cli, not llama-completion (the latter's parser rejects -md);
|
||||
# -st is required or cli sits in an interactive loop printing "> " forever.
|
||||
#
|
||||
# usage: scripts/mtp-test-strix-halo-vulkan.sh [plain|mtp|both]
|
||||
# NMAX=2,4,6 scripts/mtp-test-strix-halo-vulkan.sh mtp # sweep depths
|
||||
set -u
|
||||
|
||||
BIN=/home/user/source/llama.cpp-strix-halo-vulkan/build/bin
|
||||
TARGET=${TARGET:-/mnt/ssd2/models/qwen38-flash-next/Qwen3.8-Flash-Next-IQ4_XS.gguf}
|
||||
DRAFT=${DRAFT:-/mnt/ssd2/models/qwen38-flash-next/mtp-Qwen3.8-Flash-Next-Q8_0.gguf}
|
||||
OUT=${OUT:-/home/user/source/haloq38flash/results}
|
||||
# long-form prompt: the model emits EOS early on short factual ones, which makes
|
||||
# the tok/s figure meaningless
|
||||
PROMPT="Write a Python function that computes the nth Fibonacci number using memoization, with a docstring, type hints, and a short example. Then explain how the memoization cache works."
|
||||
CTX=8192
|
||||
NPRED=512
|
||||
NMAX_LIST=${NMAX:-6}
|
||||
|
||||
run() {
|
||||
local tag=$1; shift
|
||||
local log=$OUT/${TAGPREFIX:-}shv-$tag.log
|
||||
/usr/bin/time -v timeout 1800 "$BIN/llama-cli" \
|
||||
-m "$TARGET" "$@" \
|
||||
-dev Vulkan0 -ngl 999 -c "$CTX" -fa on -ub 2048 \
|
||||
-p "$PROMPT" -n "$NPRED" --temp 0 -no-cnv -st --simple-io > "$log" 2>&1
|
||||
local rc=$?
|
||||
# generated text sits between the "> " echo and the timing line
|
||||
awk '/^> /{f=1} /^\[ Prompt:/{f=0} f' "$log" > "$OUT/${TAGPREFIX:-}shv-$tag.txt"
|
||||
printf '%-16s exit=%s %s (%s bytes of output)\n' "$tag" "$rc" \
|
||||
"$(grep -oE 'Generation: [0-9.]+ t/s' "$log" | tail -1)" \
|
||||
"$(wc -c < "$OUT/shv-$tag.txt")"
|
||||
}
|
||||
|
||||
case ${1:-both} in
|
||||
plain) run plain ;;
|
||||
mtp)
|
||||
for n in ${NMAX_LIST//,/ }; do
|
||||
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||
done ;;
|
||||
both)
|
||||
run plain
|
||||
for n in ${NMAX_LIST//,/ }; do
|
||||
run mtp-n$n -md "$DRAFT" --spec-type draft-mtp \
|
||||
--spec-draft-n-max "$n" --spec-draft-p-min 0.75
|
||||
if diff -q "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" > /dev/null; then
|
||||
echo " greedy identity n=$n: PASS"
|
||||
else
|
||||
echo " greedy identity n=$n: FAIL"
|
||||
diff "$OUT/shv-plain.txt" "$OUT/shv-mtp-n$n.txt" | head -10
|
||||
fi
|
||||
done ;;
|
||||
*) echo "usage: $0 [plain|mtp|both]" ; exit 1 ;;
|
||||
esac
|
||||
Reference in New Issue
Block a user