From 54729f515dc7ee400a141e25ba1ffaca5317cfdc Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Thu, 1 Oct 2026 12:51:20 -0300 Subject: [PATCH] Pin ROCmFPX fc664ba: Hadamard rotation per tensor, any drafter type beside a rotated file ROCmFPX#7 makes the Hadamard activation rotation per tensor: the loader flags the Q4_0 weights of a file stamped onebit.hadamard_q4_0 = 32 and only those get rotated activations. A plain Q4_0 DFlash2 drafter beside Qwen3.8-27B-H32 now accepts 216/263 drafts on code at 45.6 tok/s (Q8_0: 217/260, 44.4); with the old process-wide GGML_Q4_0_HADAMARD it accepted 0/337. serve no longer sets GGML_Q4_0_HADAMARD (the backend reads the stamp; a value in the environment still forces the old process-wide rotation), and the #259 guard that refused a Q4_0 drafter goes, with gguf_tensor_type_count. hadamard_route.sh: no process-wide variable, Q4_0 and Q8_0 drafters both accepted; long_route.sh follows. docs/lean.md, serve.md and tools/hadamard_q4_0.py describe the per-tensor rotation. Co-Authored-By: Claude Opus 5.5 --- app/gguf_meta.h | 52 ----------------------------------- app/serve.cpp | 17 ++++-------- docs/lean.md | 6 ++-- docs/serve.md | 7 +++-- tests/hadamard_route.sh | 18 ++++++------ tests/long_route.sh | 2 +- third_party/llama.cpp-rocmfpx | 2 +- tools/hadamard_q4_0.py | 6 ++-- 8 files changed, 26 insertions(+), 84 deletions(-) diff --git a/app/gguf_meta.h b/app/gguf_meta.h index a47ad44b..ce3a2e7d 100644 --- a/app/gguf_meta.h +++ b/app/gguf_meta.h @@ -129,56 +129,4 @@ inline long long gguf_int(const std::string& path, const std::string& key) { return -1; } -// How many tensors of ggml type `type` (GGML_TYPE_Q4_0 = 2, ...) a GGUF holds, or -1 when the file -// is not a GGUF or its header cannot be read. Reads the key/value section and the tensor infos only. -inline long long gguf_tensor_type_count(const std::string& path, uint32_t type) { - std::ifstream f(path, std::ios::binary); - auto rd = [&](void* p, size_t n) { return static_cast(f.read(static_cast(p), n)); }; - uint32_t magic = 0, version = 0; - uint64_t n_tensors = 0, n_kv = 0; - if (!rd(&magic, 4) || magic != 0x46554747u || !rd(&version, 4) || version < 2 || !rd(&n_tensors, 8) || !rd(&n_kv, 8)) - return -1; - auto skip_str = [&]() { - uint64_t n = 0; - if (!rd(&n, 8) || n > (1u << 20)) return false; - return static_cast(f.seekg(static_cast(n), std::ios::cur)); - }; - static const int size[] = {1, 1, 2, 2, 4, 4, 4, 1, 0, 0, 8, 8, 8}; - auto scalar = [](uint32_t t) -> int { return t < 13 ? size[t] : -1; }; - for (uint64_t i = 0; i < n_kv; ++i) { - uint32_t t = 0; - if (!skip_str() || !rd(&t, 4)) return -1; - if (t == 8) { - if (!skip_str()) return -1; - } else if (t == 9) { - uint32_t et = 0; - uint64_t n = 0; - if (!rd(&et, 4) || !rd(&n, 8)) return -1; - if (et == 8) { - for (uint64_t j = 0; j < n; ++j) - if (!skip_str()) return -1; - } else if (scalar(et) > 0) { - f.seekg(static_cast(n * scalar(et)), std::ios::cur); - } else { - return -1; - } - } else if (scalar(t) > 0) { - f.seekg(scalar(t), std::ios::cur); - } else { - return -1; - } - if (!f) return -1; - } - long long count = 0; - for (uint64_t i = 0; i < n_tensors; ++i) { - uint32_t n_dims = 0, t = 0; - uint64_t offset = 0; - if (!skip_str() || !rd(&n_dims, 4) || n_dims > 8) return -1; - f.seekg(static_cast(8 * n_dims), std::ios::cur); - if (!rd(&t, 4) || !rd(&offset, 8)) return -1; - count += t == type; - } - return count; -} - } // namespace onebit diff --git a/app/serve.cpp b/app/serve.cpp index 90b45ab4..f5ad8877 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -674,19 +674,12 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) { "--device", device == "rocm" ? "ROCm0" : "Vulkan0", "-ngl", "99", "--jinja"}; if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } if (hadamard_q4_0(o.model)) { - // tools/hadamard_q4_0.py stamped it: its Q4_0 weights are rotated, so the activation - // quantizers must rotate too, and every Q4_0 matmul takes the W4A4 kernel (docs/lean.md). - // A value set in the environment wins (GGML_W4A4_TENSORS= keeps exact int8). + // tools/hadamard_q4_0.py stamped it: its Q4_0 weights are rotated. The lean ROCm build sees + // the stamp and rotates the activations of those weights, and of no others, so a plain + // Q4_0 drafter or MTP head works beside it (ROCmFPX#7); every Q4_0 matmul takes the W4A4 + // kernel (docs/lean.md). A value set in the environment wins (GGML_W4A4_TENSORS= keeps + // exact int8). if (device != "rocm") throw std::runtime_error(o.model + " is Hadamard-rotated: it runs on --device rocm only"); - if (!std::getenv("GGML_Q4_0_HADAMARD")) env.push_back("GGML_Q4_0_HADAMARD=1"); - // the rotation applies to every Q4_0 matmul in the backend, the drafter's too: a drafter - // or MTP head with unrotated Q4_0 tensors would read its activations rotated and accept - // nothing (a Q4_0 DFlash2 drafter did exactly that) - for (const std::string& draft : {o.dflash, o.mtp}) { - if (!draft.empty() && gguf_tensor_type_count(draft, 2) > 0 && !hadamard_q4_0(draft)) - throw std::runtime_error(draft + " has Q4_0 tensors that are not Hadamard-rotated, and " + o.model + - " rotates every Q4_0 activation in its backend: use a Q8_0 or Q4_K drafter"); - } if (!std::getenv("GGML_W4A4_TENSORS")) env.push_back("GGML_W4A4_TENSORS=all"); // micro-batch size: recipe rotated-moe-ub1024 (config/recipes.json) } diff --git a/docs/lean.md b/docs/lean.md index 3b24ebbe..2a7d130a 100644 --- a/docs/lean.md +++ b/docs/lean.md @@ -134,8 +134,10 @@ on Strix Halo, 15,182 MiB (4.66 bits per weight), 456 rotated tensors. That file **Serve it:** `1bit serve -m Qwen3.8-27B-Q4_0-H32.gguf` in a build with `-DONEBIT_LEAN=ON -DONEBIT_LEAN_ROCM=ON`. `serve` reads the stamp, runs the file on the lean ROCm build (`--device auto` picks it; any other device is refused, because only this build rotates the -activations to match), and sets `GGML_Q4_0_HADAMARD=1` and `GGML_W4A4_TENSORS=all` for it. Set -`GGML_W4A4_TENSORS=` (empty) to run the rotated file on the exact int8 path instead. +activations to match), and sets `GGML_W4A4_TENSORS=all` for it. That build reads the stamp itself +and rotates the activations of the file's Q4_0 weights only, so a plain Q4_0 drafter or MTP head +works beside it (ROCmFPX#7; `GGML_Q4_0_HADAMARD=1` still rotates every Q4_0 matmul in the +process). Set `GGML_W4A4_TENSORS=` (empty) to run the rotated file on the exact int8 path instead. `tests/hadamard_route.sh` (ctest `hadamard_route`) checks the routing and the environment without a GPU. diff --git a/docs/serve.md b/docs/serve.md index 09b7b7e3..48aa99cb 100644 --- a/docs/serve.md +++ b/docs/serve.md @@ -187,9 +187,10 @@ kept only when the model agrees, so the output is the model's own. processing and DFlash2 decode from one server, 518 t/s prompt and 46.0 / 28.4 / 17.6 tok/s decode ([lean.md](lean.md#hadamard-rotated-q4_0-w4a4-prompt-processing)). `--dflash` gets `--spec-draft-p-min` from a [recipe](recipes.md) (0.4 on rocm, 0 elsewhere) unless `--mtp-p-min` is given, and without `--mtp-max` drafts the -drafter's block minus one (its `dflash.block_size`; 16 when the file does not say). Next to a -Hadamard-rotated file a drafter must not hold plain Q4_0 tensors: the rotation applies to every Q4_0 -matmul in the backend, so serve refuses one (a Q8_0 or Q4_K drafter works). +drafter's block minus one (its `dflash.block_size`; 16 when the file does not say). Any drafter +type works next to a Hadamard-rotated file, plain Q4_0 included: the backend rotates the activations +of the rotated file's weights only. Qwen3.8-27B-H32 with a Q4_0 DFlash2 drafter runs as fast as with +the Q8_0 one. ## Recipes (`--recipes`, `--no-recipes`) diff --git a/tests/hadamard_route.sh b/tests/hadamard_route.sh index 3f579d37..59ef8de0 100755 --- a/tests/hadamard_route.sh +++ b/tests/hadamard_route.sh @@ -17,15 +17,15 @@ # How `1bit serve` routes a Hadamard-rotated Q4_0 file (tools/hadamard_q4_0.py stamps # onebit.hadamard_q4_0 = 32; docs/lean.md), without a GPU: # - --device vulkan refuses it, -# - --device auto sends it to the ROCm route (--device ROCm0) with GGML_Q4_0_HADAMARD=1 and -# GGML_W4A4_TENSORS=all in the backend's environment; a MoE file (an expert count) also gets +# - --device auto sends it to the ROCm route (--device ROCm0) with GGML_W4A4_TENSORS=all in the +# backend's environment and no process-wide GGML_Q4_0_HADAMARD (the backend rotates the stamped +# file's weights itself); a MoE file (an expert count) also gets # 1024-token micro-batches (-ub 1024), a dense one keeps llama-server's 512, # - an unstamped file with --device auto still goes to Vulkan with neither variable, # - --dflash on a rotated file adds the DFlash drafter (p-min 0.4 from its rocm recipe, n-max = the # drafter's dflash.block_size - 1, or 16 when the file does not say): one ROCm server, W4A4 # prompt processing and DFlash2 decode, -# - a drafter with unrotated Q4_0 tensors is refused next to a rotated file (the rotation is -# process-wide), one with Q8_0 tensors is not. +# - a drafter with plain Q4_0 tensors is accepted next to a rotated file, like a Q8_0 one. # # usage: tests/hadamard_route.sh path/to/1bit set -uo pipefail @@ -94,7 +94,7 @@ field() { python3 -c 'import json,sys; r=json.load(open(sys.argv[1])); print(eva run "$scratch/h32.gguf" "$scratch/h32.json" check "--device auto sends a rotated file to ROCm0" '[ "$(field "$scratch/h32.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = ROCm0 ]' -check " with GGML_Q4_0_HADAMARD=1" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_Q4_0_HADAMARD\")")" = 1 ]' +check " without a process-wide GGML_Q4_0_HADAMARD" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_Q4_0_HADAMARD\")")" = None ]' check " and GGML_W4A4_TENSORS=all" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_W4A4_TENSORS\")")" = all ]' check " dense: llama-server's own micro-batch" '[ "$(field "$scratch/h32.json" "\"-ub\" in r[\"argv\"]")" = False ]' @@ -115,12 +115,10 @@ check " and the Hadamard W4A4 environment" '[ "$(field "$scratch/df.json" "r[\" run "$scratch/h32.gguf" "$scratch/df8.json" --dflash "$scratch/draft8.gguf" check "--dflash with a block-8 drafter drafts 7" '[ "$(field "$scratch/df8.json" "r[\"argv\"][r[\"argv\"].index(\"--spec-draft-n-max\")+1]")" = 7 ]' -port=$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])') -q4=$(timeout 20 "$bin" serve -m "$scratch/h32.gguf" --device auto --port "$port" --llama-server "$scratch/backend.py" \ - --dflash "$scratch/draftq4.gguf" 2>&1) -check "a Q4_0 drafter next to a rotated file is refused" '[[ "$q4" == *"not Hadamard-rotated"* ]]' +run "$scratch/h32.gguf" "$scratch/dfq4.json" --dflash "$scratch/draftq4.gguf" +check "a Q4_0 drafter next to a rotated file is accepted" '[ "$(field "$scratch/dfq4.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq4.gguf" ]' run "$scratch/h32.gguf" "$scratch/dfq8.json" --dflash "$scratch/draftq8.gguf" -check " a Q8_0 drafter is not" '[ "$(field "$scratch/dfq8.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq8.gguf" ]' +check " so is a Q8_0 one" '[ "$(field "$scratch/dfq8.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq8.gguf" ]' if [ $fail -ne 0 ]; then for f in "$scratch"/*.log; do echo "--- $f"; cat "$f"; done; echo FAIL; exit 1; fi echo PASS diff --git a/tests/long_route.sh b/tests/long_route.sh index 454cfb0b..3b91dbfa 100755 --- a/tests/long_route.sh +++ b/tests/long_route.sh @@ -107,7 +107,7 @@ check "a long completion goes to ROCm0" '[[ "$out" == "device:ROCm0"* ]]' out=$(ask /v1/chat/completions "$(chat user "hello there" assistant "hi" user "$(words 300)")") check "a conversation that grows long stays on Vulkan0" '[[ "$out" == "device:Vulkan0"*"short"* ]]' check "the ROCm backend has the Hadamard W4A4 environment" \ - 'python3 -c "import json,sys; e=json.load(open(sys.argv[1]))[\"env\"]; sys.exit(not (e.get(\"GGML_Q4_0_HADAMARD\")==\"1\" and e.get(\"GGML_W4A4_TENSORS\")==\"all\"))" "$scratch/rec/ROCm0.json"' + 'python3 -c "import json,sys; e=json.load(open(sys.argv[1]))[\"env\"]; sys.exit(not (e.get(\"GGML_Q4_0_HADAMARD\") is None and e.get(\"GGML_W4A4_TENSORS\")==\"all\"))" "$scratch/rec/ROCm0.json"' check "the Vulkan backend does not" \ 'python3 -c "import json,sys; sys.exit(bool(json.load(open(sys.argv[1]))[\"env\"]))" "$scratch/rec/Vulkan0.json"' diff --git a/third_party/llama.cpp-rocmfpx b/third_party/llama.cpp-rocmfpx index d348b83a..fc664baf 160000 --- a/third_party/llama.cpp-rocmfpx +++ b/third_party/llama.cpp-rocmfpx @@ -1 +1 @@ -Subproject commit d348b83ae8e6affe51fd5483284906be2f2c6556 +Subproject commit fc664baf1d1468a13618ff39d30eb2367bd648f2 diff --git a/tools/hadamard_q4_0.py b/tools/hadamard_q4_0.py index c2f27b8c..104462e4 100644 --- a/tools/hadamard_q4_0.py +++ b/tools/hadamard_q4_0.py @@ -23,9 +23,9 @@ the normalized 32-point Walsh-Hadamard matrix, per 32-element block, in a copy of SOURCE. The imatrix gets the matching change (each rotated column's importance becomes its block's mean). llama-quantize then writes Q4_0 for exactly those tensors and a non-Q4_0 type for every other -matmul weight, and stamps `onebit.hadamard_q4_0 = 32`. `1bit serve` sees the stamp, sets -GGML_Q4_0_HADAMARD=1 (the ROCm activation quantizers apply the same rotation) and -GGML_W4A4_TENSORS=all, and refuses the file on devices without the rotation. +matmul weight, and stamps `onebit.hadamard_q4_0 = 32`. The lean ROCm build sees the stamp and its +activation quantizers apply the same rotation to those weights (and no others); `1bit serve` sees +it, sets GGML_W4A4_TENSORS=all, and refuses the file on devices without the rotation. The rotation is exact: on the int8 path the rotated file matches an unrotated one quantized the same way (Qwen3.8-27B: KLD 0.031 vs 0.029 against BF16). What it buys is 4-bit activations