Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 0 additions & 52 deletions app/gguf_meta.h
Original file line number Diff line number Diff line change
Expand Up @@ -129,56 +129,4 @@ inline long long gguf_int(const std::string& path, const std::string& key) {
return -1;
}

// How many tensors of ggml type `type` (GGML_TYPE_Q4_0 = 2, ...) a GGUF holds, or -1 when the file
// is not a GGUF or its header cannot be read. Reads the key/value section and the tensor infos only.
inline long long gguf_tensor_type_count(const std::string& path, uint32_t type) {
std::ifstream f(path, std::ios::binary);
auto rd = [&](void* p, size_t n) { return static_cast<bool>(f.read(static_cast<char*>(p), n)); };
uint32_t magic = 0, version = 0;
uint64_t n_tensors = 0, n_kv = 0;
if (!rd(&magic, 4) || magic != 0x46554747u || !rd(&version, 4) || version < 2 || !rd(&n_tensors, 8) || !rd(&n_kv, 8))
return -1;
auto skip_str = [&]() {
uint64_t n = 0;
if (!rd(&n, 8) || n > (1u << 20)) return false;
return static_cast<bool>(f.seekg(static_cast<std::streamoff>(n), std::ios::cur));
};
static const int size[] = {1, 1, 2, 2, 4, 4, 4, 1, 0, 0, 8, 8, 8};
auto scalar = [](uint32_t t) -> int { return t < 13 ? size[t] : -1; };
for (uint64_t i = 0; i < n_kv; ++i) {
uint32_t t = 0;
if (!skip_str() || !rd(&t, 4)) return -1;
if (t == 8) {
if (!skip_str()) return -1;
} else if (t == 9) {
uint32_t et = 0;
uint64_t n = 0;
if (!rd(&et, 4) || !rd(&n, 8)) return -1;
if (et == 8) {
for (uint64_t j = 0; j < n; ++j)
if (!skip_str()) return -1;
} else if (scalar(et) > 0) {
f.seekg(static_cast<std::streamoff>(n * scalar(et)), std::ios::cur);
} else {
return -1;
}
} else if (scalar(t) > 0) {
f.seekg(scalar(t), std::ios::cur);
} else {
return -1;
}
if (!f) return -1;
}
long long count = 0;
for (uint64_t i = 0; i < n_tensors; ++i) {
uint32_t n_dims = 0, t = 0;
uint64_t offset = 0;
if (!skip_str() || !rd(&n_dims, 4) || n_dims > 8) return -1;
f.seekg(static_cast<std::streamoff>(8 * n_dims), std::ios::cur);
if (!rd(&t, 4) || !rd(&offset, 8)) return -1;
count += t == type;
}
return count;
}

} // namespace onebit
17 changes: 5 additions & 12 deletions app/serve.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -674,19 +674,12 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) {
"--device", device == "rocm" ? "ROCm0" : "Vulkan0", "-ngl", "99", "--jinja"};
if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); }
if (hadamard_q4_0(o.model)) {
// tools/hadamard_q4_0.py stamped it: its Q4_0 weights are rotated, so the activation
// quantizers must rotate too, and every Q4_0 matmul takes the W4A4 kernel (docs/lean.md).
// A value set in the environment wins (GGML_W4A4_TENSORS= keeps exact int8).
// tools/hadamard_q4_0.py stamped it: its Q4_0 weights are rotated. The lean ROCm build sees
// the stamp and rotates the activations of those weights, and of no others, so a plain
// Q4_0 drafter or MTP head works beside it (ROCmFPX#7); every Q4_0 matmul takes the W4A4
// kernel (docs/lean.md). A value set in the environment wins (GGML_W4A4_TENSORS= keeps
// exact int8).
if (device != "rocm") throw std::runtime_error(o.model + " is Hadamard-rotated: it runs on --device rocm only");
if (!std::getenv("GGML_Q4_0_HADAMARD")) env.push_back("GGML_Q4_0_HADAMARD=1");
// the rotation applies to every Q4_0 matmul in the backend, the drafter's too: a drafter
// or MTP head with unrotated Q4_0 tensors would read its activations rotated and accept
// nothing (a Q4_0 DFlash2 drafter did exactly that)
for (const std::string& draft : {o.dflash, o.mtp}) {
if (!draft.empty() && gguf_tensor_type_count(draft, 2) > 0 && !hadamard_q4_0(draft))
throw std::runtime_error(draft + " has Q4_0 tensors that are not Hadamard-rotated, and " + o.model +
" rotates every Q4_0 activation in its backend: use a Q8_0 or Q4_K drafter");
}
if (!std::getenv("GGML_W4A4_TENSORS")) env.push_back("GGML_W4A4_TENSORS=all");
// micro-batch size: recipe rotated-moe-ub1024 (config/recipes.json)
}
Expand Down
6 changes: 4 additions & 2 deletions docs/lean.md
Original file line number Diff line number Diff line change
Expand Up @@ -134,8 +134,10 @@ on Strix Halo, 15,182 MiB (4.66 bits per weight), 456 rotated tensors. That file
**Serve it:** `1bit serve -m Qwen3.8-27B-Q4_0-H32.gguf` in a build with `-DONEBIT_LEAN=ON
-DONEBIT_LEAN_ROCM=ON`. `serve` reads the stamp, runs the file on the lean ROCm build
(`--device auto` picks it; any other device is refused, because only this build rotates the
activations to match), and sets `GGML_Q4_0_HADAMARD=1` and `GGML_W4A4_TENSORS=all` for it. Set
`GGML_W4A4_TENSORS=` (empty) to run the rotated file on the exact int8 path instead.
activations to match), and sets `GGML_W4A4_TENSORS=all` for it. That build reads the stamp itself
and rotates the activations of the file's Q4_0 weights only, so a plain Q4_0 drafter or MTP head
works beside it (ROCmFPX#7; `GGML_Q4_0_HADAMARD=1` still rotates every Q4_0 matmul in the
process). Set `GGML_W4A4_TENSORS=` (empty) to run the rotated file on the exact int8 path instead.
`tests/hadamard_route.sh` (ctest `hadamard_route`) checks the routing and the environment
without a GPU.

Expand Down
7 changes: 4 additions & 3 deletions docs/serve.md
Original file line number Diff line number Diff line change
Expand Up @@ -187,9 +187,10 @@ kept only when the model agrees, so the output is the model's own.
processing and DFlash2 decode from one server, 518 t/s prompt and 46.0 / 28.4 / 17.6 tok/s decode
([lean.md](lean.md#hadamard-rotated-q4_0-w4a4-prompt-processing)). `--dflash` gets
`--spec-draft-p-min` from a [recipe](recipes.md) (0.4 on rocm, 0 elsewhere) unless `--mtp-p-min` is given, and without `--mtp-max` drafts the
drafter's block minus one (its `dflash.block_size`; 16 when the file does not say). Next to a
Hadamard-rotated file a drafter must not hold plain Q4_0 tensors: the rotation applies to every Q4_0
matmul in the backend, so serve refuses one (a Q8_0 or Q4_K drafter works).
drafter's block minus one (its `dflash.block_size`; 16 when the file does not say). Any drafter
type works next to a Hadamard-rotated file, plain Q4_0 included: the backend rotates the activations
of the rotated file's weights only. Qwen3.8-27B-H32 with a Q4_0 DFlash2 drafter runs as fast as with
the Q8_0 one.

## Recipes (`--recipes`, `--no-recipes`)

Expand Down
18 changes: 8 additions & 10 deletions tests/hadamard_route.sh
Original file line number Diff line number Diff line change
Expand Up @@ -17,15 +17,15 @@
# How `1bit serve` routes a Hadamard-rotated Q4_0 file (tools/hadamard_q4_0.py stamps
# onebit.hadamard_q4_0 = 32; docs/lean.md), without a GPU:
# - --device vulkan refuses it,
# - --device auto sends it to the ROCm route (--device ROCm0) with GGML_Q4_0_HADAMARD=1 and
# GGML_W4A4_TENSORS=all in the backend's environment; a MoE file (an expert count) also gets
# - --device auto sends it to the ROCm route (--device ROCm0) with GGML_W4A4_TENSORS=all in the
# backend's environment and no process-wide GGML_Q4_0_HADAMARD (the backend rotates the stamped
# file's weights itself); a MoE file (an expert count) also gets
# 1024-token micro-batches (-ub 1024), a dense one keeps llama-server's 512,
# - an unstamped file with --device auto still goes to Vulkan with neither variable,
# - --dflash on a rotated file adds the DFlash drafter (p-min 0.4 from its rocm recipe, n-max = the
# drafter's dflash.block_size - 1, or 16 when the file does not say): one ROCm server, W4A4
# prompt processing and DFlash2 decode,
# - a drafter with unrotated Q4_0 tensors is refused next to a rotated file (the rotation is
# process-wide), one with Q8_0 tensors is not.
# - a drafter with plain Q4_0 tensors is accepted next to a rotated file, like a Q8_0 one.
#
# usage: tests/hadamard_route.sh path/to/1bit
set -uo pipefail
Expand Down Expand Up @@ -94,7 +94,7 @@ field() { python3 -c 'import json,sys; r=json.load(open(sys.argv[1])); print(eva

run "$scratch/h32.gguf" "$scratch/h32.json"
check "--device auto sends a rotated file to ROCm0" '[ "$(field "$scratch/h32.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = ROCm0 ]'
check " with GGML_Q4_0_HADAMARD=1" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_Q4_0_HADAMARD\")")" = 1 ]'
check " without a process-wide GGML_Q4_0_HADAMARD" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_Q4_0_HADAMARD\")")" = None ]'
check " and GGML_W4A4_TENSORS=all" '[ "$(field "$scratch/h32.json" "r[\"env\"].get(\"GGML_W4A4_TENSORS\")")" = all ]'
check " dense: llama-server's own micro-batch" '[ "$(field "$scratch/h32.json" "\"-ub\" in r[\"argv\"]")" = False ]'

Expand All @@ -115,12 +115,10 @@ check " and the Hadamard W4A4 environment" '[ "$(field "$scratch/df.json" "r[\"
run "$scratch/h32.gguf" "$scratch/df8.json" --dflash "$scratch/draft8.gguf"
check "--dflash with a block-8 drafter drafts 7" '[ "$(field "$scratch/df8.json" "r[\"argv\"][r[\"argv\"].index(\"--spec-draft-n-max\")+1]")" = 7 ]'

port=$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])')
q4=$(timeout 20 "$bin" serve -m "$scratch/h32.gguf" --device auto --port "$port" --llama-server "$scratch/backend.py" \
--dflash "$scratch/draftq4.gguf" 2>&1)
check "a Q4_0 drafter next to a rotated file is refused" '[[ "$q4" == *"not Hadamard-rotated"* ]]'
run "$scratch/h32.gguf" "$scratch/dfq4.json" --dflash "$scratch/draftq4.gguf"
check "a Q4_0 drafter next to a rotated file is accepted" '[ "$(field "$scratch/dfq4.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq4.gguf" ]'
run "$scratch/h32.gguf" "$scratch/dfq8.json" --dflash "$scratch/draftq8.gguf"
check " a Q8_0 drafter is not" '[ "$(field "$scratch/dfq8.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq8.gguf" ]'
check " so is a Q8_0 one" '[ "$(field "$scratch/dfq8.json" "r[\"argv\"][r[\"argv\"].index(\"-md\")+1]")" = "$scratch/draftq8.gguf" ]'

if [ $fail -ne 0 ]; then for f in "$scratch"/*.log; do echo "--- $f"; cat "$f"; done; echo FAIL; exit 1; fi
echo PASS
2 changes: 1 addition & 1 deletion tests/long_route.sh
Original file line number Diff line number Diff line change
Expand Up @@ -107,7 +107,7 @@ check "a long completion goes to ROCm0" '[[ "$out" == "device:ROCm0"* ]]'
out=$(ask /v1/chat/completions "$(chat user "hello there" assistant "hi" user "$(words 300)")")
check "a conversation that grows long stays on Vulkan0" '[[ "$out" == "device:Vulkan0"*"short"* ]]'
check "the ROCm backend has the Hadamard W4A4 environment" \
'python3 -c "import json,sys; e=json.load(open(sys.argv[1]))[\"env\"]; sys.exit(not (e.get(\"GGML_Q4_0_HADAMARD\")==\"1\" and e.get(\"GGML_W4A4_TENSORS\")==\"all\"))" "$scratch/rec/ROCm0.json"'
'python3 -c "import json,sys; e=json.load(open(sys.argv[1]))[\"env\"]; sys.exit(not (e.get(\"GGML_Q4_0_HADAMARD\") is None and e.get(\"GGML_W4A4_TENSORS\")==\"all\"))" "$scratch/rec/ROCm0.json"'
check "the Vulkan backend does not" \
'python3 -c "import json,sys; sys.exit(bool(json.load(open(sys.argv[1]))[\"env\"]))" "$scratch/rec/Vulkan0.json"'

Expand Down
6 changes: 3 additions & 3 deletions tools/hadamard_q4_0.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,9 @@
the normalized 32-point Walsh-Hadamard matrix, per 32-element block, in a copy of SOURCE. The
imatrix gets the matching change (each rotated column's importance becomes its block's mean).
llama-quantize then writes Q4_0 for exactly those tensors and a non-Q4_0 type for every other
matmul weight, and stamps `onebit.hadamard_q4_0 = 32`. `1bit serve` sees the stamp, sets
GGML_Q4_0_HADAMARD=1 (the ROCm activation quantizers apply the same rotation) and
GGML_W4A4_TENSORS=all, and refuses the file on devices without the rotation.
matmul weight, and stamps `onebit.hadamard_q4_0 = 32`. The lean ROCm build sees the stamp and its
activation quantizers apply the same rotation to those weights (and no others); `1bit serve` sees
it, sets GGML_W4A4_TENSORS=all, and refuses the file on devices without the rotation.

The rotation is exact: on the int8 path the rotated file matches an unrotated one quantized the
same way (Qwen3.8-27B: KLD 0.031 vs 0.029 against BF16). What it buys is 4-bit activations
Expand Down
Loading