diff --git a/app/serve.cpp b/app/serve.cpp index e99d22a1..cdca8bbb 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -731,6 +731,9 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) { // default draft length is 3, and it clamps a longer one to the drafter's trained block else if (!o.dflash.empty()) argv.insert(argv.end(), {"--spec-draft-n-max", "16"}); if (!o.mtp_p_min.empty()) { argv.push_back("--spec-draft-p-min"); argv.push_back(o.mtp_p_min); } + // a DFlash block is kept whole unless --mtp-p-min says otherwise: upstream's default p-min + // is 0, the ROCm tree's is 0.75, which cuts DFlash2 blocks from 6.7 to 5.4 tokens a step + else if (!o.dflash.empty()) argv.insert(argv.end(), {"--spec-draft-p-min", "0"}); } if (!o.mmproj.empty()) { if (device == "zinc" || device == "mlx" || device == "ds4") throw std::runtime_error("--mmproj works on the llama.cpp devices (vulkan, hrx, rocm)"); diff --git a/docs/lean.md b/docs/lean.md index 31df9e6f..8111c27b 100644 --- a/docs/lean.md +++ b/docs/lean.md @@ -159,10 +159,35 @@ each state column's rows in registers and stages the per-token inputs in shared tiled transpose for the concatenation that feeds the delta-net convolution. Both pass `test-backend-ops` against the CPU (GATED_DELTA_NET 42/42, CONCAT 129/129). -**Decode.** This route prefills; it does not decode fast. The ROCm build decodes Qwen3.8-27B at -about 11 tok/s without drafting, and its MTP drafting slowed prompt processing badly in our -runs. For decode-heavy work use `--device vulkan --dflash` ([serve.md](serve.md)): 45.7 tok/s on -code. +**One server: W4A4 prompts and DFlash2 decode.** Without a drafter this route decodes +Qwen3.8-27B at about 13 tok/s. With the DFlash2 drafter ([serve.md](serve.md#dflash-draft-models---dflash)) +the same server decodes three times faster and keeps most of the prompt speed: + +```sh +1bit serve -m Qwen3.8-27B-Q4_0-H32.gguf --dflash Qwen3.8-27B-DFlash2-q8_0.gguf +``` + +The ROCmFPX pin carries upstream's DFlash2 support (ggml-org/llama.cpp#27816, ported in +ROCmFPX#2), a HIP top-k that keeps the drafter's 248k-vocabulary candidate pick on the GPU (it +fell back to the CPU before: 11 tok/s instead of 28), and bounded recurrent-state rollback for +DFlash, so a partly accepted block rewinds the delta-net state instead of restoring a checkpoint +and replaying. `serve` passes `--spec-draft-p-min 0` with `--dflash`: this tree's default of 0.75 +cut DFlash2 blocks from 6.7 to 5.4 tokens a step. + +Measured through `1bit serve` on Strix Halo, 2026-09-28, quiet box: the 1,838-token prompt (best +/ median of 5, as `serve` reports it, context checkpoints included) and 256 tokens of greedy +decode on the code / prose / short prompts of [serve.md](serve.md) (best of 3): + +| Qwen3.8-27B-Q4_0-H32, lean ROCm route | Prompt t/s | Decode tok/s | +|---|---|---| +| no drafter | 502 / 500 | 13.0 / 13.0 / 13.4 | +| `--dflash` (DFlash2 Q8_0) | 445 / 439 | **40.9** / **26.8** / 13.7 | + +Mean accepted block: 6.71 tokens on code, 4.11 on prose, the same as upstream on this drafter. +Greedy output matches the no-drafter run on the code and short prompts; on prose one near-tie +word flips after 470 characters (batched verification rounds differently). The drafter costs the +prompt about 0.5 s on this prompt: its encoder reads the model's hidden states for every prompt +token (0.13 s), and the context checkpoints grow by the drafter's sliding-window cache. ## Build diff --git a/docs/serve.md b/docs/serve.md index 5e908064..bd4b6f7a 100644 --- a/docs/serve.md +++ b/docs/serve.md @@ -183,6 +183,11 @@ short answers (the translation prompt stops after ~16 tokens, too few to fill bl drafter is faster than BF16 (42.5 vs 38.9 on code, direct llama-server). A drafted token is kept only when the model agrees, so the output is the model's own. +`--dflash` also works on the lean ROCm route: a Hadamard-rotated Q4_0 file gets W4A4 prompt +processing and DFlash2 decode from one server, 445 t/s prompt and 40.9 / 26.8 tok/s decode +([lean.md](lean.md#hadamard-rotated-q4_0-w4a4-prompt-processing)). `--dflash` sets +`--spec-draft-p-min 0` unless `--mtp-p-min` is given. + ## MoE models larger than memory (`--moe-slots`) `--moe-slots N` (with `--device vulkan`) keeps a MoE model's routed experts in the file and diff --git a/tests/hadamard_route.sh b/tests/hadamard_route.sh index cc9802f3..dbd19078 100755 --- a/tests/hadamard_route.sh +++ b/tests/hadamard_route.sh @@ -19,7 +19,9 @@ # - --device vulkan refuses it, # - --device auto sends it to the ROCm route (--device ROCm0) with GGML_Q4_0_HADAMARD=1 and # GGML_W4A4_TENSORS=all in the backend's environment, -# - an unstamped file with --device auto still goes to Vulkan with neither variable. +# - an unstamped file with --device auto still goes to Vulkan with neither variable, +# - --dflash on a rotated file adds the DFlash drafter with full blocks (n-max 16, p-min 0): +# one ROCm server, W4A4 prompt processing and DFlash2 decode. # # usage: tests/hadamard_route.sh path/to/1bit set -uo pipefail @@ -42,6 +44,7 @@ def gguf(path, stamp): open(path, "wb").write(b"GGUF" + struct.pack("&1) check "--device vulkan refuses a rotated file" '[[ "$refused" == *"Hadamard-rotated"* ]]' -run() { # : serve with --device auto until the backend has recorded its start +run() { # [serve args]: serve with --device auto until the backend has recorded its start local port port=$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])') env -u GGML_Q4_0_HADAMARD -u GGML_W4A4_TENSORS RECORD="$2" "$bin" serve -m "$1" --device auto --port "$port" \ - --llama-server "$scratch/backend.py" >"$2.log" 2>&1 & + --llama-server "$scratch/backend.py" "${@:3}" >"$2.log" 2>&1 & pid=$! for _ in $(seq 1 100); do [ -s "$2" ] && break; sleep 0.1; done kill -9 "$pid" 2>/dev/null; wait "$pid" 2>/dev/null @@ -86,5 +89,12 @@ run "$scratch/plain.gguf" "$scratch/plain.json" check "an unstamped file still goes to Vulkan0" '[ "$(field "$scratch/plain.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = Vulkan0 ]' check " with neither variable" '[ "$(field "$scratch/plain.json" "sorted(r[\"env\"])")" = "[]" ]' +run "$scratch/h32.gguf" "$scratch/df.json" --dflash "$scratch/draft.gguf" +after() { field "$scratch/df.json" "r[\"argv\"][r[\"argv\"].index(\"$1\")+1]"; } +check "--dflash on a rotated file stays on ROCm0" '[ "$(after --device)" = ROCm0 ]' +check " with the DFlash drafter" '[ "$(after --spec-type)" = draft-dflash ] && [ "$(after -md)" = "$scratch/draft.gguf" ]' +check " full blocks: n-max 16, p-min 0" '[ "$(after --spec-draft-n-max)" = 16 ] && [ "$(after --spec-draft-p-min)" = 0 ]' +check " and the Hadamard W4A4 environment" '[ "$(field "$scratch/df.json" "r[\"env\"].get(\"GGML_W4A4_TENSORS\")")" = all ]' + if [ $fail -ne 0 ]; then for f in "$scratch"/*.log; do echo "--- $f"; cat "$f"; done; echo FAIL; exit 1; fi echo PASS diff --git a/third_party/llama.cpp-rocmfpx b/third_party/llama.cpp-rocmfpx index d572668a..8abd563d 160000 --- a/third_party/llama.cpp-rocmfpx +++ b/third_party/llama.cpp-rocmfpx @@ -1 +1 @@ -Subproject commit d572668a633eeef53d686cfca828eeeed4c00fb0 +Subproject commit 8abd563db5a3666adbd6eaa50c6826dfbe9abfef