diff --git a/app/serve.cpp b/app/serve.cpp index 4398f87..5b1c7b4 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -229,8 +229,9 @@ bool prism_hadamard(const std::string& model) { gguf_int(model, "prism.hadamard.version") > 0; } -// PrismML's own ternary types (general.file_type PQ2_0 / PTQ1_0 in its fork): no pinned -// llama.cpp reads them; tools/ternary_to_q4_0.py writes the same weights as Q4_0. +// PrismML's own ternary types (general.file_type PQ2_0 / PTQ1_0 in its fork): only the HRX +// build reads them (llama.cpp #62); tools/ternary_to_q4_0.py writes the same weights as Q4_0 +// for any other route. bool prism_ternary_types(const std::string& model) { const long long ft = gguf_int(model, "general.file_type"); return ft == 128 || ft == 129 || ft == 141 || ft == 142 || ft == 143; @@ -1207,10 +1208,17 @@ int serve_child(const Options& given) { else std::fprintf(stderr, "1bit serve: %s is Hadamard-rotated: lean ROCm route, W4A4 prompt processing\n", o.model.c_str()); } + if (prism_ternary_types(o.model) && !prism_hadamard(o.model)) { + // PQ2_0 / PTQ1_0 without a rotation (e.g. Ternary-Bonsai-1.7B): only the HRX build reads the types + if (o.device == "auto") o.device = "hrx"; + if (o.device != "hrx") + throw std::runtime_error(o.model + " stores PrismML's ternary types: it runs on --device hrx, or convert it " + "with tools/ternary_to_q4_0.py (the same weights as Q4_0), docs/hrx.md"); + if (o.laya_auto || !o.laya_model.empty()) + throw std::runtime_error("--laya routes among devices; a file in PrismML's ternary types runs on hrx only"); + std::fprintf(stderr, "1bit serve: %s stores PrismML's ternary types: HRX route\n", o.model.c_str()); + } if (prism_hadamard(o.model)) { - if (prism_ternary_types(o.model)) - throw std::runtime_error(o.model + " stores PrismML's ternary types: convert it first with " - "tools/ternary_to_q4_0.py (the same weights as Q4_0), docs/hrx.md"); // a rotated file has one route: our llama.cpp on HRX, which rotates the activations if (o.device == "auto") o.device = "hrx"; if (o.device != "hrx") diff --git a/docs/hrx.md b/docs/hrx.md index 56159d3..72b917e 100644 --- a/docs/hrx.md +++ b/docs/hrx.md @@ -192,18 +192,32 @@ were observed. ### Ternary Bonsai (PrismML's Hadamard-folded GGUFs) PrismML's [Ternary-Bonsai-2-27B](https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf) is Qwen3.8-27B -trained to weights of -1, 0 and +1, with one fp16 scale per 128 weights. It runs on `HRX0`: +trained to weights of -1, 0 and +1, with one fp16 scale per 128 weights. It runs on `HRX0`, +straight from PrismML's file: ```sh -tools/ternary_to_q4_0.py Ternary-Bonsai-2-27B-PTQ1_0.gguf Ternary-Bonsai-2-27B-Q4_0.gguf -1bit serve -m Ternary-Bonsai-2-27B-Q4_0.gguf # --device auto picks hrx for this file +1bit serve -m Ternary-Bonsai-2-27B-PTQ1_0.gguf # --device auto picks hrx for this file ``` -- **The weights.** PrismML's PTQ1_0 and PQ2_0 types are not in any llama.cpp the engine pins. - `tools/ternary_to_q4_0.py` writes the same weights as Q4_0 (q = trit + 8 with the group's own - fp16 scale in each of its four blocks), so nothing is lost: every group is decoded back and - compared before the file is kept, and `tests/ternary_to_q4_0_test.py` checks it against - PrismML's own encoder. The file is 14.1 GiB instead of 5.5. +- **The weights** ([llama.cpp #62](https://github.com/1bit-MONSTER/llama.cpp/pull/62)). PrismML's + PQ2_0 (ggml type 142) and PTQ1_0 (143) are native types in our HRX build, with the layouts of + PrismML's llama.cpp fork. Our CPU decode of every tensor of both 27B files matches PrismML's own + build bit for bit. On `HRX0` the K-quant decode kernels read them directly, so one copy of the + weights is resident: + + | Ternary-Bonsai-2-27B | weights | GPU peak over idle | tg128 (tok/s) | pp512 (tok/s) | + |---|---|---|---|---| + | Q4_0 + packed ternary (below) | 14.13 GiB | 21.8 GiB | 15.3 | 13.1 | + | PQ2_0 | 6.70 GiB | 10.1 GiB | 15.8 | 14.2 | + | PTQ1_0 | 5.53 GiB | 8.9 GiB | 14.3 | 14.2 | + + Balanced power mode, llama-bench `-fa 1`, medians of 3. Against the CPU logits of the + bit-identical Q4_0 copy (3 x 512) all three give the same KLD, 0.000129 at decode (`-ub 8`) and + 0.002051 at `-ub 512`. Files in these types run on `--device hrx` only, with or without + `prism.hadamard` keys (Ternary-Bonsai-1.7B has none). For another route, + `tools/ternary_to_q4_0.py` writes the same weights as Q4_0 (q = trit + 8 with the group's own fp16 + scale in each of its four blocks); every group is decoded back and compared before the file is + kept, and `tests/ternary_to_q4_0_test.py` checks it against PrismML's own encoder. - **Packed ternary decode** ([llama.cpp #54](https://github.com/1bit-MONSTER/llama.cpp/pull/54)). For a file stamped `onebit.ternary_q4_0`, `1bit serve` sets `GGML_HRX_TERNARY_Q4_0=1`. HRX0's K-quant decode kernels then read those Q4_0 weights from a 2-bit copy made at load (68 bytes per @@ -285,6 +299,18 @@ run on HRX with `1bit serve --device hrx`. down its generic path: #55 routes Q4_K, Q5_K and IQ4_XS only. Until Q4_0 gets the same routing, `--device auto` keeps these files on the lean ROCm route. +### Server memory: the graph program cache cap (llama.cpp #63) + +HRX builds a graph program for every new graph shape (each new prompt or batch length) and kept every +one: a long-running `llama-server` with prompts of varying length grew its GPU memory without bound +(ZAYA1-8B: about 1 GiB per request, 24.6 GiB after 28 requests). Since +[llama.cpp #63](https://github.com/1bit-MONSTER/llama.cpp/pull/63) the cache holds at most +`GGML_HRX_GRAPH_PROGRAM_CACHE` programs (default 64, `0` = unbounded): when a new program takes it over +the limit, HRX waits for the stream and drops every other program. On ZAYA1-8B, 40 requests of varying +length peak at 8.7 GiB instead of growing; greedy answers are identical, and llama-bench pp512/tg128 +is unchanged. A repeated prompt length right after a flush costs 1.75 s instead of 1.12 s, since its +program is rebuilt. + ### Known issues on `HRX0` - **Several sequences per batch fail.** `llama-perplexity` with `n_seq` > 1 stops on an @@ -385,6 +411,11 @@ BF16 is 0.063. UD files no longer fall back to the generic path for pairs that mix formats. Qwen3.8-27B UD-IQ2_S tg128 3.12 to 8.32 tok/s; UD-Q4_K_XL unchanged (11.88 / 11.81); KLD 0.00013 vs the previous build, `test-backend-ops -b HRX0` 996/996. +- **PrismML's PQ2_0 / PTQ1_0 types** ([llama.cpp #62](https://github.com/1bit-MONSTER/llama.cpp/pull/62)): + native ggml types 142 / 143 with a CPU reference and HRX decode on the K-quant kernels; Q1_0 decode + moves onto them too ([Ternary Bonsai](#ternary-bonsai-prismmls-hadamard-folded-ggufs)). +- **A cap on the graph program cache** ([llama.cpp #63](https://github.com/1bit-MONSTER/llama.cpp/pull/63)): + `GGML_HRX_GRAPH_PROGRAM_CACHE`, default 64 ([section above](#server-memory-the-graph-program-cache-cap-llamacpp-63)). - **Hadamard-rotated Q4_0 files** ([llama.cpp #58](https://github.com/1bit-MONSTER/llama.cpp/pull/58)): llama loads the engine's `onebit.hadamard_q4_0` files through `llama-hadamard`. Qwen3.8-27B-Q4_0-H32 on HRX vs BF16 (wikitext 40 x 512): PPL 6.021, KLD 0.0292, same top token 92.5%. Q4_0 prompt matmuls diff --git a/registry/architectures.json b/registry/architectures.json index c45bf43..f1e3277 100644 --- a/registry/architectures.json +++ b/registry/architectures.json @@ -2,7 +2,7 @@ "about": "HF architecture -> GGUF architecture and the backends whose code accepts it. Generated by tools/registry_build.py from the pinned sources; do not edit.", "sources": { "llama.cpp (vulkan)": "8a5e8d8ae71394b27a3a0aa1810c2b4c0c8f3775", - "llama.cpp (hrx)": "bd5b2970b05ad0632a7be0d4cd39fc761dc793e6", + "llama.cpp (hrx)": "d60cc4f3ca53d51f05252a802e3ec6188b4dcc94", "zinc": "29bc350ac4cbf9f110ec628b9e177ea04ac816fb" }, "counts": { diff --git a/tests/prism_route.sh b/tests/prism_route.sh index bd2ea04..de9275f 100755 --- a/tests/prism_route.sh +++ b/tests/prism_route.sh @@ -18,7 +18,8 @@ # docs/hrx.md), without a GPU: # - --device auto sends a converted file (tools/ternary_to_q4_0.py) to HRX0, # - --device vulkan refuses it (upstream llama.cpp would ignore the rotation), -# - a file still in PrismML's ternary types is refused with the converter's name, +# - a file in PrismML's ternary types goes to HRX0 too (llama.cpp #62), with or without +# prism.hadamard keys, and --device vulkan refuses it with the converter's name, # - a plain file with --device auto still goes to the GPU route (HRX0 with HRX, else Vulkan0). # # usage: tests/prism_route.sh path/to/1bit @@ -50,6 +51,7 @@ def gguf(path, prism, file_type): open(path, "wb").write(b"GGUF" + struct.pack("&1) check "--device vulkan refuses a Hadamard-folded file" '[[ "$refused" == *"Hadamard-folded"* ]]' -refused=$("$bin" serve -m "$scratch/bonsai-ptq1_0.gguf" --device auto --port 1 --llama-server "$scratch/backend.py" 2>&1) -check "PrismML ternary types are refused with the converter's name" '[[ "$refused" == *"ternary_to_q4_0.py"* ]]' +refused=$("$bin" serve -m "$scratch/bonsai-pq2_0-plain.gguf" --device vulkan --port 1 --llama-server "$scratch/backend.py" 2>&1) +check "--device vulkan refuses PrismML ternary types with the converter's name" '[[ "$refused" == *"ternary_to_q4_0.py"* ]]' run() { # : serve with --device auto until the backend has recorded its start local port @@ -87,6 +89,10 @@ field() { python3 -c 'import json,sys; r=json.load(open(sys.argv[1])); print(eva run "$scratch/bonsai-q4_0.gguf" "$scratch/b.json" check "--device auto sends a Hadamard-folded file to HRX0" '[ "$(field "$scratch/b.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = HRX0 ]' +run "$scratch/bonsai-ptq1_0.gguf" "$scratch/t.json" +check "--device auto sends a PTQ1_0 file to HRX0" '[ "$(field "$scratch/t.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = HRX0 ]' +run "$scratch/bonsai-pq2_0-plain.gguf" "$scratch/p.json" +check "--device auto sends an unrotated PQ2_0 file to HRX0" '[ "$(field "$scratch/p.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = HRX0 ]' run "$scratch/plain.gguf" "$scratch/plain.json" check "a plain file still goes to $gpu" '[ "$(field "$scratch/plain.json" "r[\"argv\"][r[\"argv\"].index(\"--device\")+1]")" = "$gpu" ]' diff --git a/third_party/llama.cpp b/third_party/llama.cpp index bd5b297..d60cc4f 160000 --- a/third_party/llama.cpp +++ b/third_party/llama.cpp @@ -1 +1 @@ -Subproject commit bd5b2970b05ad0632a7be0d4cd39fc761dc793e6 +Subproject commit d60cc4f3ca53d51f05252a802e3ec6188b4dcc94