diff --git a/.github/workflows/bump-ds4.yml b/.github/workflows/bump-ds4.yml new file mode 100644 index 00000000..e8a254a1 --- /dev/null +++ b/.github/workflows/bump-ds4.yml @@ -0,0 +1,97 @@ +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Keep third_party/ds4 on upstream antirez/ds4 main (docs/dwarfstar.md): when +# upstream moves, open a PR here moving the submodule. GitHub-hosted CI has no +# GPU; rerun scripts/build-ds4.sh and the checks in docs/dwarfstar.md on Strix +# Halo before merging. +# +# Uses the secret HRX_BUMP_TOKEN (Contents and Pull requests read/write on +# 1bit-MONSTER/engine): a PR opened with the default GITHUB_TOKEN would not +# start CI. +name: bump-ds4 + +on: + schedule: + - cron: "41 6 * * *" + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: bump-ds4 + cancel-in-progress: false + +jobs: + bump: + runs-on: ubuntu-latest + steps: + - name: Require the token + env: + HRX_BUMP_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + run: | + if [ -z "$HRX_BUMP_TOKEN" ]; then + echo "::error::secret HRX_BUMP_TOKEN is not set (see the header of this workflow)" + exit 1 + fi + + - uses: actions/checkout@v4 + with: + token: ${{ secrets.HRX_BUMP_TOKEN }} + + - name: Compare with upstream + id: pins + env: + GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + run: | + set -euo pipefail + upstream=$(gh api repos/antirez/ds4/commits/main --jq .sha) + ours=$(git ls-tree HEAD third_party/ds4 | awk '{print $3}') + echo "upstream $upstream, ours $ours" + { + echo "upstream=$upstream"; echo "ours=$ours" + if [ "$upstream" = "$ours" ]; then echo "changed=false"; else echo "changed=true"; fi + } >> "$GITHUB_OUTPUT" + + - name: Open the bump PR + if: steps.pins.outputs.changed == 'true' + env: + GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + UPSTREAM: ${{ steps.pins.outputs.upstream }} + OURS: ${{ steps.pins.outputs.ours }} + run: | + set -euo pipefail + branch="bump-ds4/${UPSTREAM:0:12}" + if git ls-remote --exit-code origin "refs/heads/$branch" > /dev/null; then + echo "$branch already exists"; exit 0 + fi + log=$(gh api "repos/antirez/ds4/compare/${OURS}...${UPSTREAM}" \ + --jq '.commits[-30:][] | "- \(.sha[0:9]) \(.commit.message | split("\n")[0])"' || true) + git config user.name "ds4-bump" + git config user.email "ds4-bump@users.noreply.github.com" + git switch -c "$branch" + git update-index --cacheinfo "160000,$UPSTREAM,third_party/ds4" + git commit -q -m "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}" + git push -q origin "$branch" + gh pr create --base main --head "$branch" \ + --title "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}" \ + --body "Moves third_party/ds4 from \`${OURS:0:12}\` to upstream main \`${UPSTREAM:0:12}\`. + + Upstream commits (last 30): + ${log} + + CI here has no GPU. Before merging, on Strix Halo: + \`DS4_TEST=1 scripts/build-ds4.sh ~/.cache/ds4-pin rocm\` (builds, then runs DwarfStar's model-free routed-MoE GPU test), then \`1bit serve --device ds4 -m --ctx-size 8192\` answers a chat request (docs/dwarfstar.md)." diff --git a/.gitmodules b/.gitmodules index 53170bbc..7eb314bf 100644 --- a/.gitmodules +++ b/.gitmodules @@ -81,3 +81,10 @@ [submodule "third_party/ryzenai-server"] path = third_party/ryzenai-server url = https://github.com/lemonade-sdk/ryzenai-server +# DwarfStar (docs/dwarfstar.md): upstream antirez/ds4, built by scripts/build-ds4.sh; +# .github/workflows/bump-ds4.yml keeps it current. +[submodule "third_party/ds4"] + path = third_party/ds4 + url = https://github.com/antirez/ds4.git + shallow = true + branch = main diff --git a/CMakeLists.txt b/CMakeLists.txt index 7b29089d..8fdc6f76 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -163,6 +163,31 @@ if(ONEBIT_ZINC) endif() endif() +# ── DwarfStar (docs/dwarfstar.md) ───────────────────────────────────────── +# Off by default: it builds third_party/ds4 (antirez/ds4, MIT) with scripts/build-ds4.sh for +# one backend (rocm: Strix Halo, needs HIP/hipBLAS(Lt)/rocBLAS/rocWMMA/hipCUB; cuda; metal; +# cpu). `1bit serve --device ds4` then runs its ds4-server on DwarfStar's own GGUFs. +option(ONEBIT_DS4 "Build DwarfStar (third_party/ds4) for --device ds4" OFF) +if(ONEBIT_DS4) + if(APPLE) + set(_ds4_default metal) + else() + set(_ds4_default rocm) + endif() + set(ONEBIT_DS4_BACKEND "${_ds4_default}" CACHE STRING "DwarfStar backend: rocm, cuda, metal or cpu") + set_property(CACHE ONEBIT_DS4_BACKEND PROPERTY STRINGS rocm cuda metal cpu) + if(NOT EXISTS "${CMAKE_SOURCE_DIR}/third_party/ds4/ds4_server.c") + message(FATAL_ERROR "ONEBIT_DS4 needs third_party/ds4: git submodule update --init third_party/ds4") + endif() + set(ONEBIT_DS4_SERVER "${CMAKE_BINARY_DIR}/ds4/${ONEBIT_DS4_BACKEND}/ds4-server") + add_custom_target(ds4 + COMMAND ${CMAKE_SOURCE_DIR}/scripts/build-ds4.sh ${CMAKE_BINARY_DIR}/ds4 ${ONEBIT_DS4_BACKEND} + BYPRODUCTS ${ONEBIT_DS4_SERVER} + USES_TERMINAL) + add_dependencies(onebit ds4) + target_compile_definitions(onebit PRIVATE ONEBIT_DS4_SERVER="${ONEBIT_DS4_SERVER}") +endif() + # ── ComfyUI.cpp (docs/comfyui.md) ────────────────────────────────────────── # Off by default: it needs LibTorch. Builds third_party/comfyui.cpp (GPL-3.0) with # scripts/build-comfyui.sh as its own program; `1bit comfy ` runs it as a diff --git a/NOTICE b/NOTICE index b268113a..eaddba60 100644 --- a/NOTICE +++ b/NOTICE @@ -82,6 +82,11 @@ ZINC (third_party/zinc) Copyright (c) 2025 ZINC Contributors License: MIT +DwarfStar (third_party/ds4) + https://github.com/antirez/ds4 + Copyright (c) 2026 The ds4.c authors + License: MIT + Lemonade (third_party/lemonade) https://github.com/1bit-MONSTER/lemonade (a fork of https://github.com/lemonade-sdk/lemonade, with the onebit recipe) diff --git a/README.md b/README.md index f34d2807..fc8abc0e 100644 --- a/README.md +++ b/README.md @@ -28,6 +28,7 @@ serves each model behind an OpenAI-compatible API (`1bit serve`), whatever devic - Vulkan on the Radeon iGPU, from upstream llama.cpp's latest release, so new architectures land the day upstream ships them; architectures upstream lacks, such as Zyphra's ZAYA1, run from our llama.cpp ([docs/vulkan.md](docs/vulkan.md#zaya1-zyphra-from-our-llamacpp)) - a lean option, ROCmFPX's ROCmFP4 and ROCmI4 formats: faster, less accurate ([docs/lean.md](docs/lean.md)) - ZINC, which also reaches NVIDIA GPUs (CUDA) and Apple GPUs (Metal) +- DwarfStar, for DeepSeek V4 Flash, GLM 5.x and Qwen3.8-Flash-Next in its own GGUFs, with SSD expert streaming - MLX on Apple Silicon, through lemon-mlx-engine - ONNX Runtime GenAI models (Lemonade's ONNX format) on the CPU, and on the Radeon through ONNX Runtime's WebGPU provider ([docs/onnx.md](docs/onnx.md)) - Laya, which decides where each request runs @@ -86,6 +87,7 @@ The repositories this engine is built on, in order of importance: | 9 | [huggingface/tokenizers](https://github.com/huggingface/tokenizers) | Every model's `tokenizer.json`, byte-exact, behind our C ABI | Apache-2.0 | | 10 | [NandhaKishorM/laya](https://github.com/NandhaKishorM/laya) | The router that decides where each request runs | Apache-2.0 | | 11 | [ROCm/FastFlowLM](https://github.com/ROCm/FastFlowLM) | The Q4NX NPU model format and its models on Hugging Face (`FastFlowLM/*-NPU2`), which the engine's NPU route runs on its own kernels | MIT | +| 12 | [antirez/ds4](https://github.com/antirez/ds4) (DwarfStar) | DeepSeek V4 Flash, GLM 5.x and Qwen3.8-Flash-Next on its own kernels (ROCm on Strix Halo, CUDA, Metal) | MIT | Also built on [nlohmann/json](https://github.com/nlohmann/json) (MIT) and [yhirose/cpp-httplib](https://github.com/yhirose/cpp-httplib) (MIT). diff --git a/app/serve.cpp b/app/serve.cpp index 9e0697a5..877a7cef 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -39,6 +39,9 @@ // long prompt prefixes on HRX0 over one // shared KV cache (docs/hrx.md) // a .gguf, --device zinc -> this build's zinc +// a DwarfStar .gguf, --device ds4 -> this build's DwarfStar ds4-server (DeepSeek +// V4/V4.1 Flash, GLM 5.x, Qwen3.8-Flash-Next in +// DwarfStar's own GGUFs; docs/dwarfstar.md) // a .gguf, --device rocm -> the ROCm build's llama-server on ROCm0 // (ONEBIT_LEAN_ROCM: any GGUF, and ROCmI4 with // the gfx1151 W4A4 path; docs/lean.md) @@ -112,7 +115,8 @@ struct Options { std::string model, host = "127.0.0.1", device = "auto", alias; int port = 8000, ctx_size = 0; std::string flash_attn; // -fa on/off forwarded to the child llama-server - std::string llama_server, zinc, hrx_libhsa, mlx; + std::string llama_server, zinc, hrx_libhsa, mlx, ds4; + bool ssd_streaming = false; // --device ds4: stream routed experts from the SSD std::string prefill_device; int prefill_min_tokens = 0; bool lean = false; @@ -202,6 +206,16 @@ bool is_onnx_model_dir(const std::string& dir) { return fs::is_directory(dir) && fs::exists(fs::path(dir) / "genai_config.json"); } +// DwarfStar's ds4-server (third_party/ds4, MIT), built by scripts/build-ds4.sh +std::string default_ds4() { + if (const char* e = std::getenv("ONEBIT_DS4"); e && *e) return e; +#ifdef ONEBIT_DS4_SERVER + return ONEBIT_DS4_SERVER; +#else + return "ds4-server"; +#endif +} + std::string default_zinc() { if (const char* e = std::getenv("ONEBIT_ZINC"); e && *e) return e; #ifdef ONEBIT_ZINC_SERVER @@ -253,8 +267,9 @@ int free_port() { // closes, so the backend dies with `1bit serve` for any reason (Linux: PR_SET_PDEATHSIG). class Child { public: - Child(const std::vector& argv, const std::vector& env_extra, int port) - : port_(port) { + Child(const std::vector& argv, const std::vector& env_extra, int port, + std::string ready_path = "/health") + : port_(port), ready_path_(std::move(ready_path)) { // the command line, each argument quoted the way CommandLineToArgvW reads it back std::string cmd; for (const auto& a : argv) { @@ -323,8 +338,9 @@ class Child { #else class Child { public: - Child(const std::vector& argv, const std::vector& env_extra, int port) - : port_(port) { + Child(const std::vector& argv, const std::vector& env_extra, int port, + std::string ready_path = "/health") + : port_(port), ready_path_(std::move(ready_path)) { std::vector args; for (const auto& a : argv) args.push_back(const_cast(a.c_str())); args.push_back(nullptr); @@ -377,14 +393,15 @@ class Child { bool alive() const { return pid_ > 0 && ::waitpid(pid_, nullptr, WNOHANG) == 0; } #endif - // Polls the child's /health until it answers 200, it exits, or time runs out. + // Polls the child's ready path (/health; /v1/models for servers without one, which answer + // only once the model is loaded) until it answers 200, it exits, or time runs out. bool wait_ready(std::chrono::seconds timeout, const std::atomic& stop) const { httplib::Client c("127.0.0.1", port_); c.set_connection_timeout(1); const auto end = std::chrono::steady_clock::now() + timeout; while (std::chrono::steady_clock::now() < end) { if (!alive() || stop) return false; - if (auto r = c.Get("/health"); r && r->status == 200) return true; + if (auto r = c.Get(ready_path_); r && r->status == 200) return true; std::this_thread::sleep_for(std::chrono::milliseconds(250)); } return false; @@ -399,6 +416,7 @@ class Child { pid_t pid_ = -1; #endif int port_; + std::string ready_path_; }; std::string model_id(const Options& o) { @@ -499,6 +517,7 @@ struct Launch { bool drop_model = false; std::string set_model; bool mtp = false; + std::string ready_path = "/health"; }; // The backend process for one device: its command line and environment. @@ -510,6 +529,7 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) { std::vector& env = l.env; bool& drop_model = l.drop_model; std::string& set_model = l.set_model; + if (o.ssd_streaming && device != "ds4") throw std::runtime_error("--ssd-streaming works with --device ds4"); if (device == "onnx") { // ryzenai-server: `-m --port

`; the execution mode (CPU, NPU, hybrid) comes from // the model's genai_config.json @@ -555,23 +575,30 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) { const std::string hsa = hrx_libhsa(o.hrx_libhsa); if (!hsa.empty()) env.push_back("IREE_HAL_AMDGPU_LIBHSA_PATH=" + hsa); } + } else if (device == "ds4") { + // DwarfStar: its own GGUF layouts only; it opens its port after the model has loaded + argv = {o.ds4.empty() ? default_ds4() : o.ds4, "-m", o.model, + "--host", "127.0.0.1", "--port", std::to_string(child_port)}; + if (o.ctx_size > 0) { argv.push_back("--ctx"); argv.push_back(std::to_string(o.ctx_size)); } + if (o.ssd_streaming) argv.push_back("--ssd-streaming"); + l.ready_path = "/v1/models"; } else if (device == "zinc") { argv = {o.zinc.empty() ? default_zinc() : o.zinc, "-m", o.model, "-p", std::to_string(child_port)}; if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } env.push_back("RADV_PERFTEST=coop_matrix"); drop_model = true; // zinc rejects any model id but its own } else { - throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx, rocm or zinc)"); + throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx, rocm, zinc or ds4)"); } if (o.parallel > 1) { // continuous batching: N requests decode together, one read of the weights per step // for all of them; llama-server splits --ctx-size across the slots - if (device == "zinc" || device == "mlx") throw std::runtime_error("--parallel works on the llama.cpp devices (vulkan, hrx, rocm)"); + if (device == "zinc" || device == "mlx" || device == "ds4") throw std::runtime_error("--parallel works on the llama.cpp devices (vulkan, hrx, rocm)"); argv.insert(argv.end(), {"-np", std::to_string(o.parallel)}); } if (!o.mtp.empty()) { // the MTP head drafts tokens on the same device; the model checks them in one batch - if (device == "zinc" || device == "mlx") throw std::runtime_error("--mtp works on the llama.cpp devices (vulkan, hrx, rocm)"); + if (device == "zinc" || device == "mlx" || device == "ds4") throw std::runtime_error("--mtp works on the llama.cpp devices (vulkan, hrx, rocm)"); l.mtp = true; argv.insert(argv.end(), {"--spec-type", "draft-mtp", "-md", o.mtp, "-ngld", "99"}); if (o.mtp_max > 0) { argv.push_back("--spec-draft-n-max"); argv.push_back(std::to_string(o.mtp_max)); } @@ -724,7 +751,7 @@ int serve_child(const Options& o) { if (started[i]) return true; const Launch& b = backends[i]; std::fprintf(stderr, "1bit serve: %s on %s (%s), picked by laya\n", id.c_str(), b.device.c_str(), b.argv[0].c_str()); - auto c = std::make_unique(b.argv, b.env, b.port); + auto c = std::make_unique(b.argv, b.env, b.port, b.ready_path); if (!c->wait_ready(std::chrono::seconds(600), g_stop)) return false; started[i] = c.get(); children.push_back(std::move(c)); @@ -876,7 +903,7 @@ int serve_child(const Options& o) { #endif for (const auto& b : laya ? std::vector{} : backends) { std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", id.c_str(), b.device.c_str(), b.argv[0].c_str()); - children.push_back(std::make_unique(b.argv, b.env, b.port)); + children.push_back(std::make_unique(b.argv, b.env, b.port, b.ready_path)); } for (const auto& b : rag) { std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", b.argv[2].c_str(), b.device.c_str(), b.argv[0].c_str()); @@ -916,8 +943,9 @@ int serve_child(const Options& o) { void usage(FILE* out) { std::fprintf(out, "usage: 1bit serve -m [--port 8000] [--host 127.0.0.1]\n" - " [--device auto|npu|vulkan|hrx|rocm|zinc|mlx|onnx] [--ctx-size N] [--alias NAME]\n" + " [--device auto|npu|vulkan|hrx|rocm|zinc|ds4|mlx|onnx] [--ctx-size N] [--alias NAME]\n" " [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH] [--mlx-server PATH]\n" + " [--ds4 PATH] [--ssd-streaming] DwarfStar (--device ds4; docs/dwarfstar.md)\n" " [--prefill-device hrx] [--prefill-min-tokens N] (with --device vulkan)\n" " [--lean] ROCmFPX formats: ROCmFP4 on vulkan, ROCmI4 with --device rocm\n" " [--mtp HEAD.gguf] [--mtp-max N] [--mtp-p-min P] multi-token prediction (vulkan, hrx, rocm)\n" @@ -951,6 +979,8 @@ int run_serve(int argc, char** argv) { else if (a == "--alias") o.alias = next(); else if (a == "--llama-server") o.llama_server = next(); else if (a == "--zinc") o.zinc = next(); + else if (a == "--ds4") o.ds4 = next(); + else if (a == "--ssd-streaming") o.ssd_streaming = true; else if (a == "--hrx-libhsa") o.hrx_libhsa = next(); else if (a == "--mlx-server") o.mlx = next(); else if (a == "--prefill-device") o.prefill_device = next(); diff --git a/docs/dwarfstar.md b/docs/dwarfstar.md new file mode 100644 index 00000000..2f0ca433 --- /dev/null +++ b/docs/dwarfstar.md @@ -0,0 +1,84 @@ + +# DwarfStar + +[DwarfStar](https://github.com/antirez/ds4) (`antirez/ds4`, MIT) is a native +inference engine written for a few large MoE models: DeepSeek V4 Flash (and V4.1 +Flash and PRO on bigger machines), GLM 5.2/5.3 Flash and Qwen3.8-Flash-Next. It has +its own kernels for Metal, CUDA and ROCm, runs Strix Halo (`gfx1151`) as a first-class +target, and can stream routed experts from the SSD. It is not a general GGUF runner: +it loads its own GGUF layouts (`antirez/deepseek-v4-gguf`, +`antirez/deepseek-v4.1-flash-gguf`, and the others its `download_model.sh` lists), and +llama.cpp cannot load those. + +`1bit serve --device ds4 -m ` runs its `ds4-server` as the model's +backend, behind the same OpenAI API as every other device. + +## Build + +`third_party/ds4` pins upstream `antirez/ds4` main; `.github/workflows/bump-ds4.yml` +opens a PR when it moves. + +``` +git submodule update --init --depth 1 third_party/ds4 +cmake -B build -G Ninja -DONEBIT_DS4=ON # ONEBIT_DS4_BACKEND=rocm|cuda|metal|cpu +cmake --build build +``` + +`scripts/build-ds4.sh [backend]` does the work: it builds a copy of the +source under `/src/` (DwarfStar builds in its tree; the submodule +stays clean) and puts `ds4-server`, `ds4` and `ds4-bench` in `/`. +`DS4_TEST=1` then runs DwarfStar's model-free routed-MoE test on the GPU. + +**ROCm (Strix Halo)** needs HIP, hipBLAS, hipBLASLt, rocBLAS, rocWMMA and hipCUB. The +script takes `ROCM_PATH`, else TheRock's SDK +(`/opt/rocm-therock/lib/python3*/site-packages/_rocm_sdk_devel`), else `/opt/rocm`. +It passes that SDK's `include/` with `-isystem`: clang otherwise searches it after +`/usr/include`, and a distro HIP there (another version) shadows TheRock's headers +and the build fails (`use of undeclared identifier '__ocml_exp10_f32'`). + +## Serving + +``` +1bit serve --device ds4 -m ~/models/ds4/DeepSeek-V4-Flash-IQ2XXS-w2Q2K-AProjQ8-SExpQ8-OutQ8-chat-v2-imatrix-0731.gguf --ctx-size 8192 +``` + +| `1bit serve` | `ds4-server` | +|---|---| +| `-m FILE` | `-m FILE` | +| `--ctx-size N` | `--ctx N` | +| `--ssd-streaming` | `--ssd-streaming`: routed experts from the SSD instead of full residency (GLM 5.x needs it on 128 GB) | +| `--ds4 PATH` | a different `ds4-server` (else `$ONEBIT_DS4`, else this build's, else `ds4-server` on PATH) | + +`ds4-server` opens its port only after the model has loaded, so `1bit serve` waits +for its `/v1/models` (it has no `/health`). `--parallel` and `--mtp` are llama.cpp +options and are refused on `ds4`; DwarfStar has its own batching +(`--batched-session`) and MTP/DSpark drafting, which `serve` does not expose yet. + +Memory: the resident DeepSeek V4 Flash Q2 needs about 81 GiB plus runtime buffers, +and the GPU must see that much (`amdgpu.gttsize` / `ttm.pages_limit`, DwarfStar's +[STRIX_HALO.md](https://github.com/antirez/ds4/blob/main/docs/STRIX_HALO.md)). + +## Verified (Strix Halo, ds4 `0aaea5a238fb`, TheRock ROCm) + +| Check | Result | +|---|---| +| `DS4_TEST=1 scripts/build-ds4.sh build/ds4 rocm` | builds; routed-MoE MXFP4 test PASS (0 failures at 128 and 512 tokens, variants bitwise OK) | +| `-DONEBIT_DS4=ON` engine build | builds; ctest 7/7 | +| `1bit serve --device ds4 -m ` | DwarfStar's "cannot open model", then serve exits: backend did not become ready | +| `--ssd-streaming` with another device | refused | +| a DeepSeek V4 Flash Q2 chat through `1bit serve --device ds4` | not yet run: the box had no room for the 81 GiB model | diff --git a/docs/serve.md b/docs/serve.md index 54917140..090752d1 100644 --- a/docs/serve.md +++ b/docs/serve.md @@ -23,8 +23,8 @@ OpenAI client. ```sh 1bit serve -m [--port 8000] [--host 127.0.0.1] - [--device auto|npu|vulkan|hrx|rocm|zinc|mlx] [--ctx-size N] [--alias NAME] - [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH] [--mlx-server PATH] + [--device auto|npu|vulkan|hrx|rocm|zinc|ds4|mlx] [--ctx-size N] [--alias NAME] + [--llama-server PATH] [--zinc PATH] [--ds4 PATH] [--ssd-streaming] [--hrx-libhsa PATH] [--mlx-server PATH] [--prefill-device hrx] [--prefill-min-tokens N] [--lean] [--mtp HEAD.gguf] [--mtp-max N] [--mtp-p-min P] [--parallel N] [--adaptive] [--adaptive-at N] @@ -49,7 +49,7 @@ One model per process: "llama-server devices" means `vulkan`, `hrx` and `rocm`. These routes go to the llama-server behind the model, which is how Lemonade's llamacpp backend reaches them (the `onebit` recipe -inherits it). On `npu`, `zinc` and `mlx` they answer 501. +inherits it). On `npu`, `zinc`, `ds4` and `mlx` they answer 501. ## Where the model runs @@ -63,6 +63,7 @@ inherits it). On `npu`, `zinc` and `mlx` they answer 501. | ROCmFP4 `.gguf` | `auto`, `vulkan` with `--lean` | the lean (ROCmFPX) build's llama-server on `Vulkan0` (docs/lean.md) | | `.gguf` | `rocm` | the ROCm build's llama-server on `ROCm0` (ROCmFPX's tree, `ONEBIT_LEAN_ROCM`); ROCmI4 files take its W4A4 path (docs/lean.md) | | `.gguf` | `zinc` | this build's ZINC (Vulkan, ROCm or CUDA, whichever it was built for; docs/zinc.md) | +| DwarfStar `.gguf` (DeepSeek V4 Flash, GLM 5.x, Qwen3.8-Flash-Next in its own layouts) | `ds4` | this build's DwarfStar `ds4-server` (ROCm, CUDA or Metal; `--ssd-streaming` streams routed experts; docs/dwarfstar.md) | | Hugging Face id | `mlx` | lemon-mlx-engine's server, on Apple Silicon (docs/apple.md) | A build without the private add-on answers a Qwen3.6-35B-A3B directory with "the @@ -91,8 +92,8 @@ PM4-emulation probe, and then HRX registers no device. `serve` sets `IREE_HAL_AMDGPU_LIBHSA_PATH` itself unless you did. It uses `--hrx-libhsa`, else the build's copy, else the first one under `/opt/rocm-therock`. -The child binaries default to this build's (`-DONEBIT_HRX`, `-DONEBIT_ZINC`), -then `$ONEBIT_LLAMA_SERVER` / `$ONEBIT_ZINC`, then `llama-server` / `zinc` on PATH. +The child binaries default to this build's (`-DONEBIT_HRX`, `-DONEBIT_ZINC`, `-DONEBIT_DS4`), +then `$ONEBIT_LLAMA_SERVER` / `$ONEBIT_ZINC` / `$ONEBIT_DS4`, then `llama-server` / `zinc` / `ds4-server` on PATH. ## Multi-token prediction (`--mtp`) diff --git a/scripts/build-ds4.sh b/scripts/build-ds4.sh new file mode 100755 index 00000000..b69caddc --- /dev/null +++ b/scripts/build-ds4.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# build-ds4.sh [backend] +# +# Builds DwarfStar pinned in third_party/ds4 (upstream antirez/ds4) into +# //ds4-server (with ds4 and ds4-bench beside it). backend is +# one of rocm (default on Linux), cuda, metal (macOS), cpu: +# rocm Strix Halo / gfx1151 (make strix-halo). Needs HIP, hipBLAS, hipBLASLt, +# rocBLAS, rocWMMA and hipCUB: ROCM_PATH, else TheRock's SDK under +# /opt/rocm-therock, else /opt/rocm +# cuda CUDA_ARCH (default native: make cuda-generic) +# metal Apple Silicon (plain make on macOS) +# cpu CPU only +# +# DwarfStar builds in its source tree, so the build runs on a copy under +# /src and the submodule stays clean (docs/dwarfstar.md). DS4_TEST=1 then runs +# DwarfStar's model-free routed-MoE test on the GPU (rocm only: make test-mxfp4-rocm). +set -euo pipefail +prefix=${1:?usage: build-ds4.sh [rocm|cuda|metal|cpu]} +if [ "$(uname -s)" = Darwin ]; then default=metal; else default=rocm; fi +backend=${2:-$default} +case "$backend" in rocm|cuda|metal|cpu) ;; *) echo "unknown backend: $backend"; exit 1 ;; esac +root=$(cd "$(dirname "$0")/.." && pwd) +src=$root/third_party/ds4 +mkdir -p "$prefix" +prefix=$(cd "$prefix" && pwd) + +[ -f "$src/ds4_server.c" ] || { echo "third_party/ds4 is empty: git submodule update --init third_party/ds4"; exit 1; } + +work=$prefix/src/$backend +mkdir -p "$work" +# copy the tree, keeping objects from an earlier build of the same backend +(cd "$src" && tar --exclude=.git -cf - .) | (cd "$work" && tar -xf -) +jobs=$(nproc 2>/dev/null || sysctl -n hw.ncpu) + +case "$backend" in +rocm) + rocm=${ROCM_PATH:-} + if [ -z "$rocm" ]; then + for d in /opt/rocm-therock/lib/python3*/site-packages/_rocm_sdk_devel /opt/rocm; do + [ -x "$d/bin/hipcc" ] && { rocm=$d; break; } + done + fi + [ -n "$rocm" ] && [ -x "$rocm/bin/hipcc" ] || { echo "no ROCm with bin/hipcc: set ROCM_PATH"; exit 1; } + # -isystem: clang otherwise appends the ROCm include directory after /usr/include, + # so a distro HIP there (another version) shadows this one and the build fails + cflags="-O3 -ffast-math -g -fno-finite-math-only -pthread -D__HIP_PLATFORM_AMD__ -Wno-unused-command-line-argument --offload-arch=${ROCM_ARCH:-gfx1151} -isystem $rocm/include" + libs="-L$rocm/lib -Wl,-rpath,$rocm/lib -lm -pthread -lhipblas -lhipblaslt -lrocblas" + (cd "$work" && ROCM_PATH=$rocm HIP_PATH=$rocm make -j"$jobs" strix-halo \ + HIPCC="$rocm/bin/hipcc" ROCM_CFLAGS="$cflags" ROCM_LDLIBS="$libs") + if [ "${DS4_TEST:-0}" = 1 ]; then + (cd "$work" && ROCM_PATH=$rocm HIP_PATH=$rocm make test-mxfp4-rocm \ + HIPCC="$rocm/bin/hipcc" ROCM_CFLAGS="$cflags" ROCM_LDLIBS="$libs") + fi + ;; +cuda) (cd "$work" && make -j"$jobs" cuda CUDA_ARCH="${CUDA_ARCH:-native}") ;; +metal) (cd "$work" && make -j"$jobs") ;; +cpu) (cd "$work" && make -j"$jobs" cpu) ;; +esac + +out=$prefix/$backend +mkdir -p "$out" +for b in ds4-server ds4 ds4-bench; do cp "$work/$b" "$out/$b"; done +commit=$(git -C "$src" rev-parse --short=12 HEAD 2>/dev/null || echo unknown) +echo "DwarfStar ($backend, $commit): $out/ds4-server" diff --git a/third_party/ds4 b/third_party/ds4 new file mode 160000 index 00000000..0aaea5a2 --- /dev/null +++ b/third_party/ds4 @@ -0,0 +1 @@ +Subproject commit 0aaea5a238fb41a35106a551e73c8409dfb751ac