Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
97 changes: 97 additions & 0 deletions .github/workflows/bump-ds4.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
# Copyright 2026 bong-water-water-bong
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Keep third_party/ds4 on upstream antirez/ds4 main (docs/dwarfstar.md): when
# upstream moves, open a PR here moving the submodule. GitHub-hosted CI has no
# GPU; rerun scripts/build-ds4.sh and the checks in docs/dwarfstar.md on Strix
# Halo before merging.
#
# Uses the secret HRX_BUMP_TOKEN (Contents and Pull requests read/write on
# 1bit-MONSTER/engine): a PR opened with the default GITHUB_TOKEN would not
# start CI.
name: bump-ds4

on:
schedule:
- cron: "41 6 * * *"
workflow_dispatch:

permissions:
contents: read

concurrency:
group: bump-ds4
cancel-in-progress: false

jobs:
bump:
runs-on: ubuntu-latest
steps:
- name: Require the token
env:
HRX_BUMP_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }}
run: |
if [ -z "$HRX_BUMP_TOKEN" ]; then
echo "::error::secret HRX_BUMP_TOKEN is not set (see the header of this workflow)"
exit 1
fi

- uses: actions/checkout@v4
with:
token: ${{ secrets.HRX_BUMP_TOKEN }}

- name: Compare with upstream
id: pins
env:
GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }}
run: |
set -euo pipefail
upstream=$(gh api repos/antirez/ds4/commits/main --jq .sha)
ours=$(git ls-tree HEAD third_party/ds4 | awk '{print $3}')
echo "upstream $upstream, ours $ours"
{
echo "upstream=$upstream"; echo "ours=$ours"
if [ "$upstream" = "$ours" ]; then echo "changed=false"; else echo "changed=true"; fi
} >> "$GITHUB_OUTPUT"

- name: Open the bump PR
if: steps.pins.outputs.changed == 'true'
env:
GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }}
UPSTREAM: ${{ steps.pins.outputs.upstream }}
OURS: ${{ steps.pins.outputs.ours }}
run: |
set -euo pipefail
branch="bump-ds4/${UPSTREAM:0:12}"
if git ls-remote --exit-code origin "refs/heads/$branch" > /dev/null; then
echo "$branch already exists"; exit 0
fi
log=$(gh api "repos/antirez/ds4/compare/${OURS}...${UPSTREAM}" \
--jq '.commits[-30:][] | "- \(.sha[0:9]) \(.commit.message | split("\n")[0])"' || true)
git config user.name "ds4-bump"
git config user.email "ds4-bump@users.noreply.github.com"
git switch -c "$branch"
git update-index --cacheinfo "160000,$UPSTREAM,third_party/ds4"
git commit -q -m "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}"
git push -q origin "$branch"
gh pr create --base main --head "$branch" \
--title "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}" \
--body "Moves third_party/ds4 from \`${OURS:0:12}\` to upstream main \`${UPSTREAM:0:12}\`.

Upstream commits (last 30):
${log}

CI here has no GPU. Before merging, on Strix Halo:
\`DS4_TEST=1 scripts/build-ds4.sh ~/.cache/ds4-pin rocm\` (builds, then runs DwarfStar's model-free routed-MoE GPU test), then \`1bit serve --device ds4 -m <DeepSeek V4 Flash Q2 .gguf> --ctx-size 8192\` answers a chat request (docs/dwarfstar.md)."
7 changes: 7 additions & 0 deletions .gitmodules
Original file line number Diff line number Diff line change
Expand Up @@ -81,3 +81,10 @@
[submodule "third_party/ryzenai-server"]
path = third_party/ryzenai-server
url = https://github.com/lemonade-sdk/ryzenai-server
# DwarfStar (docs/dwarfstar.md): upstream antirez/ds4, built by scripts/build-ds4.sh;
# .github/workflows/bump-ds4.yml keeps it current.
[submodule "third_party/ds4"]
path = third_party/ds4
url = https://github.com/antirez/ds4.git
shallow = true
branch = main
25 changes: 25 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,31 @@ if(ONEBIT_ZINC)
endif()
endif()

# ── DwarfStar (docs/dwarfstar.md) ─────────────────────────────────────────
# Off by default: it builds third_party/ds4 (antirez/ds4, MIT) with scripts/build-ds4.sh for
# one backend (rocm: Strix Halo, needs HIP/hipBLAS(Lt)/rocBLAS/rocWMMA/hipCUB; cuda; metal;
# cpu). `1bit serve --device ds4` then runs its ds4-server on DwarfStar's own GGUFs.
option(ONEBIT_DS4 "Build DwarfStar (third_party/ds4) for --device ds4" OFF)
if(ONEBIT_DS4)
if(APPLE)
set(_ds4_default metal)
else()
set(_ds4_default rocm)
endif()
set(ONEBIT_DS4_BACKEND "${_ds4_default}" CACHE STRING "DwarfStar backend: rocm, cuda, metal or cpu")
set_property(CACHE ONEBIT_DS4_BACKEND PROPERTY STRINGS rocm cuda metal cpu)
if(NOT EXISTS "${CMAKE_SOURCE_DIR}/third_party/ds4/ds4_server.c")
message(FATAL_ERROR "ONEBIT_DS4 needs third_party/ds4: git submodule update --init third_party/ds4")
endif()
set(ONEBIT_DS4_SERVER "${CMAKE_BINARY_DIR}/ds4/${ONEBIT_DS4_BACKEND}/ds4-server")
add_custom_target(ds4
COMMAND ${CMAKE_SOURCE_DIR}/scripts/build-ds4.sh ${CMAKE_BINARY_DIR}/ds4 ${ONEBIT_DS4_BACKEND}
BYPRODUCTS ${ONEBIT_DS4_SERVER}
USES_TERMINAL)
add_dependencies(onebit ds4)
target_compile_definitions(onebit PRIVATE ONEBIT_DS4_SERVER="${ONEBIT_DS4_SERVER}")
endif()

# ── ComfyUI.cpp (docs/comfyui.md) ──────────────────────────────────────────
# Off by default: it needs LibTorch. Builds third_party/comfyui.cpp (GPL-3.0) with
# scripts/build-comfyui.sh as its own program; `1bit comfy <workflow.json>` runs it as a
Expand Down
5 changes: 5 additions & 0 deletions NOTICE
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,11 @@ ZINC (third_party/zinc)
Copyright (c) 2025 ZINC Contributors
License: MIT

DwarfStar (third_party/ds4)
https://github.com/antirez/ds4
Copyright (c) 2026 The ds4.c authors
License: MIT

Lemonade (third_party/lemonade)
https://github.com/1bit-MONSTER/lemonade
(a fork of https://github.com/lemonade-sdk/lemonade, with the onebit recipe)
Expand Down
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@ serves each model behind an OpenAI-compatible API (`1bit serve`), whatever devic
- Vulkan on the Radeon iGPU, from upstream llama.cpp's latest release, so new architectures land the day upstream ships them; architectures upstream lacks, such as Zyphra's ZAYA1, run from our llama.cpp ([docs/vulkan.md](docs/vulkan.md#zaya1-zyphra-from-our-llamacpp))
- a lean option, ROCmFPX's ROCmFP4 and ROCmI4 formats: faster, less accurate ([docs/lean.md](docs/lean.md))
- ZINC, which also reaches NVIDIA GPUs (CUDA) and Apple GPUs (Metal)
- DwarfStar, for DeepSeek V4 Flash, GLM 5.x and Qwen3.8-Flash-Next in its own GGUFs, with SSD expert streaming
- MLX on Apple Silicon, through lemon-mlx-engine
- ONNX Runtime GenAI models (Lemonade's ONNX format) on the CPU, and on the Radeon through ONNX Runtime's WebGPU provider ([docs/onnx.md](docs/onnx.md))
- Laya, which decides where each request runs
Expand Down Expand Up @@ -86,6 +87,7 @@ The repositories this engine is built on, in order of importance:
| 9 | [huggingface/tokenizers](https://github.com/huggingface/tokenizers) | Every model's `tokenizer.json`, byte-exact, behind our C ABI | Apache-2.0 |
| 10 | [NandhaKishorM/laya](https://github.com/NandhaKishorM/laya) | The router that decides where each request runs | Apache-2.0 |
| 11 | [ROCm/FastFlowLM](https://github.com/ROCm/FastFlowLM) | The Q4NX NPU model format and its models on Hugging Face (`FastFlowLM/*-NPU2`), which the engine's NPU route runs on its own kernels | MIT |
| 12 | [antirez/ds4](https://github.com/antirez/ds4) (DwarfStar) | DeepSeek V4 Flash, GLM 5.x and Qwen3.8-Flash-Next on its own kernels (ROCm on Strix Halo, CUDA, Metal) | MIT |

Also built on [nlohmann/json](https://github.com/nlohmann/json) (MIT)
and [yhirose/cpp-httplib](https://github.com/yhirose/cpp-httplib) (MIT).
Expand Down
56 changes: 43 additions & 13 deletions app/serve.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,9 @@
// long prompt prefixes on HRX0 over one
// shared KV cache (docs/hrx.md)
// a .gguf, --device zinc -> this build's zinc
// a DwarfStar .gguf, --device ds4 -> this build's DwarfStar ds4-server (DeepSeek
// V4/V4.1 Flash, GLM 5.x, Qwen3.8-Flash-Next in
// DwarfStar's own GGUFs; docs/dwarfstar.md)
// a .gguf, --device rocm -> the ROCm build's llama-server on ROCm0
// (ONEBIT_LEAN_ROCM: any GGUF, and ROCmI4 with
// the gfx1151 W4A4 path; docs/lean.md)
Expand Down Expand Up @@ -112,7 +115,8 @@ struct Options {
std::string model, host = "127.0.0.1", device = "auto", alias;
int port = 8000, ctx_size = 0;
std::string flash_attn; // -fa on/off forwarded to the child llama-server
std::string llama_server, zinc, hrx_libhsa, mlx;
std::string llama_server, zinc, hrx_libhsa, mlx, ds4;
bool ssd_streaming = false; // --device ds4: stream routed experts from the SSD
std::string prefill_device;
int prefill_min_tokens = 0;
bool lean = false;
Expand Down Expand Up @@ -202,6 +206,16 @@ bool is_onnx_model_dir(const std::string& dir) {
return fs::is_directory(dir) && fs::exists(fs::path(dir) / "genai_config.json");
}

// DwarfStar's ds4-server (third_party/ds4, MIT), built by scripts/build-ds4.sh
std::string default_ds4() {
if (const char* e = std::getenv("ONEBIT_DS4"); e && *e) return e;
#ifdef ONEBIT_DS4_SERVER
return ONEBIT_DS4_SERVER;
#else
return "ds4-server";
#endif
}

std::string default_zinc() {
if (const char* e = std::getenv("ONEBIT_ZINC"); e && *e) return e;
#ifdef ONEBIT_ZINC_SERVER
Expand Down Expand Up @@ -253,8 +267,9 @@ int free_port() {
// closes, so the backend dies with `1bit serve` for any reason (Linux: PR_SET_PDEATHSIG).
class Child {
public:
Child(const std::vector<std::string>& argv, const std::vector<std::string>& env_extra, int port)
: port_(port) {
Child(const std::vector<std::string>& argv, const std::vector<std::string>& env_extra, int port,
std::string ready_path = "/health")
: port_(port), ready_path_(std::move(ready_path)) {
// the command line, each argument quoted the way CommandLineToArgvW reads it back
std::string cmd;
for (const auto& a : argv) {
Expand Down Expand Up @@ -323,8 +338,9 @@ class Child {
#else
class Child {
public:
Child(const std::vector<std::string>& argv, const std::vector<std::string>& env_extra, int port)
: port_(port) {
Child(const std::vector<std::string>& argv, const std::vector<std::string>& env_extra, int port,
std::string ready_path = "/health")
: port_(port), ready_path_(std::move(ready_path)) {
std::vector<char*> args;
for (const auto& a : argv) args.push_back(const_cast<char*>(a.c_str()));
args.push_back(nullptr);
Expand Down Expand Up @@ -377,14 +393,15 @@ class Child {
bool alive() const { return pid_ > 0 && ::waitpid(pid_, nullptr, WNOHANG) == 0; }
#endif

// Polls the child's /health until it answers 200, it exits, or time runs out.
// Polls the child's ready path (/health; /v1/models for servers without one, which answer
// only once the model is loaded) until it answers 200, it exits, or time runs out.
bool wait_ready(std::chrono::seconds timeout, const std::atomic<bool>& stop) const {
httplib::Client c("127.0.0.1", port_);
c.set_connection_timeout(1);
const auto end = std::chrono::steady_clock::now() + timeout;
while (std::chrono::steady_clock::now() < end) {
if (!alive() || stop) return false;
if (auto r = c.Get("/health"); r && r->status == 200) return true;
if (auto r = c.Get(ready_path_); r && r->status == 200) return true;
std::this_thread::sleep_for(std::chrono::milliseconds(250));
}
return false;
Expand All @@ -399,6 +416,7 @@ class Child {
pid_t pid_ = -1;
#endif
int port_;
std::string ready_path_;
};

std::string model_id(const Options& o) {
Expand Down Expand Up @@ -499,6 +517,7 @@ struct Launch {
bool drop_model = false;
std::string set_model;
bool mtp = false;
std::string ready_path = "/health";
};

// The backend process for one device: its command line and environment.
Expand All @@ -510,6 +529,7 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) {
std::vector<std::string>& env = l.env;
bool& drop_model = l.drop_model;
std::string& set_model = l.set_model;
if (o.ssd_streaming && device != "ds4") throw std::runtime_error("--ssd-streaming works with --device ds4");
if (device == "onnx") {
// ryzenai-server: `-m <dir> --port <p>`; the execution mode (CPU, NPU, hybrid) comes from
// the model's genai_config.json
Expand Down Expand Up @@ -555,23 +575,30 @@ Launch launch_for(const Options& o, const std::string& device, int child_port) {
const std::string hsa = hrx_libhsa(o.hrx_libhsa);
if (!hsa.empty()) env.push_back("IREE_HAL_AMDGPU_LIBHSA_PATH=" + hsa);
}
} else if (device == "ds4") {
// DwarfStar: its own GGUF layouts only; it opens its port after the model has loaded
argv = {o.ds4.empty() ? default_ds4() : o.ds4, "-m", o.model,
"--host", "127.0.0.1", "--port", std::to_string(child_port)};
if (o.ctx_size > 0) { argv.push_back("--ctx"); argv.push_back(std::to_string(o.ctx_size)); }
if (o.ssd_streaming) argv.push_back("--ssd-streaming");
l.ready_path = "/v1/models";
} else if (device == "zinc") {
argv = {o.zinc.empty() ? default_zinc() : o.zinc, "-m", o.model, "-p", std::to_string(child_port)};
if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); }
env.push_back("RADV_PERFTEST=coop_matrix");
drop_model = true; // zinc rejects any model id but its own
} else {
throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx, rocm or zinc)");
throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx, rocm, zinc or ds4)");
}
if (o.parallel > 1) {
// continuous batching: N requests decode together, one read of the weights per step
// for all of them; llama-server splits --ctx-size across the slots
if (device == "zinc" || device == "mlx") throw std::runtime_error("--parallel works on the llama.cpp devices (vulkan, hrx, rocm)");
if (device == "zinc" || device == "mlx" || device == "ds4") throw std::runtime_error("--parallel works on the llama.cpp devices (vulkan, hrx, rocm)");
argv.insert(argv.end(), {"-np", std::to_string(o.parallel)});
}
if (!o.mtp.empty()) {
// the MTP head drafts tokens on the same device; the model checks them in one batch
if (device == "zinc" || device == "mlx") throw std::runtime_error("--mtp works on the llama.cpp devices (vulkan, hrx, rocm)");
if (device == "zinc" || device == "mlx" || device == "ds4") throw std::runtime_error("--mtp works on the llama.cpp devices (vulkan, hrx, rocm)");
l.mtp = true;
argv.insert(argv.end(), {"--spec-type", "draft-mtp", "-md", o.mtp, "-ngld", "99"});
if (o.mtp_max > 0) { argv.push_back("--spec-draft-n-max"); argv.push_back(std::to_string(o.mtp_max)); }
Expand Down Expand Up @@ -724,7 +751,7 @@ int serve_child(const Options& o) {
if (started[i]) return true;
const Launch& b = backends[i];
std::fprintf(stderr, "1bit serve: %s on %s (%s), picked by laya\n", id.c_str(), b.device.c_str(), b.argv[0].c_str());
auto c = std::make_unique<Child>(b.argv, b.env, b.port);
auto c = std::make_unique<Child>(b.argv, b.env, b.port, b.ready_path);
if (!c->wait_ready(std::chrono::seconds(600), g_stop)) return false;
started[i] = c.get();
children.push_back(std::move(c));
Expand Down Expand Up @@ -876,7 +903,7 @@ int serve_child(const Options& o) {
#endif
for (const auto& b : laya ? std::vector<Launch>{} : backends) {
std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", id.c_str(), b.device.c_str(), b.argv[0].c_str());
children.push_back(std::make_unique<Child>(b.argv, b.env, b.port));
children.push_back(std::make_unique<Child>(b.argv, b.env, b.port, b.ready_path));
}
for (const auto& b : rag) {
std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", b.argv[2].c_str(), b.device.c_str(), b.argv[0].c_str());
Expand Down Expand Up @@ -916,8 +943,9 @@ int serve_child(const Options& o) {
void usage(FILE* out) {
std::fprintf(out,
"usage: 1bit serve -m <model> [--port 8000] [--host 127.0.0.1]\n"
" [--device auto|npu|vulkan|hrx|rocm|zinc|mlx|onnx] [--ctx-size N] [--alias NAME]\n"
" [--device auto|npu|vulkan|hrx|rocm|zinc|ds4|mlx|onnx] [--ctx-size N] [--alias NAME]\n"
" [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH] [--mlx-server PATH]\n"
" [--ds4 PATH] [--ssd-streaming] DwarfStar (--device ds4; docs/dwarfstar.md)\n"
" [--prefill-device hrx] [--prefill-min-tokens N] (with --device vulkan)\n"
" [--lean] ROCmFPX formats: ROCmFP4 on vulkan, ROCmI4 with --device rocm\n"
" [--mtp HEAD.gguf] [--mtp-max N] [--mtp-p-min P] multi-token prediction (vulkan, hrx, rocm)\n"
Expand Down Expand Up @@ -951,6 +979,8 @@ int run_serve(int argc, char** argv) {
else if (a == "--alias") o.alias = next();
else if (a == "--llama-server") o.llama_server = next();
else if (a == "--zinc") o.zinc = next();
else if (a == "--ds4") o.ds4 = next();
else if (a == "--ssd-streaming") o.ssd_streaming = true;
else if (a == "--hrx-libhsa") o.hrx_libhsa = next();
else if (a == "--mlx-server") o.mlx = next();
else if (a == "--prefill-device") o.prefill_device = next();
Expand Down
Loading
Loading