From 169876474338db8553b0050e555cdd9b1ccc08d9 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Wed, 23 Sep 2026 18:53:32 -0300 Subject: [PATCH 1/3] 1bit serve: one model on any device behind one OpenAI-compatible API The engine as a backend a host launches (geramyL's point: embed the engine into Lemonade, not Lemonade into the engine). One model per process: /health (+/v1/health) 503 -> 200, /v1/models, /v1/chat/completions (SSE or not), /v1/completions. - NPU model directory -> the NPU fast lane in process (unified.cpp, which now also answers /health). - .gguf -> this build's llama-server (Vulkan0 or HRX0) or ZINC as a private loopback child; OpenAI routes forwarded, streaming relayed as it arrives, replies carry the served name, and ZINC gets requests without 'model'. - HRX: sets IREE_HAL_AMDGPU_LIBHSA_PATH to TheRock's libhsa itself (--hrx-libhsa, the build's copy, or /opt/rocm-therock). - tests/serve_e2e.sh + ctest serve_e2e_ (-DONEBIT_SERVE_TEST_GGUF). Verified on Strix Halo with Qwen3-0.6B Q4_K_M: vulkan, hrx and zinc all PASS (health, models, 'Paris.' under the served name, SSE streaming). The NPU route is not yet run through serve_e2e. Co-Authored-By: Claude Opus 5.5 --- CMakeLists.txt | 16 +- app/main.cpp | 10 ++ app/serve.cpp | 365 +++++++++++++++++++++++++++++++++++++++++++++ app/serve.h | 25 ++++ app/unified.cpp | 7 +- docs/serve.md | 75 ++++++++++ tests/serve_e2e.sh | 63 ++++++++ 7 files changed, 559 insertions(+), 2 deletions(-) create mode 100644 app/serve.cpp create mode 100644 app/serve.h create mode 100644 docs/serve.md create mode 100755 tests/serve_e2e.sh diff --git a/CMakeLists.txt b/CMakeLists.txt index 41e8b4cf..37209418 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -35,10 +35,14 @@ set(BUILD_WEB_APP OFF CACHE BOOL "" FORCE) set(BUILD_TESTING OFF CACHE BOOL "" FORCE) add_subdirectory(third_party/lemonade EXCLUDE_FROM_ALL) -add_executable(onebit app/main.cpp) +add_executable(onebit app/main.cpp app/serve.cpp) set_target_properties(onebit PROPERTIES OUTPUT_NAME 1bit) target_link_libraries(onebit PRIVATE lemonade-server-core) +# ── `1bit serve` end to end (docs/serve.md) ─────────────────────────────── +# Needs a GGUF (e.g. Qwen3-0.6B Q4_K_M); tests each device this build has. +set(ONEBIT_SERVE_TEST_GGUF "" CACHE FILEPATH "GGUF the serve_e2e tests load") + # ── Step 2: HRX + Vulkan in one llama.cpp build (docs/hrx.md) ───────────── # Off by default: it needs TheRock/ROCm and the three HRX submodules, and it # only runs on AMD GPUs. `1bit lemonade` then uses this llama-server for both @@ -48,6 +52,12 @@ if(ONEBIT_HRX) include(cmake/hrx.cmake) add_dependencies(onebit llama_hrx) target_compile_definitions(onebit PRIVATE ONEBIT_HRX_SERVER="${ONEBIT_HRX_SERVER}" ONEBIT_HRX_LIBHSA="${ONEBIT_HRX_LIBHSA}") + if(ONEBIT_SERVE_TEST_GGUF) + foreach(_dev vulkan hrx) + add_test(NAME serve_e2e_${_dev} + COMMAND ${CMAKE_SOURCE_DIR}/tests/serve_e2e.sh $ ${ONEBIT_SERVE_TEST_GGUF} ${_dev}) + endforeach() + endif() add_test(NAME hrx_lemonade_e2e COMMAND ${CMAKE_SOURCE_DIR}/tests/hrx_lemonade_e2e.sh $ ${ONEBIT_HRX_SERVER}) endif() @@ -70,6 +80,10 @@ if(ONEBIT_ZINC) USES_TERMINAL) add_dependencies(onebit zinc) target_compile_definitions(onebit PRIVATE ONEBIT_ZINC_SERVER="${ONEBIT_ZINC_SERVER}") + if(ONEBIT_SERVE_TEST_GGUF) + add_test(NAME serve_e2e_zinc + COMMAND ${CMAKE_SOURCE_DIR}/tests/serve_e2e.sh $ ${ONEBIT_SERVE_TEST_GGUF} zinc) + endif() add_test(NAME zinc_lemonade_e2e COMMAND ${CMAKE_SOURCE_DIR}/tests/zinc_lemonade_e2e.sh $ ${ONEBIT_ZINC_SERVER}) endif() diff --git a/app/main.cpp b/app/main.cpp index 0ca333dd..673d8b1d 100644 --- a/app/main.cpp +++ b/app/main.cpp @@ -38,6 +38,7 @@ #include "model.h" #include "unified.h" #endif +#include "serve.h" #include #include @@ -61,6 +62,7 @@ void usage(FILE* out) { "\n" "commands:\n" " lemonade [lemond options] run Lemonade's server with the engine behind it\n" + " serve -m one model on any device, OpenAI-compatible API\n" #ifdef ONEBIT_NPU " unified -m serve one NPU model (OpenAI endpoints)\n" " npu-run [options] generate on the NPU fast lane (npu-run --help)\n" @@ -237,6 +239,14 @@ int main(int argc, char** argv) { for (int i = 2; i < argc; ++i) args.push_back(argv[i]); return run_lemonade(static_cast(args.size()), args.data()); } + if (cmd == "serve") { + try { + return onebit::run_serve(argc - 2, argv + 2); + } catch (const std::exception& e) { + std::fprintf(stderr, "1bit serve: %s\n", e.what()); + return 1; + } + } #ifdef ONEBIT_NPU if (cmd == "unified") { try { diff --git a/app/serve.cpp b/app/serve.cpp new file mode 100644 index 00000000..24d994e9 --- /dev/null +++ b/app/serve.cpp @@ -0,0 +1,365 @@ +// Copyright 2026 bong-water-water-bong +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// `1bit serve -m [--port 8000] [--device auto|npu|vulkan|hrx|zinc]` +// +// The engine's one front door (docs/serve.md): one model per process behind an +// OpenAI-compatible API, the contract a host such as Lemonade launches a +// backend with: +// +// GET /health, /v1/health 503 while the model loads, then 200 +// GET /v1/models the one model +// POST /v1/chat/completions (stream or not) +// POST /v1/completions +// +// Which device runs the model follows from the model and --device: +// an NPU model directory (model.q4nx + npu/) -> the NPU fast lane, in process +// a .gguf, --device vulkan|hrx -> this build's llama-server +// a .gguf, --device zinc -> this build's zinc +// For a .gguf, the engine starts that server as a private child on a loopback +// port and forwards the OpenAI routes to it, streaming included. `auto` picks +// Vulkan for GGUF (the fastest measured device for standard quants, +// docs/hrx.md) until the Laya router (docs/laya.md) makes that choice. +#include "serve.h" + +#ifdef ONEBIT_NPU +#include "unified.h" +#endif + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +extern char** environ; + +namespace onebit { + +namespace { + +using json = nlohmann::json; +namespace fs = std::filesystem; + +struct Options { + std::string model, host = "127.0.0.1", device = "auto", alias; + int port = 8000, ctx_size = 0; + std::string llama_server, zinc, hrx_libhsa; +}; + +// HRX dlopens the HSA runtime, and a distro libhsa rejects gfx1151's +// PM4-emulation probe, so HRX registers no device (docs/hrx.md). Use, in order: +// --hrx-libhsa, the build's TheRock copy, or the first one under /opt/rocm-therock. +std::string hrx_libhsa(const std::string& option) { + if (!option.empty()) return option; +#ifdef ONEBIT_HRX_LIBHSA + if (fs::exists(ONEBIT_HRX_LIBHSA)) return ONEBIT_HRX_LIBHSA; +#endif + std::error_code ec; + for (fs::recursive_directory_iterator it("/opt/rocm-therock", fs::directory_options::skip_permission_denied, ec), end; + !ec && it != end; it.increment(ec)) { + if (it.depth() > 6) { it.disable_recursion_pending(); continue; } + if (it->path().filename() == "libhsa-runtime64.so.1") return it->path().string(); + } + return ""; +} + +std::string default_llama_server() { + if (const char* e = std::getenv("ONEBIT_LLAMA_SERVER"); e && *e) return e; +#ifdef ONEBIT_HRX_SERVER + return ONEBIT_HRX_SERVER; +#else + return "llama-server"; +#endif +} + +std::string default_zinc() { + if (const char* e = std::getenv("ONEBIT_ZINC"); e && *e) return e; +#ifdef ONEBIT_ZINC_SERVER + return ONEBIT_ZINC_SERVER; +#else + return "zinc"; +#endif +} + +int free_port() { + const int s = ::socket(AF_INET, SOCK_STREAM, 0); + sockaddr_in a{}; + a.sin_family = AF_INET; + a.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + a.sin_port = 0; + socklen_t len = sizeof(a); + if (s < 0 || ::bind(s, reinterpret_cast(&a), sizeof(a)) != 0 || + ::getsockname(s, reinterpret_cast(&a), &len) != 0) { + if (s >= 0) ::close(s); + throw std::runtime_error("cannot find a free loopback port"); + } + const int port = ntohs(a.sin_port); + ::close(s); + return port; +} + +// A child OpenAI-compatible server on a loopback port. +class Child { +public: + Child(const std::vector& argv, const std::vector& env_extra, int port) + : port_(port) { + std::vector args; + for (const auto& a : argv) args.push_back(const_cast(a.c_str())); + args.push_back(nullptr); + std::vector env_store; + for (char** e = environ; *e; ++e) env_store.emplace_back(*e); + for (const auto& e : env_extra) { + const std::string key = e.substr(0, e.find('=') + 1); + bool present = false; + for (const auto& have : env_store) present = present || have.rfind(key, 0) == 0; + if (!present) env_store.push_back(e); // a value the user set wins + } + std::vector envp; + for (auto& e : env_store) envp.push_back(e.data()); + envp.push_back(nullptr); + if (::posix_spawnp(&pid_, args[0], nullptr, nullptr, args.data(), envp.data()) != 0) + throw std::runtime_error("cannot start " + argv[0] + ": " + std::strerror(errno)); + } + ~Child() { stop(); } + Child(const Child&) = delete; + Child& operator=(const Child&) = delete; + + void stop() { + if (pid_ <= 0) return; + ::kill(pid_, SIGTERM); + for (int i = 0; i < 50; i++) { + if (::waitpid(pid_, nullptr, WNOHANG) == pid_) { pid_ = -1; return; } + std::this_thread::sleep_for(std::chrono::milliseconds(100)); + } + ::kill(pid_, SIGKILL); + ::waitpid(pid_, nullptr, 0); + pid_ = -1; + } + + bool alive() const { return pid_ > 0 && ::waitpid(pid_, nullptr, WNOHANG) == 0; } + + // Polls the child's /health until it answers 200, it exits, or time runs out. + bool wait_ready(std::chrono::seconds timeout) const { + httplib::Client c("127.0.0.1", port_); + c.set_connection_timeout(1); + const auto end = std::chrono::steady_clock::now() + timeout; + while (std::chrono::steady_clock::now() < end) { + if (!alive()) return false; + if (auto r = c.Get("/health"); r && r->status == 200) return true; + std::this_thread::sleep_for(std::chrono::milliseconds(250)); + } + return false; + } + + int port() const { return port_; } + +private: + pid_t pid_ = -1; + int port_; +}; + +std::string model_id(const Options& o) { + if (!o.alias.empty()) return o.alias; + fs::path p(o.model); + if (p.has_filename() == false) p = p.parent_path(); + return p.stem().string(); +} + +// Forwards an OpenAI POST to the child: the model id becomes the child's (a +// child such as zinc rejects ids it did not load), and replies carry ours. +void forward(const Child& child, const std::string& our_id, bool drop_model, const httplib::Request& req, + httplib::Response& res) { + json body; + try { + body = json::parse(req.body); + } catch (const std::exception&) { + res.status = 400; + res.set_content(R"({"error":{"message":"request body is not JSON"}})", "application/json"); + return; + } + if (drop_model) body.erase("model"); + const bool stream = body.value("stream", false); + const std::string payload = body.dump(); + const std::string path = req.path; + const int port = child.port(); + if (!stream) { + httplib::Client c("127.0.0.1", port); + c.set_read_timeout(3600); + auto r = c.Post(path, payload, "application/json"); + if (!r) { + res.status = 502; + res.set_content(R"({"error":{"message":"backend did not answer"}})", "application/json"); + return; + } + res.status = r->status; + try { + json out = json::parse(r->body); + if (out.is_object() && out.contains("model")) out["model"] = our_id; + res.set_content(out.dump(), "application/json"); + } catch (const std::exception&) { + res.set_content(r->body, r->get_header_value("Content-Type")); + } + return; + } + // Streaming: relay the child's SSE bytes as they arrive. + res.set_chunked_content_provider("text/event-stream", [port, path, payload](size_t, httplib::DataSink& sink) { + httplib::Client c("127.0.0.1", port); + c.set_read_timeout(3600); + c.Post(path, httplib::Headers{}, payload, "application/json", + [&sink](const char* data, size_t n) { return sink.write(data, n); }); + sink.done(); + return true; + }); +} + +int serve_gguf(const Options& o) { + std::string device = o.device == "auto" ? "vulkan" : o.device; + const int child_port = free_port(); + std::vector argv, env; + bool drop_model = false; + if (device == "vulkan" || device == "hrx") { + argv = {o.llama_server.empty() ? default_llama_server() : o.llama_server, + "-m", o.model, "--host", "127.0.0.1", "--port", std::to_string(child_port), + "--device", device == "hrx" ? "HRX0" : "Vulkan0", "-ngl", "99", "--jinja"}; + if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } + if (device == "hrx") { + const std::string hsa = hrx_libhsa(o.hrx_libhsa); + if (!hsa.empty()) env.push_back("IREE_HAL_AMDGPU_LIBHSA_PATH=" + hsa); + } + } else if (device == "zinc") { + argv = {o.zinc.empty() ? default_zinc() : o.zinc, "-m", o.model, "-p", std::to_string(child_port)}; + if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } + env.push_back("RADV_PERFTEST=coop_matrix"); + drop_model = true; // zinc rejects any model id but its own + } else { + throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx or zinc)"); + } + + const std::string id = model_id(o); + std::atomic ready{false}; + httplib::Server srv; + auto health = [&](const httplib::Request&, httplib::Response& res) { + res.status = ready ? 200 : 503; + res.set_content(ready ? R"({"status":"ok"})" : R"({"status":"loading"})", "application/json"); + }; + srv.Get("/health", health); + srv.Get("/v1/health", health); + srv.Get("/v1/models", [&](const httplib::Request&, httplib::Response& res) { + res.set_content(json{{"object", "list"}, + {"data", json::array({{{"id", id}, {"object", "model"}, {"owned_by", "1bit"}, + {"device", device}}})}}.dump(), + "application/json"); + }); + std::unique_ptr child; + auto post = [&](const httplib::Request& q, httplib::Response& r) { + if (!ready || !child) { + r.status = 503; + r.set_content(R"({"error":{"message":"model is loading"}})", "application/json"); + return; + } + forward(*child, id, drop_model, q, r); + }; + srv.Post("/v1/chat/completions", post); + srv.Post("/v1/completions", post); + + if (!srv.bind_to_port(o.host, o.port)) + throw std::runtime_error("cannot listen on " + o.host + ":" + std::to_string(o.port)); + std::thread listener([&] { srv.listen_after_bind(); }); + struct Stop { + httplib::Server& s; + std::thread& t; + ~Stop() { s.stop(); if (t.joinable()) t.join(); } + } stop{srv, listener}; + + std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", id.c_str(), device.c_str(), argv[0].c_str()); + child = std::make_unique(argv, env, child_port); + if (!child->wait_ready(std::chrono::seconds(600))) { + std::fprintf(stderr, "1bit serve: %s did not become ready\n", argv[0].c_str()); + return 1; + } + ready = true; + std::fprintf(stderr, "1bit serve: ready on http://%s:%d\n", o.host.c_str(), o.port); + while (child->alive()) std::this_thread::sleep_for(std::chrono::milliseconds(500)); + std::fprintf(stderr, "1bit serve: the %s backend exited\n", device.c_str()); + return 1; +} + +void usage(FILE* out) { + std::fprintf(out, + "usage: 1bit serve -m [--port 8000] [--host 127.0.0.1]\n" + " [--device auto|npu|vulkan|hrx|zinc] [--ctx-size N] [--alias NAME]\n" + " [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH]\n" + " : an NPU model directory (model.q4nx + npu/) or a .gguf file\n"); +} + +} // namespace + +int run_serve(int argc, char** argv) { + Options o; + for (int i = 0; i < argc; i++) { + const std::string a = argv[i]; + auto next = [&]() -> std::string { + if (i + 1 >= argc) throw std::runtime_error(a + " needs a value"); + return argv[++i]; + }; + if (a == "-m" || a == "--model") o.model = next(); + else if (a == "-p" || a == "--port") o.port = std::stoi(next()); + else if (a == "--host") o.host = next(); + else if (a == "--device") o.device = next(); + else if (a == "-c" || a == "--ctx-size") o.ctx_size = std::stoi(next()); + else if (a == "--alias") o.alias = next(); + else if (a == "--llama-server") o.llama_server = next(); + else if (a == "--zinc") o.zinc = next(); + else if (a == "--hrx-libhsa") o.hrx_libhsa = next(); + else if (a == "-h" || a == "--help") { usage(stdout); return 0; } + else throw std::runtime_error("unknown option " + a); + } + if (o.model.empty()) { usage(stderr); return 2; } + + if (fs::is_directory(o.model)) { + if (o.device != "auto" && o.device != "npu") + throw std::runtime_error("an NPU model directory runs on --device npu"); +#ifdef ONEBIT_NPU + if (!is_npu_model_dir(o.model)) throw std::runtime_error(o.model + " is not an NPU model directory"); + // The NPU lane serves in process (unified.cpp). + std::vector args = {"-m", o.model, "-p", std::to_string(o.port), "--host", o.host}; + std::vector av; + for (auto& s : args) av.push_back(s.data()); + return run_unified(int(av.size()), av.data()); +#else + throw std::runtime_error("this build has no NPU lane (configure with -DONEBIT_NPU=ON)"); +#endif + } + if (fs::path(o.model).extension() == ".gguf") return serve_gguf(o); + throw std::runtime_error(o.model + ": expected an NPU model directory or a .gguf file"); +} + +} // namespace onebit diff --git a/app/serve.h b/app/serve.h new file mode 100644 index 00000000..41e5bcb9 --- /dev/null +++ b/app/serve.h @@ -0,0 +1,25 @@ +// Copyright 2026 bong-water-water-bong +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// `1bit serve`: the engine's one front door (docs/serve.md). One model per +// process, behind an OpenAI-compatible API, whatever device runs it. +#pragma once + +namespace onebit { + +// argv holds the options after "serve". +int run_serve(int argc, char** argv); + +} // namespace onebit diff --git a/app/unified.cpp b/app/unified.cpp index ed550678..13c917b3 100644 --- a/app/unified.cpp +++ b/app/unified.cpp @@ -17,7 +17,7 @@ // behind the OpenAI endpoints Lemonade's `onebit` backend forwards to // (third_party/lemonade/src/cpp/server/backends/onebit/onebit_server.cpp): // -// GET /v1/health 200 once the model is on the device +// GET /health, /v1/health 200 once the model is on the device // GET /v1/models the one model // POST /v1/chat/completions ChatML prompt, greedy decoding, optional SSE stream // POST /v1/completions raw prompt @@ -233,6 +233,11 @@ int run_unified(int argc, char** argv) { res.set_content(json{{"status", ready ? "ok" : "loading"}, {"model", e.id}, {"device", "npu"}}.dump(), "application/json"); }); + srv.Get("/health", [&](const httplib::Request&, httplib::Response& res) { + res.status = ready ? 200 : 503; + res.set_content(json{{"status", ready ? "ok" : "loading"}, {"model", e.id}, {"device", "npu"}}.dump(), + "application/json"); + }); srv.Get("/v1/models", [&](const httplib::Request&, httplib::Response& res) { res.set_content(json{{"object", "list"}, {"data", json::array({{{"id", e.id}, {"object", "model"}, {"owned_by", "1bit"}}})}}.dump(), "application/json"); diff --git a/docs/serve.md b/docs/serve.md new file mode 100644 index 00000000..f1dc9d98 --- /dev/null +++ b/docs/serve.md @@ -0,0 +1,75 @@ + +# `1bit serve`: the engine behind one OpenAI-compatible API + +The engine is a backend that a host launches, the way Lemonade launches +`llama-server`. It exposes nothing but an OpenAI-compatible API. (The embedded +Lemonade of step 1 is on its way out; see PORTING.md.) + +```sh +1bit serve -m [--port 8000] [--host 127.0.0.1] + [--device auto|npu|vulkan|hrx|zinc] [--ctx-size N] [--alias NAME] + [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH] +``` + +One model per process: + +| Endpoint | | +|---|---| +| `GET /health`, `GET /v1/health` | 503 while the model loads, then 200 | +| `GET /v1/models` | the one model (`--alias`, else the file or directory name) | +| `POST /v1/chat/completions` | streamed (SSE) or not | +| `POST /v1/completions` | | + +## Where the model runs + +| Model | `--device` | Runs on | +|---|---|---| +| NPU model directory (`model.q4nx` + `npu/`, docs/npu.md) | `auto`, `npu` | the NPU fast lane, in process | +| `.gguf` | `auto`, `vulkan` | this build's llama-server on `Vulkan0` | +| `.gguf` | `hrx` | this build's llama-server on `HRX0` | +| `.gguf` | `zinc` | this build's ZINC (Vulkan, ROCm or CUDA, whichever it was built for; docs/zinc.md) | + +For a `.gguf` the engine starts that server as a private child on a loopback +port and forwards the OpenAI routes to it, streaming included. Replies carry +the served model name. For ZINC, which rejects foreign model ids, requests go +out without `model`. `auto` means Vulkan for GGUF, the fastest measured device +for standard quants (docs/hrx.md), until the Laya router (docs/laya.md) makes +that choice per request. + +HRX needs TheRock's HSA runtime: the distro `libhsa` rejects gfx1151's +PM4-emulation probe, and then HRX registers no device. `serve` sets +`IREE_HAL_AMDGPU_LIBHSA_PATH` itself unless you did. It uses `--hrx-libhsa`, +else the build's copy, else the first one under `/opt/rocm-therock`. + +The child binaries default to this build's (`-DONEBIT_HRX`, `-DONEBIT_ZINC`), +then `$ONEBIT_LLAMA_SERVER` / `$ONEBIT_ZINC`, then `llama-server` / `zinc` on PATH. + +## Verified (Strix Halo, 2026-09-23) + +`tests/serve_e2e.sh` with Qwen3-0.6B Q4_K_M. The test checks `/health` 200, +`/v1/models`, a chat that answers "Paris." under the served name, and +streaming: + +| Device | Result | +|---|---| +| `vulkan` | PASS (32 SSE chunks) | +| `hrx` | PASS (32 SSE chunks), with no environment set up | +| `zinc` | PASS (6 SSE chunks) | + +ctest runs these as `serve_e2e_` when configured with +`-DONEBIT_SERVE_TEST_GGUF=`. diff --git a/tests/serve_e2e.sh b/tests/serve_e2e.sh new file mode 100755 index 00000000..7d5d5c41 --- /dev/null +++ b/tests/serve_e2e.sh @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# End to end for `1bit serve` (docs/serve.md): serves one model on one device +# and checks the OpenAI-compatible API a host such as Lemonade relies on. +# /health goes 200, /v1/models names the model, a chat answers "Paris" under +# the served name, and a streamed chat arrives as SSE chunks. +# +# usage: tests/serve_e2e.sh path/to/1bit [extra serve args] +set -uo pipefail + +bin=${1:?usage: serve_e2e.sh path/to/1bit [args]} +model=${2:?} +device=${3:?} +shift 3 +port=$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])') +log=$(mktemp) +"$bin" serve -m "$model" --device "$device" --port "$port" --alias e2e-model "$@" >"$log" 2>&1 & +pid=$! +cleanup() { kill "$pid" 2>/dev/null; wait "$pid" 2>/dev/null; rm -f "$log"; } +trap cleanup EXIT +fail=0 +check() { if eval "$2"; then echo "ok $1"; else echo "FAIL $1"; fail=1; fi; } +api="http://127.0.0.1:$port" + +code=000 +for _ in $(seq 1 1200); do + code=$(curl -s -o /dev/null -w '%{http_code}' "$api/health") + [ "$code" = 200 ] && break + kill -0 "$pid" 2>/dev/null || break + sleep 0.5 +done +check "$device: /health 200" '[ "$code" = 200 ]' +models=$(curl -s "$api/v1/models") +check "$device: /v1/models names e2e-model" '[[ "$models" == *e2e-model* ]]' +body='{"model": "e2e-model", "messages": [{"role": "user", "content": "What is the capital of France? One word."}], + "temperature": 0, "max_tokens": 48, "chat_template_kwargs": {"enable_thinking": false}}' +reply=$(curl -s "$api/v1/chat/completions" -H 'Content-Type: application/json' -d "$body") +content=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["choices"][0]["message"]["content"])' "$reply" 2>/dev/null) +name=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["model"])' "$reply" 2>/dev/null) +echo " said: ${content:-}" +check "$device: answers Paris" '[[ "$content" == *Paris* ]]' +check "$device: reply names e2e-model" '[ "$name" = e2e-model ]' +chunks=$(curl -sN "$api/v1/chat/completions" -H 'Content-Type: application/json' \ + -d '{"model": "e2e-model", "stream": true, "messages": [{"role": "user", "content": "Count to five."}], "max_tokens": 32}' \ + | grep -c '^data: {') +check "$device: streams ($chunks chunks)" '[ "$chunks" -gt 3 ]' + +if [ $fail -ne 0 ]; then echo "--- serve log (tail)"; tail -30 "$log"; echo FAIL; exit 1; fi +echo PASS From c2c08042d5274f6cd50005754de36a0f4e126194 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Wed, 23 Sep 2026 19:23:18 -0300 Subject: [PATCH 2/3] 1bit serve: never leave a backend running The e2e test left llama-server/zinc children behind: SIGTERM ended serve without its destructor, and nothing could clean up after SIGKILL. The child is now forked with PR_SET_PDEATHSIG (the kernel ends it when serve dies, SIGKILL included), and SIGTERM/SIGINT stop the server and the child cleanly. Verified on Strix Halo: after SIGTERM and after SIGKILL of serve, the child is gone; serve_e2e vulkan still passes. Co-Authored-By: Claude Opus 5.5 --- app/serve.cpp | 37 ++++++++++++++++++++++++++++++------- 1 file changed, 30 insertions(+), 7 deletions(-) diff --git a/app/serve.cpp b/app/serve.cpp index 24d994e9..afcc3b34 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -49,7 +49,7 @@ #include #include #include -#include +#include #include #include #include @@ -109,6 +109,10 @@ std::string default_zinc() { #endif } +// SIGTERM / SIGINT: stop serving and take the backend down with us. +std::atomic g_stop{false}; +extern "C" void on_stop_signal(int) { g_stop = true; } + int free_port() { const int s = ::socket(AF_INET, SOCK_STREAM, 0); sockaddr_in a{}; @@ -145,8 +149,17 @@ class Child { std::vector envp; for (auto& e : env_store) envp.push_back(e.data()); envp.push_back(nullptr); - if (::posix_spawnp(&pid_, args[0], nullptr, nullptr, args.data(), envp.data()) != 0) - throw std::runtime_error("cannot start " + argv[0] + ": " + std::strerror(errno)); + const pid_t parent = ::getpid(); + pid_ = ::fork(); + if (pid_ < 0) throw std::runtime_error("cannot fork for " + argv[0] + ": " + std::strerror(errno)); + if (pid_ == 0) { + // The kernel ends the child when `1bit serve` dies for any reason, + // SIGKILL included, so a backend is never left running on its own. + ::prctl(PR_SET_PDEATHSIG, SIGTERM); + if (::getppid() != parent) ::_exit(127); // parent already gone + ::execvpe(args[0], args.data(), envp.data()); + ::_exit(127); + } } ~Child() { stop(); } Child(const Child&) = delete; @@ -167,12 +180,12 @@ class Child { bool alive() const { return pid_ > 0 && ::waitpid(pid_, nullptr, WNOHANG) == 0; } // Polls the child's /health until it answers 200, it exits, or time runs out. - bool wait_ready(std::chrono::seconds timeout) const { + bool wait_ready(std::chrono::seconds timeout, const std::atomic& stop) const { httplib::Client c("127.0.0.1", port_); c.set_connection_timeout(1); const auto end = std::chrono::steady_clock::now() + timeout; while (std::chrono::steady_clock::now() < end) { - if (!alive()) return false; + if (!alive() || stop) return false; if (auto r = c.Get("/health"); r && r->status == 200) return true; std::this_thread::sleep_for(std::chrono::milliseconds(250)); } @@ -299,15 +312,25 @@ int serve_gguf(const Options& o) { ~Stop() { s.stop(); if (t.joinable()) t.join(); } } stop{srv, listener}; + struct sigaction sa{}; + sa.sa_handler = on_stop_signal; + ::sigaction(SIGTERM, &sa, nullptr); + ::sigaction(SIGINT, &sa, nullptr); std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", id.c_str(), device.c_str(), argv[0].c_str()); child = std::make_unique(argv, env, child_port); - if (!child->wait_ready(std::chrono::seconds(600))) { + if (!child->wait_ready(std::chrono::seconds(600), g_stop)) { + if (g_stop) return 0; std::fprintf(stderr, "1bit serve: %s did not become ready\n", argv[0].c_str()); return 1; } ready = true; std::fprintf(stderr, "1bit serve: ready on http://%s:%d\n", o.host.c_str(), o.port); - while (child->alive()) std::this_thread::sleep_for(std::chrono::milliseconds(500)); + while (child->alive() && !g_stop) std::this_thread::sleep_for(std::chrono::milliseconds(200)); + if (g_stop) { + std::fprintf(stderr, "1bit serve: stopping\n"); + child->stop(); + return 0; + } std::fprintf(stderr, "1bit serve: the %s backend exited\n", device.c_str()); return 1; } From 8f892347c6aea91d26ec2c0fb99a335103854a54 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Wed, 23 Sep 2026 19:33:58 -0300 Subject: [PATCH 3/3] 1bit serve: NPU route passes end to end; --alias and enable_thinking on the NPU serve passes --alias through to the in-process NPU server (unified now takes --alias), and the NPU route honours chat_template_kwargs.enable_thinking=false the way Qwen3's template does (an empty think block after the assistant prefix), so every device answers alike. serve_e2e npu PASS on Strix Halo (Qwen3-0.6B NPU model dir: 'Paris' under the served name, 33 SSE chunks). Co-Authored-By: Claude Opus 5.5 --- app/serve.cpp | 1 + app/unified.cpp | 20 ++++++++++++++------ docs/serve.md | 7 ++++++- 3 files changed, 21 insertions(+), 7 deletions(-) diff --git a/app/serve.cpp b/app/serve.cpp index afcc3b34..02ffa629 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -374,6 +374,7 @@ int run_serve(int argc, char** argv) { if (!is_npu_model_dir(o.model)) throw std::runtime_error(o.model + " is not an NPU model directory"); // The NPU lane serves in process (unified.cpp). std::vector args = {"-m", o.model, "-p", std::to_string(o.port), "--host", o.host}; + if (!o.alias.empty()) { args.push_back("--alias"); args.push_back(o.alias); } std::vector av; for (auto& s : args) av.push_back(s.data()); return run_unified(int(av.size()), av.data()); diff --git a/app/unified.cpp b/app/unified.cpp index 13c917b3..139e96be 100644 --- a/app/unified.cpp +++ b/app/unified.cpp @@ -59,7 +59,7 @@ struct Engine { // The chat scaffold for the model families the lane serves. Hugging Face ships // the template as Jinja; ChatML is what Qwen2 and Qwen3 render it to for plain // text messages, with generation starting after "<|im_start|>assistant\n". -std::string chatml(const json& messages) { +std::string chatml(const json& messages, bool thinking = true) { std::string p; for (const auto& m : messages) { std::string content; @@ -72,7 +72,11 @@ std::string chatml(const json& messages) { } p += "<|im_start|>" + m.at("role").get() + "\n" + content + "<|im_end|>\n"; } - return p + "<|im_start|>assistant\n"; + p += "<|im_start|>assistant\n"; + // chat_template_kwargs.enable_thinking = false: what Qwen3's own template + // emits, an empty think block, so the model answers directly. + if (!thinking) p += "\n\n\n\n"; + return p; } // The longest prefix of `s` that does not end inside a UTF-8 sequence. @@ -102,7 +106,10 @@ void handle_generate(Engine& e, const httplib::Request& req, httplib::Response& } std::string prompt; try { - prompt = chat ? chatml(body.at("messages")) : body.at("prompt").get(); + bool thinking = true; + if (body.contains("chat_template_kwargs") && body["chat_template_kwargs"].is_object()) + thinking = body["chat_template_kwargs"].value("enable_thinking", true); + prompt = chat ? chatml(body.at("messages"), thinking) : body.at("prompt").get(); } catch (const std::exception& ex) { res.status = 400; res.set_content(json{{"error", {{"message", std::string("bad request: ") + ex.what()}}}}.dump(), "application/json"); @@ -203,7 +210,7 @@ bool is_npu_model_dir(const std::string& dir) { } int run_unified(int argc, char** argv) { - std::string model_dir, host = "127.0.0.1"; + std::string model_dir, host = "127.0.0.1", alias; int port = 8000; for (int i = 0; i < argc; ++i) { const std::string a = argv[i]; @@ -214,8 +221,9 @@ int run_unified(int argc, char** argv) { if (a == "-m" || a == "--model") model_dir = next(); else if (a == "-p" || a == "--port") port = std::stoi(next()); else if (a == "--host") host = next(); + else if (a == "--alias") alias = next(); else if (a == "--help" || a == "-h") { - std::printf("usage: 1bit unified -m [-p 8000] [--host 127.0.0.1]\n" + std::printf("usage: 1bit unified -m [-p 8000] [--host 127.0.0.1] [--alias NAME]\n" " holds model.q4nx, config.json, tokenizer.json and npu/ (the lane's kernels)\n"); return 0; } else throw std::runtime_error("unknown option " + a); @@ -225,7 +233,7 @@ int run_unified(int argc, char** argv) { if (std::filesystem::is_regular_file(model_dir)) model_dir = std::filesystem::path(model_dir).parent_path().string(); Engine e; - e.id = std::filesystem::path(model_dir).filename().string(); + e.id = alias.empty() ? std::filesystem::path(model_dir).filename().string() : alias; httplib::Server srv; std::atomic ready{false}; srv.Get("/v1/health", [&](const httplib::Request&, httplib::Response& res) { diff --git a/docs/serve.md b/docs/serve.md index f1dc9d98..557771a2 100644 --- a/docs/serve.md +++ b/docs/serve.md @@ -44,6 +44,10 @@ One model per process: | `.gguf` | `hrx` | this build's llama-server on `HRX0` | | `.gguf` | `zinc` | this build's ZINC (Vulkan, ROCm or CUDA, whichever it was built for; docs/zinc.md) | +`chat_template_kwargs.enable_thinking: false` works on every device. The GPU +backends apply the model's own chat template. The NPU route emits what Qwen3's +template does: an empty think block after the assistant prefix. + For a `.gguf` the engine starts that server as a private child on a loopback port and forwards the OpenAI routes to it, streaming included. Replies carry the served model name. For ZINC, which rejects foreign model ids, requests go @@ -61,12 +65,13 @@ then `$ONEBIT_LLAMA_SERVER` / `$ONEBIT_ZINC`, then `llama-server` / `zinc` on PA ## Verified (Strix Halo, 2026-09-23) -`tests/serve_e2e.sh` with Qwen3-0.6B Q4_K_M. The test checks `/health` 200, +`tests/serve_e2e.sh` with Qwen3-0.6B: the Q4_K_M GGUF for the GPU devices and the Q4NX model directory for the NPU. The test checks `/health` 200, `/v1/models`, a chat that answers "Paris." under the served name, and streaming: | Device | Result | |---|---| +| `npu` | PASS (33 SSE chunks): Qwen3-0.6B NPU model directory on the fast lane | | `vulkan` | PASS (32 SSE chunks) | | `hrx` | PASS (32 SSE chunks), with no environment set up | | `zinc` | PASS (6 SSE chunks) |