diff --git a/CMakeLists.txt b/CMakeLists.txt index 41e8b4cf..37209418 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -35,10 +35,14 @@ set(BUILD_WEB_APP OFF CACHE BOOL "" FORCE) set(BUILD_TESTING OFF CACHE BOOL "" FORCE) add_subdirectory(third_party/lemonade EXCLUDE_FROM_ALL) -add_executable(onebit app/main.cpp) +add_executable(onebit app/main.cpp app/serve.cpp) set_target_properties(onebit PROPERTIES OUTPUT_NAME 1bit) target_link_libraries(onebit PRIVATE lemonade-server-core) +# ── `1bit serve` end to end (docs/serve.md) ─────────────────────────────── +# Needs a GGUF (e.g. Qwen3-0.6B Q4_K_M); tests each device this build has. +set(ONEBIT_SERVE_TEST_GGUF "" CACHE FILEPATH "GGUF the serve_e2e tests load") + # ── Step 2: HRX + Vulkan in one llama.cpp build (docs/hrx.md) ───────────── # Off by default: it needs TheRock/ROCm and the three HRX submodules, and it # only runs on AMD GPUs. `1bit lemonade` then uses this llama-server for both @@ -48,6 +52,12 @@ if(ONEBIT_HRX) include(cmake/hrx.cmake) add_dependencies(onebit llama_hrx) target_compile_definitions(onebit PRIVATE ONEBIT_HRX_SERVER="${ONEBIT_HRX_SERVER}" ONEBIT_HRX_LIBHSA="${ONEBIT_HRX_LIBHSA}") + if(ONEBIT_SERVE_TEST_GGUF) + foreach(_dev vulkan hrx) + add_test(NAME serve_e2e_${_dev} + COMMAND ${CMAKE_SOURCE_DIR}/tests/serve_e2e.sh $ ${ONEBIT_SERVE_TEST_GGUF} ${_dev}) + endforeach() + endif() add_test(NAME hrx_lemonade_e2e COMMAND ${CMAKE_SOURCE_DIR}/tests/hrx_lemonade_e2e.sh $ ${ONEBIT_HRX_SERVER}) endif() @@ -70,6 +80,10 @@ if(ONEBIT_ZINC) USES_TERMINAL) add_dependencies(onebit zinc) target_compile_definitions(onebit PRIVATE ONEBIT_ZINC_SERVER="${ONEBIT_ZINC_SERVER}") + if(ONEBIT_SERVE_TEST_GGUF) + add_test(NAME serve_e2e_zinc + COMMAND ${CMAKE_SOURCE_DIR}/tests/serve_e2e.sh $ ${ONEBIT_SERVE_TEST_GGUF} zinc) + endif() add_test(NAME zinc_lemonade_e2e COMMAND ${CMAKE_SOURCE_DIR}/tests/zinc_lemonade_e2e.sh $ ${ONEBIT_ZINC_SERVER}) endif() diff --git a/app/main.cpp b/app/main.cpp index 0ca333dd..673d8b1d 100644 --- a/app/main.cpp +++ b/app/main.cpp @@ -38,6 +38,7 @@ #include "model.h" #include "unified.h" #endif +#include "serve.h" #include #include @@ -61,6 +62,7 @@ void usage(FILE* out) { "\n" "commands:\n" " lemonade [lemond options] run Lemonade's server with the engine behind it\n" + " serve -m one model on any device, OpenAI-compatible API\n" #ifdef ONEBIT_NPU " unified -m serve one NPU model (OpenAI endpoints)\n" " npu-run [options] generate on the NPU fast lane (npu-run --help)\n" @@ -237,6 +239,14 @@ int main(int argc, char** argv) { for (int i = 2; i < argc; ++i) args.push_back(argv[i]); return run_lemonade(static_cast(args.size()), args.data()); } + if (cmd == "serve") { + try { + return onebit::run_serve(argc - 2, argv + 2); + } catch (const std::exception& e) { + std::fprintf(stderr, "1bit serve: %s\n", e.what()); + return 1; + } + } #ifdef ONEBIT_NPU if (cmd == "unified") { try { diff --git a/app/serve.cpp b/app/serve.cpp new file mode 100644 index 00000000..02ffa629 --- /dev/null +++ b/app/serve.cpp @@ -0,0 +1,389 @@ +// Copyright 2026 bong-water-water-bong +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// `1bit serve -m [--port 8000] [--device auto|npu|vulkan|hrx|zinc]` +// +// The engine's one front door (docs/serve.md): one model per process behind an +// OpenAI-compatible API, the contract a host such as Lemonade launches a +// backend with: +// +// GET /health, /v1/health 503 while the model loads, then 200 +// GET /v1/models the one model +// POST /v1/chat/completions (stream or not) +// POST /v1/completions +// +// Which device runs the model follows from the model and --device: +// an NPU model directory (model.q4nx + npu/) -> the NPU fast lane, in process +// a .gguf, --device vulkan|hrx -> this build's llama-server +// a .gguf, --device zinc -> this build's zinc +// For a .gguf, the engine starts that server as a private child on a loopback +// port and forwards the OpenAI routes to it, streaming included. `auto` picks +// Vulkan for GGUF (the fastest measured device for standard quants, +// docs/hrx.md) until the Laya router (docs/laya.md) makes that choice. +#include "serve.h" + +#ifdef ONEBIT_NPU +#include "unified.h" +#endif + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +extern char** environ; + +namespace onebit { + +namespace { + +using json = nlohmann::json; +namespace fs = std::filesystem; + +struct Options { + std::string model, host = "127.0.0.1", device = "auto", alias; + int port = 8000, ctx_size = 0; + std::string llama_server, zinc, hrx_libhsa; +}; + +// HRX dlopens the HSA runtime, and a distro libhsa rejects gfx1151's +// PM4-emulation probe, so HRX registers no device (docs/hrx.md). Use, in order: +// --hrx-libhsa, the build's TheRock copy, or the first one under /opt/rocm-therock. +std::string hrx_libhsa(const std::string& option) { + if (!option.empty()) return option; +#ifdef ONEBIT_HRX_LIBHSA + if (fs::exists(ONEBIT_HRX_LIBHSA)) return ONEBIT_HRX_LIBHSA; +#endif + std::error_code ec; + for (fs::recursive_directory_iterator it("/opt/rocm-therock", fs::directory_options::skip_permission_denied, ec), end; + !ec && it != end; it.increment(ec)) { + if (it.depth() > 6) { it.disable_recursion_pending(); continue; } + if (it->path().filename() == "libhsa-runtime64.so.1") return it->path().string(); + } + return ""; +} + +std::string default_llama_server() { + if (const char* e = std::getenv("ONEBIT_LLAMA_SERVER"); e && *e) return e; +#ifdef ONEBIT_HRX_SERVER + return ONEBIT_HRX_SERVER; +#else + return "llama-server"; +#endif +} + +std::string default_zinc() { + if (const char* e = std::getenv("ONEBIT_ZINC"); e && *e) return e; +#ifdef ONEBIT_ZINC_SERVER + return ONEBIT_ZINC_SERVER; +#else + return "zinc"; +#endif +} + +// SIGTERM / SIGINT: stop serving and take the backend down with us. +std::atomic g_stop{false}; +extern "C" void on_stop_signal(int) { g_stop = true; } + +int free_port() { + const int s = ::socket(AF_INET, SOCK_STREAM, 0); + sockaddr_in a{}; + a.sin_family = AF_INET; + a.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + a.sin_port = 0; + socklen_t len = sizeof(a); + if (s < 0 || ::bind(s, reinterpret_cast(&a), sizeof(a)) != 0 || + ::getsockname(s, reinterpret_cast(&a), &len) != 0) { + if (s >= 0) ::close(s); + throw std::runtime_error("cannot find a free loopback port"); + } + const int port = ntohs(a.sin_port); + ::close(s); + return port; +} + +// A child OpenAI-compatible server on a loopback port. +class Child { +public: + Child(const std::vector& argv, const std::vector& env_extra, int port) + : port_(port) { + std::vector args; + for (const auto& a : argv) args.push_back(const_cast(a.c_str())); + args.push_back(nullptr); + std::vector env_store; + for (char** e = environ; *e; ++e) env_store.emplace_back(*e); + for (const auto& e : env_extra) { + const std::string key = e.substr(0, e.find('=') + 1); + bool present = false; + for (const auto& have : env_store) present = present || have.rfind(key, 0) == 0; + if (!present) env_store.push_back(e); // a value the user set wins + } + std::vector envp; + for (auto& e : env_store) envp.push_back(e.data()); + envp.push_back(nullptr); + const pid_t parent = ::getpid(); + pid_ = ::fork(); + if (pid_ < 0) throw std::runtime_error("cannot fork for " + argv[0] + ": " + std::strerror(errno)); + if (pid_ == 0) { + // The kernel ends the child when `1bit serve` dies for any reason, + // SIGKILL included, so a backend is never left running on its own. + ::prctl(PR_SET_PDEATHSIG, SIGTERM); + if (::getppid() != parent) ::_exit(127); // parent already gone + ::execvpe(args[0], args.data(), envp.data()); + ::_exit(127); + } + } + ~Child() { stop(); } + Child(const Child&) = delete; + Child& operator=(const Child&) = delete; + + void stop() { + if (pid_ <= 0) return; + ::kill(pid_, SIGTERM); + for (int i = 0; i < 50; i++) { + if (::waitpid(pid_, nullptr, WNOHANG) == pid_) { pid_ = -1; return; } + std::this_thread::sleep_for(std::chrono::milliseconds(100)); + } + ::kill(pid_, SIGKILL); + ::waitpid(pid_, nullptr, 0); + pid_ = -1; + } + + bool alive() const { return pid_ > 0 && ::waitpid(pid_, nullptr, WNOHANG) == 0; } + + // Polls the child's /health until it answers 200, it exits, or time runs out. + bool wait_ready(std::chrono::seconds timeout, const std::atomic& stop) const { + httplib::Client c("127.0.0.1", port_); + c.set_connection_timeout(1); + const auto end = std::chrono::steady_clock::now() + timeout; + while (std::chrono::steady_clock::now() < end) { + if (!alive() || stop) return false; + if (auto r = c.Get("/health"); r && r->status == 200) return true; + std::this_thread::sleep_for(std::chrono::milliseconds(250)); + } + return false; + } + + int port() const { return port_; } + +private: + pid_t pid_ = -1; + int port_; +}; + +std::string model_id(const Options& o) { + if (!o.alias.empty()) return o.alias; + fs::path p(o.model); + if (p.has_filename() == false) p = p.parent_path(); + return p.stem().string(); +} + +// Forwards an OpenAI POST to the child: the model id becomes the child's (a +// child such as zinc rejects ids it did not load), and replies carry ours. +void forward(const Child& child, const std::string& our_id, bool drop_model, const httplib::Request& req, + httplib::Response& res) { + json body; + try { + body = json::parse(req.body); + } catch (const std::exception&) { + res.status = 400; + res.set_content(R"({"error":{"message":"request body is not JSON"}})", "application/json"); + return; + } + if (drop_model) body.erase("model"); + const bool stream = body.value("stream", false); + const std::string payload = body.dump(); + const std::string path = req.path; + const int port = child.port(); + if (!stream) { + httplib::Client c("127.0.0.1", port); + c.set_read_timeout(3600); + auto r = c.Post(path, payload, "application/json"); + if (!r) { + res.status = 502; + res.set_content(R"({"error":{"message":"backend did not answer"}})", "application/json"); + return; + } + res.status = r->status; + try { + json out = json::parse(r->body); + if (out.is_object() && out.contains("model")) out["model"] = our_id; + res.set_content(out.dump(), "application/json"); + } catch (const std::exception&) { + res.set_content(r->body, r->get_header_value("Content-Type")); + } + return; + } + // Streaming: relay the child's SSE bytes as they arrive. + res.set_chunked_content_provider("text/event-stream", [port, path, payload](size_t, httplib::DataSink& sink) { + httplib::Client c("127.0.0.1", port); + c.set_read_timeout(3600); + c.Post(path, httplib::Headers{}, payload, "application/json", + [&sink](const char* data, size_t n) { return sink.write(data, n); }); + sink.done(); + return true; + }); +} + +int serve_gguf(const Options& o) { + std::string device = o.device == "auto" ? "vulkan" : o.device; + const int child_port = free_port(); + std::vector argv, env; + bool drop_model = false; + if (device == "vulkan" || device == "hrx") { + argv = {o.llama_server.empty() ? default_llama_server() : o.llama_server, + "-m", o.model, "--host", "127.0.0.1", "--port", std::to_string(child_port), + "--device", device == "hrx" ? "HRX0" : "Vulkan0", "-ngl", "99", "--jinja"}; + if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } + if (device == "hrx") { + const std::string hsa = hrx_libhsa(o.hrx_libhsa); + if (!hsa.empty()) env.push_back("IREE_HAL_AMDGPU_LIBHSA_PATH=" + hsa); + } + } else if (device == "zinc") { + argv = {o.zinc.empty() ? default_zinc() : o.zinc, "-m", o.model, "-p", std::to_string(child_port)}; + if (o.ctx_size > 0) { argv.push_back("-c"); argv.push_back(std::to_string(o.ctx_size)); } + env.push_back("RADV_PERFTEST=coop_matrix"); + drop_model = true; // zinc rejects any model id but its own + } else { + throw std::runtime_error("--device " + o.device + " cannot run a .gguf (vulkan, hrx or zinc)"); + } + + const std::string id = model_id(o); + std::atomic ready{false}; + httplib::Server srv; + auto health = [&](const httplib::Request&, httplib::Response& res) { + res.status = ready ? 200 : 503; + res.set_content(ready ? R"({"status":"ok"})" : R"({"status":"loading"})", "application/json"); + }; + srv.Get("/health", health); + srv.Get("/v1/health", health); + srv.Get("/v1/models", [&](const httplib::Request&, httplib::Response& res) { + res.set_content(json{{"object", "list"}, + {"data", json::array({{{"id", id}, {"object", "model"}, {"owned_by", "1bit"}, + {"device", device}}})}}.dump(), + "application/json"); + }); + std::unique_ptr child; + auto post = [&](const httplib::Request& q, httplib::Response& r) { + if (!ready || !child) { + r.status = 503; + r.set_content(R"({"error":{"message":"model is loading"}})", "application/json"); + return; + } + forward(*child, id, drop_model, q, r); + }; + srv.Post("/v1/chat/completions", post); + srv.Post("/v1/completions", post); + + if (!srv.bind_to_port(o.host, o.port)) + throw std::runtime_error("cannot listen on " + o.host + ":" + std::to_string(o.port)); + std::thread listener([&] { srv.listen_after_bind(); }); + struct Stop { + httplib::Server& s; + std::thread& t; + ~Stop() { s.stop(); if (t.joinable()) t.join(); } + } stop{srv, listener}; + + struct sigaction sa{}; + sa.sa_handler = on_stop_signal; + ::sigaction(SIGTERM, &sa, nullptr); + ::sigaction(SIGINT, &sa, nullptr); + std::fprintf(stderr, "1bit serve: %s on %s (%s)\n", id.c_str(), device.c_str(), argv[0].c_str()); + child = std::make_unique(argv, env, child_port); + if (!child->wait_ready(std::chrono::seconds(600), g_stop)) { + if (g_stop) return 0; + std::fprintf(stderr, "1bit serve: %s did not become ready\n", argv[0].c_str()); + return 1; + } + ready = true; + std::fprintf(stderr, "1bit serve: ready on http://%s:%d\n", o.host.c_str(), o.port); + while (child->alive() && !g_stop) std::this_thread::sleep_for(std::chrono::milliseconds(200)); + if (g_stop) { + std::fprintf(stderr, "1bit serve: stopping\n"); + child->stop(); + return 0; + } + std::fprintf(stderr, "1bit serve: the %s backend exited\n", device.c_str()); + return 1; +} + +void usage(FILE* out) { + std::fprintf(out, + "usage: 1bit serve -m [--port 8000] [--host 127.0.0.1]\n" + " [--device auto|npu|vulkan|hrx|zinc] [--ctx-size N] [--alias NAME]\n" + " [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH]\n" + " : an NPU model directory (model.q4nx + npu/) or a .gguf file\n"); +} + +} // namespace + +int run_serve(int argc, char** argv) { + Options o; + for (int i = 0; i < argc; i++) { + const std::string a = argv[i]; + auto next = [&]() -> std::string { + if (i + 1 >= argc) throw std::runtime_error(a + " needs a value"); + return argv[++i]; + }; + if (a == "-m" || a == "--model") o.model = next(); + else if (a == "-p" || a == "--port") o.port = std::stoi(next()); + else if (a == "--host") o.host = next(); + else if (a == "--device") o.device = next(); + else if (a == "-c" || a == "--ctx-size") o.ctx_size = std::stoi(next()); + else if (a == "--alias") o.alias = next(); + else if (a == "--llama-server") o.llama_server = next(); + else if (a == "--zinc") o.zinc = next(); + else if (a == "--hrx-libhsa") o.hrx_libhsa = next(); + else if (a == "-h" || a == "--help") { usage(stdout); return 0; } + else throw std::runtime_error("unknown option " + a); + } + if (o.model.empty()) { usage(stderr); return 2; } + + if (fs::is_directory(o.model)) { + if (o.device != "auto" && o.device != "npu") + throw std::runtime_error("an NPU model directory runs on --device npu"); +#ifdef ONEBIT_NPU + if (!is_npu_model_dir(o.model)) throw std::runtime_error(o.model + " is not an NPU model directory"); + // The NPU lane serves in process (unified.cpp). + std::vector args = {"-m", o.model, "-p", std::to_string(o.port), "--host", o.host}; + if (!o.alias.empty()) { args.push_back("--alias"); args.push_back(o.alias); } + std::vector av; + for (auto& s : args) av.push_back(s.data()); + return run_unified(int(av.size()), av.data()); +#else + throw std::runtime_error("this build has no NPU lane (configure with -DONEBIT_NPU=ON)"); +#endif + } + if (fs::path(o.model).extension() == ".gguf") return serve_gguf(o); + throw std::runtime_error(o.model + ": expected an NPU model directory or a .gguf file"); +} + +} // namespace onebit diff --git a/app/serve.h b/app/serve.h new file mode 100644 index 00000000..41e5bcb9 --- /dev/null +++ b/app/serve.h @@ -0,0 +1,25 @@ +// Copyright 2026 bong-water-water-bong +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// `1bit serve`: the engine's one front door (docs/serve.md). One model per +// process, behind an OpenAI-compatible API, whatever device runs it. +#pragma once + +namespace onebit { + +// argv holds the options after "serve". +int run_serve(int argc, char** argv); + +} // namespace onebit diff --git a/app/unified.cpp b/app/unified.cpp index ed550678..139e96be 100644 --- a/app/unified.cpp +++ b/app/unified.cpp @@ -17,7 +17,7 @@ // behind the OpenAI endpoints Lemonade's `onebit` backend forwards to // (third_party/lemonade/src/cpp/server/backends/onebit/onebit_server.cpp): // -// GET /v1/health 200 once the model is on the device +// GET /health, /v1/health 200 once the model is on the device // GET /v1/models the one model // POST /v1/chat/completions ChatML prompt, greedy decoding, optional SSE stream // POST /v1/completions raw prompt @@ -59,7 +59,7 @@ struct Engine { // The chat scaffold for the model families the lane serves. Hugging Face ships // the template as Jinja; ChatML is what Qwen2 and Qwen3 render it to for plain // text messages, with generation starting after "<|im_start|>assistant\n". -std::string chatml(const json& messages) { +std::string chatml(const json& messages, bool thinking = true) { std::string p; for (const auto& m : messages) { std::string content; @@ -72,7 +72,11 @@ std::string chatml(const json& messages) { } p += "<|im_start|>" + m.at("role").get() + "\n" + content + "<|im_end|>\n"; } - return p + "<|im_start|>assistant\n"; + p += "<|im_start|>assistant\n"; + // chat_template_kwargs.enable_thinking = false: what Qwen3's own template + // emits, an empty think block, so the model answers directly. + if (!thinking) p += "\n\n\n\n"; + return p; } // The longest prefix of `s` that does not end inside a UTF-8 sequence. @@ -102,7 +106,10 @@ void handle_generate(Engine& e, const httplib::Request& req, httplib::Response& } std::string prompt; try { - prompt = chat ? chatml(body.at("messages")) : body.at("prompt").get(); + bool thinking = true; + if (body.contains("chat_template_kwargs") && body["chat_template_kwargs"].is_object()) + thinking = body["chat_template_kwargs"].value("enable_thinking", true); + prompt = chat ? chatml(body.at("messages"), thinking) : body.at("prompt").get(); } catch (const std::exception& ex) { res.status = 400; res.set_content(json{{"error", {{"message", std::string("bad request: ") + ex.what()}}}}.dump(), "application/json"); @@ -203,7 +210,7 @@ bool is_npu_model_dir(const std::string& dir) { } int run_unified(int argc, char** argv) { - std::string model_dir, host = "127.0.0.1"; + std::string model_dir, host = "127.0.0.1", alias; int port = 8000; for (int i = 0; i < argc; ++i) { const std::string a = argv[i]; @@ -214,8 +221,9 @@ int run_unified(int argc, char** argv) { if (a == "-m" || a == "--model") model_dir = next(); else if (a == "-p" || a == "--port") port = std::stoi(next()); else if (a == "--host") host = next(); + else if (a == "--alias") alias = next(); else if (a == "--help" || a == "-h") { - std::printf("usage: 1bit unified -m [-p 8000] [--host 127.0.0.1]\n" + std::printf("usage: 1bit unified -m [-p 8000] [--host 127.0.0.1] [--alias NAME]\n" " holds model.q4nx, config.json, tokenizer.json and npu/ (the lane's kernels)\n"); return 0; } else throw std::runtime_error("unknown option " + a); @@ -225,7 +233,7 @@ int run_unified(int argc, char** argv) { if (std::filesystem::is_regular_file(model_dir)) model_dir = std::filesystem::path(model_dir).parent_path().string(); Engine e; - e.id = std::filesystem::path(model_dir).filename().string(); + e.id = alias.empty() ? std::filesystem::path(model_dir).filename().string() : alias; httplib::Server srv; std::atomic ready{false}; srv.Get("/v1/health", [&](const httplib::Request&, httplib::Response& res) { @@ -233,6 +241,11 @@ int run_unified(int argc, char** argv) { res.set_content(json{{"status", ready ? "ok" : "loading"}, {"model", e.id}, {"device", "npu"}}.dump(), "application/json"); }); + srv.Get("/health", [&](const httplib::Request&, httplib::Response& res) { + res.status = ready ? 200 : 503; + res.set_content(json{{"status", ready ? "ok" : "loading"}, {"model", e.id}, {"device", "npu"}}.dump(), + "application/json"); + }); srv.Get("/v1/models", [&](const httplib::Request&, httplib::Response& res) { res.set_content(json{{"object", "list"}, {"data", json::array({{{"id", e.id}, {"object", "model"}, {"owned_by", "1bit"}}})}}.dump(), "application/json"); diff --git a/docs/serve.md b/docs/serve.md new file mode 100644 index 00000000..557771a2 --- /dev/null +++ b/docs/serve.md @@ -0,0 +1,80 @@ + +# `1bit serve`: the engine behind one OpenAI-compatible API + +The engine is a backend that a host launches, the way Lemonade launches +`llama-server`. It exposes nothing but an OpenAI-compatible API. (The embedded +Lemonade of step 1 is on its way out; see PORTING.md.) + +```sh +1bit serve -m [--port 8000] [--host 127.0.0.1] + [--device auto|npu|vulkan|hrx|zinc] [--ctx-size N] [--alias NAME] + [--llama-server PATH] [--zinc PATH] [--hrx-libhsa PATH] +``` + +One model per process: + +| Endpoint | | +|---|---| +| `GET /health`, `GET /v1/health` | 503 while the model loads, then 200 | +| `GET /v1/models` | the one model (`--alias`, else the file or directory name) | +| `POST /v1/chat/completions` | streamed (SSE) or not | +| `POST /v1/completions` | | + +## Where the model runs + +| Model | `--device` | Runs on | +|---|---|---| +| NPU model directory (`model.q4nx` + `npu/`, docs/npu.md) | `auto`, `npu` | the NPU fast lane, in process | +| `.gguf` | `auto`, `vulkan` | this build's llama-server on `Vulkan0` | +| `.gguf` | `hrx` | this build's llama-server on `HRX0` | +| `.gguf` | `zinc` | this build's ZINC (Vulkan, ROCm or CUDA, whichever it was built for; docs/zinc.md) | + +`chat_template_kwargs.enable_thinking: false` works on every device. The GPU +backends apply the model's own chat template. The NPU route emits what Qwen3's +template does: an empty think block after the assistant prefix. + +For a `.gguf` the engine starts that server as a private child on a loopback +port and forwards the OpenAI routes to it, streaming included. Replies carry +the served model name. For ZINC, which rejects foreign model ids, requests go +out without `model`. `auto` means Vulkan for GGUF, the fastest measured device +for standard quants (docs/hrx.md), until the Laya router (docs/laya.md) makes +that choice per request. + +HRX needs TheRock's HSA runtime: the distro `libhsa` rejects gfx1151's +PM4-emulation probe, and then HRX registers no device. `serve` sets +`IREE_HAL_AMDGPU_LIBHSA_PATH` itself unless you did. It uses `--hrx-libhsa`, +else the build's copy, else the first one under `/opt/rocm-therock`. + +The child binaries default to this build's (`-DONEBIT_HRX`, `-DONEBIT_ZINC`), +then `$ONEBIT_LLAMA_SERVER` / `$ONEBIT_ZINC`, then `llama-server` / `zinc` on PATH. + +## Verified (Strix Halo, 2026-09-23) + +`tests/serve_e2e.sh` with Qwen3-0.6B: the Q4_K_M GGUF for the GPU devices and the Q4NX model directory for the NPU. The test checks `/health` 200, +`/v1/models`, a chat that answers "Paris." under the served name, and +streaming: + +| Device | Result | +|---|---| +| `npu` | PASS (33 SSE chunks): Qwen3-0.6B NPU model directory on the fast lane | +| `vulkan` | PASS (32 SSE chunks) | +| `hrx` | PASS (32 SSE chunks), with no environment set up | +| `zinc` | PASS (6 SSE chunks) | + +ctest runs these as `serve_e2e_` when configured with +`-DONEBIT_SERVE_TEST_GGUF=`. diff --git a/tests/serve_e2e.sh b/tests/serve_e2e.sh new file mode 100755 index 00000000..7d5d5c41 --- /dev/null +++ b/tests/serve_e2e.sh @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# End to end for `1bit serve` (docs/serve.md): serves one model on one device +# and checks the OpenAI-compatible API a host such as Lemonade relies on. +# /health goes 200, /v1/models names the model, a chat answers "Paris" under +# the served name, and a streamed chat arrives as SSE chunks. +# +# usage: tests/serve_e2e.sh path/to/1bit [extra serve args] +set -uo pipefail + +bin=${1:?usage: serve_e2e.sh path/to/1bit [args]} +model=${2:?} +device=${3:?} +shift 3 +port=$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1])') +log=$(mktemp) +"$bin" serve -m "$model" --device "$device" --port "$port" --alias e2e-model "$@" >"$log" 2>&1 & +pid=$! +cleanup() { kill "$pid" 2>/dev/null; wait "$pid" 2>/dev/null; rm -f "$log"; } +trap cleanup EXIT +fail=0 +check() { if eval "$2"; then echo "ok $1"; else echo "FAIL $1"; fail=1; fi; } +api="http://127.0.0.1:$port" + +code=000 +for _ in $(seq 1 1200); do + code=$(curl -s -o /dev/null -w '%{http_code}' "$api/health") + [ "$code" = 200 ] && break + kill -0 "$pid" 2>/dev/null || break + sleep 0.5 +done +check "$device: /health 200" '[ "$code" = 200 ]' +models=$(curl -s "$api/v1/models") +check "$device: /v1/models names e2e-model" '[[ "$models" == *e2e-model* ]]' +body='{"model": "e2e-model", "messages": [{"role": "user", "content": "What is the capital of France? One word."}], + "temperature": 0, "max_tokens": 48, "chat_template_kwargs": {"enable_thinking": false}}' +reply=$(curl -s "$api/v1/chat/completions" -H 'Content-Type: application/json' -d "$body") +content=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["choices"][0]["message"]["content"])' "$reply" 2>/dev/null) +name=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["model"])' "$reply" 2>/dev/null) +echo " said: ${content:-}" +check "$device: answers Paris" '[[ "$content" == *Paris* ]]' +check "$device: reply names e2e-model" '[ "$name" = e2e-model ]' +chunks=$(curl -sN "$api/v1/chat/completions" -H 'Content-Type: application/json' \ + -d '{"model": "e2e-model", "stream": true, "messages": [{"role": "user", "content": "Count to five."}], "max_tokens": 32}' \ + | grep -c '^data: {') +check "$device: streams ($chunks chunks)" '[ "$chunks" -gt 3 ]' + +if [ $fail -ne 0 ]; then echo "--- serve log (tail)"; tail -30 "$log"; echo FAIL; exit 1; fi +echo PASS