Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,6 @@ jobs:
- name: Configure
run: cmake -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
- name: Build
run: cmake --build build --target onebit npu_full_elf_test npu_pack_test npu_q4nx_test tokenizer_test -j
run: cmake --build build --target onebit npu_full_elf_test npu_pack_test npu_q4nx_test npu_lax_test tokenizer_test -j
- name: Test
run: ctest --test-dir build --output-on-failure
6 changes: 6 additions & 0 deletions .gitmodules
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,12 @@
branch = main
# ZINC (docs/zinc.md): upstream zolotukhin/zinc, built by scripts/build-zinc.sh;
# .github/workflows/bump-zinc.yml keeps it current.
# The 35B MoE whole-layer decode (docs/npu-lax.md): the open kernels, MIT, built by
# scripts/build-lax.sh; the branch carries our fixes and optimisations of the lax design.
[submodule "third_party/OpenFlowLM-Next"]
path = third_party/OpenFlowLM-Next
url = https://github.com/bong-water-water-bong/OpenFlowLM-Next.git
branch = 1bit/lax-35b
[submodule "third_party/zinc"]
path = third_party/zinc
url = https://github.com/zolotukhin/zinc.git
Expand Down
61 changes: 58 additions & 3 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -182,7 +182,11 @@ option(ONEBIT_NPU_HOST "Build the NPU host library and its tests" ${_onebit_npu_
if(ONEBIT_NPU_HOST)
find_package(PkgConfig REQUIRED)
pkg_check_modules(PCRE2 REQUIRED IMPORTED_TARGET libpcre2-8)
add_library(onebit_npu STATIC npu/full_elf.cpp npu/model.cpp npu/pack.cpp npu/q4nx.cpp npu/tokenizer.cpp)
add_library(onebit_npu STATIC npu/full_elf.cpp npu/lax_elf.cpp npu/lax_pack.cpp npu/lax_stream.cpp npu/lax_turns.cpp npu/model.cpp npu/pack.cpp npu/q4nx.cpp
npu/tokenizer.cpp)
# The lax packer's q8 -> q4_1 re-quantization must round every float32 step on its own,
# as the reference packer's NumPy does (npu/lax_pack.h): no fused multiply-adds.
set_source_files_properties(npu/lax_pack.cpp PROPERTIES COMPILE_OPTIONS -ffp-contract=off)
target_include_directories(onebit_npu PUBLIC npu)
target_link_libraries(onebit_npu PUBLIC nlohmann_json::nlohmann_json PRIVATE PkgConfig::PCRE2)

Expand Down Expand Up @@ -215,6 +219,16 @@ endif()
if(ONEBIT_NPU_MODEL AND ONEBIT_NPU_BO_DUMP)
add_test(NAME npu_pack_model COMMAND npu_pack_test ${ONEBIT_NPU_MODEL} ${ONEBIT_NPU_BO_DUMP})
endif()
# The Qwen3.6-35B-A3B lax decode's host side (docs/npu-lax.md): CI checks the transforms
# against the reference packer's hashes; with the model, all 83 buffers against its table.
add_executable(npu_lax_test tests/npu_lax_test.cpp)
target_link_libraries(npu_lax_test PRIVATE onebit_npu)
add_test(NAME npu_lax_host COMMAND npu_lax_test)
set(ONEBIT_NPU_LAX_MODEL "" CACHE PATH "Qwen3.6-35B-A3B Q4NX directory, for the lax packer and decode tests")
if(ONEBIT_NPU_LAX_MODEL)
add_test(NAME npu_lax_pack_model
COMMAND npu_lax_test --model ${ONEBIT_NPU_LAX_MODEL} --sha256 ${CMAKE_SOURCE_DIR}/tests/golden/npu_lax/sha256.tsv)
endif()
endif() # ONEBIT_NPU_HOST

# The lane itself: needs XRT (the NPU runtime) and an XDNA NPU to run.
Expand All @@ -227,10 +241,10 @@ if(ONEBIT_NPU)
if(NOT EXISTS "${ONEBIT_XRT_ROOT}/include/xrt/xrt_device.h")
message(FATAL_ERROR "ONEBIT_NPU needs XRT; none at ONEBIT_XRT_ROOT=${ONEBIT_XRT_ROOT}")
endif()
add_library(onebit_npu_lane STATIC npu/lane.cpp npu/generate.cpp)
add_library(onebit_npu_lane STATIC npu/lane.cpp npu/generate.cpp npu/lax.cpp npu/lax_elf_kernels.cpp)
target_include_directories(onebit_npu_lane SYSTEM PUBLIC ${ONEBIT_XRT_ROOT}/include)
target_link_libraries(onebit_npu_lane PUBLIC onebit_npu ${ONEBIT_XRT_ROOT}/lib/libxrt_coreutil.so)
target_sources(onebit PRIVATE app/unified.cpp)
target_sources(onebit PRIVATE app/unified.cpp app/npu_lax.cpp)
target_link_libraries(onebit PRIVATE onebit_npu_lane)
target_compile_definitions(onebit PRIVATE ONEBIT_NPU)
set_target_properties(onebit PROPERTIES BUILD_RPATH ${ONEBIT_XRT_ROOT}/lib INSTALL_RPATH ${ONEBIT_XRT_ROOT}/lib)
Expand All @@ -241,6 +255,47 @@ if(ONEBIT_NPU)
COMMAND ${CMAKE_SOURCE_DIR}/tests/npu_lane_e2e.sh $<TARGET_FILE:onebit>
${ONEBIT_NPU_MODEL} ${ONEBIT_NPU_KERNELS} ${ONEBIT_NPU_REFERENCE})
endif()
# `1bit npu-lax`: parity on three positions against the fp64 reference, then one chat turn.
set(ONEBIT_NPU_LAX_KERNELS "" CACHE PATH "scripts/build-lax.sh's <prefix>/kernels")
set(ONEBIT_NPU_LAX_REF "" CACHE PATH "make_decode.py --requant --tokens 3 output (xres<t>.bin, ref_logits*.bin)")
if(ONEBIT_NPU_LAX_KERNELS)
# The full-attention stream's position patches, against the reference harness's table
# for the pinned build (tests/golden/npu_lax/lax_a_patches.tsv names its md5).
add_test(NAME npu_lax_patches
COMMAND npu_lax_test --insts ${ONEBIT_NPU_LAX_KERNELS}/lax_a/insts.bin
--patches ${CMAKE_SOURCE_DIR}/tests/golden/npu_lax/lax_a_patches.tsv)
# The full-ELF position sites of lax_a/insts.elf (tests/golden/npu_lax/lax_a_elf_sites.tsv
# names its md5); with the harness's full ELFs for the same build, every assembled ELF
# byte for byte.
set(ONEBIT_NPU_LAX_ELF_GOLDEN "" CACHE PATH "harness/full_elf.py lax output for the same kernel build (lax_init.elf, lxf.elf, axf_p<N>.elf, ln.elf, lm.elf)")
set(_golden)
if(ONEBIT_NPU_LAX_ELF_GOLDEN)
set(_golden --golden ${ONEBIT_NPU_LAX_ELF_GOLDEN})
endif()
add_test(NAME npu_lax_elf
COMMAND npu_lax_test --elf ${ONEBIT_NPU_LAX_KERNELS}
--sites ${CMAKE_SOURCE_DIR}/tests/golden/npu_lax/lax_a_elf_sites.tsv ${_golden})
endif()
if(ONEBIT_NPU_LAX_MODEL AND ONEBIT_NPU_LAX_KERNELS AND ONEBIT_NPU_LAX_REF)
# Full ELFs (the default), and the classic xclbin path kept for A/B.
add_test(NAME npu_lax_e2e
COMMAND ${CMAKE_SOURCE_DIR}/tests/npu_lax_cpp.sh $<TARGET_FILE:onebit>
${ONEBIT_NPU_LAX_MODEL} ${ONEBIT_NPU_LAX_KERNELS} ${ONEBIT_NPU_LAX_REF} elf)
add_test(NAME npu_lax_e2e_classic
COMMAND ${CMAKE_SOURCE_DIR}/tests/npu_lax_cpp.sh $<TARGET_FILE:onebit>
${ONEBIT_NPU_LAX_MODEL} ${ONEBIT_NPU_LAX_KERNELS} ${ONEBIT_NPU_LAX_REF} classic)
endif()
if(ONEBIT_NPU_LAX_MODEL AND ONEBIT_NPU_LAX_KERNELS)
# A DeltaNet state snapshot restored after another continuation gives the logits of the
# same prefix fed from scratch, bit for bit; a Session follow-up from it, the same tokens.
add_test(NAME npu_lax_snapshot
COMMAND onebit npu-lax --model ${ONEBIT_NPU_LAX_MODEL} --kernels ${ONEBIT_NPU_LAX_KERNELS}
--snapshot-check)
# `1bit serve --device npu` on the 35B: the route, one chat round trip, no xclbin opened.
add_test(NAME npu_lax_serve
COMMAND ${CMAKE_SOURCE_DIR}/tests/npu_lax_serve.sh $<TARGET_FILE:onebit>
${ONEBIT_NPU_LAX_MODEL} ${ONEBIT_NPU_LAX_KERNELS})
endif()
endif()


Expand Down
12 changes: 12 additions & 0 deletions NOTICE
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,18 @@ ZINC (third_party/zinc)
Copyright (c) 2025 ZINC Contributors
License: MIT

OpenFlowLM-Next open kernels (third_party/OpenFlowLM-Next; only open_kernels/ is built)
https://github.com/bong-water-water-bong/OpenFlowLM-Next
(a fork of https://github.com/Atomic-Germ/OpenFlowLM-Next)
open_kernels/: Copyright (c) 2026 Cyrus Attoun and phlegm contributors;
portions derived from OpenFlowLM (https://github.com/OpenFlowLM/OpenFlowLM),
Copyright (c) 2026 Advanced Micro Devices, Inc.
License: MIT (open_kernels/LICENSE); the rest of the tree: Copyright (c)
OpenFlowLM Community, MIT (LICENSE_OPEN_RUNTIME.md)
Includes GEMV tile arithmetic and a smoke test adapted from vegah/LLMNpuTest
(https://github.com/vegah/LLMNpuTest), Apache-2.0
(open_kernels/designs/rot13/LICENSE.LLMNpuTest; attributed in the file headers)

Hugging Face tokenizers (third_party/tokenizers)
https://github.com/huggingface/tokenizers
Copyright Hugging Face, Inc. and the tokenizers authors
Expand Down
4 changes: 3 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,9 @@ serves each model behind an OpenAI-compatible API (`1bit serve`), whatever devic
> vendors Lemonade: Lemonade is the host, 1bit is the engine inside it. Ported so far: HRX on AMD's live ggml-hrx
> ([docs/hrx.md](docs/hrx.md)), Vulkan from upstream llama.cpp's latest release ([docs/vulkan.md](docs/vulkan.md)), the NPU engine on full ELFs with the
> upstream XDNA stack pinned ([docs/npu.md](docs/npu.md); its layer kernel is not yet built from
> source), ZINC ([docs/zinc.md](docs/zinc.md)) and MLX ([docs/apple.md](docs/apple.md)). Next is step 4,
> source), ZINC ([docs/zinc.md](docs/zinc.md)) and MLX ([docs/apple.md](docs/apple.md)). Experimental,
> and served by `1bit serve --device npu` on full ELFs: Qwen3.6-35B-A3B on the NPU, the whole 40-layer
> MoE token as one runlist submit, parity against fp64, 16.3-16.5 tok/s decode ([docs/npu-lax.md](docs/npu-lax.md)). Next is step 4,
> the Laya router. The working engine is being ported from
> [1bit-MONSTER](https://github.com/1bit-MONSTER/1bit-MONSTER); see [docs/PORTING.md](docs/PORTING.md).
> Measured results are on the [wiki](https://github.com/1bit-MONSTER/engine/wiki). This repository
Expand Down
12 changes: 12 additions & 0 deletions app/main.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -21,12 +21,15 @@
// OpenAI endpoints; serve's NPU route.
// 1bit npu-run [options] token ids in, token ids out, on the NPU fast
// lane (docs/npu.md); for checks and benchmarks.
// 1bit npu-lax [options] Qwen3.6-35B-A3B on the NPU lax kernels: greedy
// chat, or the parity check (docs/npu-lax.md).
//
// See docs/PORTING.md for what lands next.

#ifdef ONEBIT_NPU
#include "generate.h"
#include "model.h"
#include "npu_lax.h"
#include "unified.h"
#endif
#include "serve.h"
Expand Down Expand Up @@ -56,6 +59,7 @@ void usage(FILE* out) {
#ifdef ONEBIT_NPU
" unified -m <model dir> serve one NPU model (OpenAI endpoints)\n"
" npu-run [options] generate on the NPU fast lane (npu-run --help)\n"
" npu-lax [options] Qwen3.6-35B-A3B on the NPU lax kernels (npu-lax --help)\n"
#endif
" version print the version\n"
" help show this help\n");
Expand Down Expand Up @@ -148,6 +152,14 @@ int main(int argc, char** argv) {
return 1;
}
}
if (cmd == "npu-lax") {
try {
return onebit::run_npu_lax(argc - 2, argv + 2);
} catch (const std::exception& e) {
std::fprintf(stderr, "1bit npu-lax: %s\n", e.what());
return 1;
}
}
#endif
if (cmd == "version" || cmd == "--version") {
std::printf("1bit %s\n", kVersion);
Expand Down
Loading
Loading