diff --git a/.github/workflows/bump-ds4.yml b/.github/workflows/bump-ds4.yml index 9f034fbd..14d18150 100644 --- a/.github/workflows/bump-ds4.yml +++ b/.github/workflows/bump-ds4.yml @@ -13,14 +13,17 @@ # See the License for the specific language governing permissions and # limitations under the License. # -# Keep third_party/ds4 on upstream antirez/ds4 main (docs/dwarfstar.md): when -# upstream moves, open a PR here moving the submodule. GitHub-hosted CI has no -# GPU; rerun scripts/build-ds4.sh and the checks in docs/dwarfstar.md on Strix -# Halo before merging. +# Keep third_party/ds4 on upstream antirez/ds4 main (docs/dwarfstar.md). The submodule is +# our fork 1bit-MONSTER/ds4, branch 1bit/main: upstream main plus the commits we carry (1BP +# packages, docs/1BP.md in the fork). When upstream main moves, this workflow rebases those +# commits onto it (one upstream has since taken becomes empty and drops out), tags the old tip +# ds4-main- so pinned commits stay reachable, and opens a PR here moving the submodule. +# A commit that no longer applies stops the bump for a hand rebase. GitHub-hosted CI has no +# GPU; run the checks in the PR body on Strix Halo before merging. # -# Uses the secret HRX_BUMP_TOKEN (Contents and Pull requests read/write on -# 1bit-MONSTER/engine): a PR opened with the default GITHUB_TOKEN would not -# start CI. +# Uses the secret HRX_BUMP_TOKEN (Contents read/write on 1bit-MONSTER/ds4 and +# 1bit-MONSTER/engine, Pull requests read/write on 1bit-MONSTER/engine): a PR opened with the +# default GITHUB_TOKEN would not start CI, and that token cannot push to the fork. name: bump-ds4 on: @@ -60,10 +63,51 @@ jobs: set -euo pipefail upstream=$(gh api repos/antirez/ds4/commits/main --jq .sha) ours=$(git ls-tree HEAD third_party/ds4 | awk '{print $3}') - echo "upstream $upstream, ours $ours" + # where our commits branch off antirez's history (the fork shares its object network) + ours_base=$(gh api "repos/antirez/ds4/compare/${upstream}...${ours}" --jq .merge_base_commit.sha) + echo "upstream ${upstream}, ours ${ours} on ${ours_base}" { - echo "upstream=$upstream"; echo "ours=$ours" - if [ "$upstream" = "$ours" ]; then echo "changed=false"; else echo "changed=true"; fi + echo "upstream=$upstream"; echo "ours=$ours"; echo "ours_base=$ours_base" + if [ "$upstream" = "$ours_base" ]; then echo "changed=false"; else echo "changed=true"; fi + } >> "$GITHUB_OUTPUT" + + - name: Rebase our commits onto upstream + id: fork + if: steps.pins.outputs.changed == 'true' + env: + GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + UPSTREAM: ${{ steps.pins.outputs.upstream }} + OURS: ${{ steps.pins.outputs.ours }} + OURS_BASE: ${{ steps.pins.outputs.ours_base }} + run: | + set -euo pipefail + # the fork's main mirrors antirez/ds4 main + gh api -X POST repos/1bit-MONSTER/ds4/merge-upstream -f branch=main > /dev/null + git init -q fork && cd fork + git remote add fork "https://x-access-token:${GH_TOKEN}@github.com/1bit-MONSTER/ds4.git" + git remote add upstream https://github.com/antirez/ds4.git + git fetch -q --filter=blob:none upstream "$UPSTREAM" + git fetch -q --filter=blob:none fork 1bit/main + carried=$(git rev-parse FETCH_HEAD) + if [ "$carried" != "$OURS" ]; then + echo "::error::1bit/main (${carried:0:12}) is not the engine's pin (${OURS:0:12}); reconcile them by hand" + exit 1 + fi + git fetch -q --filter=blob:none upstream "$OURS_BASE" + git checkout -q --detach "$carried" + if ! git -c user.name=ds4-bump -c user.email=ds4-bump@users.noreply.github.com \ + rebase -q --onto "$UPSTREAM" "$OURS_BASE"; then + git rebase --abort || true + echo "::error::our commits on 1bit/main do not rebase onto antirez/ds4 ${UPSTREAM:0:12}; rebase them by hand" + exit 1 + fi + rebased=$(git rev-parse HEAD) + git tag "ds4-main-${carried:0:12}" "$carried" + git push -q fork "refs/tags/ds4-main-${carried:0:12}" + git push -q --force-with-lease="refs/heads/1bit/main:$carried" fork "$rebased:refs/heads/1bit/main" + { + echo "rebased=$rebased" + echo "patches<> "$GITHUB_OUTPUT" - name: Open the bump PR @@ -72,26 +116,32 @@ jobs: GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} UPSTREAM: ${{ steps.pins.outputs.upstream }} OURS: ${{ steps.pins.outputs.ours }} + OURS_BASE: ${{ steps.pins.outputs.ours_base }} + REBASED: ${{ steps.fork.outputs.rebased }} + PATCHES: ${{ steps.fork.outputs.patches }} run: | set -euo pipefail branch="bump-ds4/${UPSTREAM:0:12}" if git ls-remote --exit-code origin "refs/heads/$branch" > /dev/null; then echo "$branch already exists"; exit 0 fi - log=$(gh api "repos/antirez/ds4/compare/${OURS}...${UPSTREAM}" \ + log=$(gh api "repos/antirez/ds4/compare/${OURS_BASE}...${UPSTREAM}" \ --jq '.commits[-30:][] | "- \(.sha[0:9]) \(.commit.message | split("\n")[0])"' || true) git config user.name "ds4-bump" git config user.email "ds4-bump@users.noreply.github.com" git switch -c "$branch" - git update-index --cacheinfo "160000,$UPSTREAM,third_party/ds4" + git update-index --cacheinfo "160000,$REBASED,third_party/ds4" git commit -q -m "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}" git push -q origin "$branch" gh pr create --base main --head "$branch" \ --title "Bump DwarfStar: antirez/ds4 ${UPSTREAM:0:12}" \ - --body "Moves third_party/ds4 from \`${OURS:0:12}\` to upstream main \`${UPSTREAM:0:12}\`. + --body "Moves third_party/ds4 (1bit-MONSTER/ds4 1bit/main) from \`${OURS:0:12}\` to \`${REBASED:0:12}\`: upstream main \`${UPSTREAM:0:12}\` plus the commits we carry. Upstream commits (last 30): ${log} + Our commits, rebased onto upstream (none left means upstream has them all): + ${PATCHES:-none} + CI here has no GPU. Before merging, on Strix Halo: - \`DS4_TEST=1 scripts/build-ds4.sh ~/.cache/ds4-pin rocm\` (builds, then runs DwarfStar's model-free routed-MoE GPU test), then \`1bit serve --device ds4 -m --ctx-size 8192\` answers a chat request (docs/dwarfstar.md)." + \`DS4_TEST=1 scripts/build-ds4.sh ~/.cache/ds4-pin rocm\` (builds, then runs DwarfStar's model-free routed-MoE GPU test), then \`1bit serve --device ds4 -m --ctx-size 8192\` and the same with its \`.1bp\` (\`gguf-tools/gguf_to_1bp.py\`) answer the same chat request (docs/dwarfstar.md)." diff --git a/.gitmodules b/.gitmodules index 7eb314bf..21e1eb28 100644 --- a/.gitmodules +++ b/.gitmodules @@ -81,10 +81,11 @@ [submodule "third_party/ryzenai-server"] path = third_party/ryzenai-server url = https://github.com/lemonade-sdk/ryzenai-server -# DwarfStar (docs/dwarfstar.md): upstream antirez/ds4, built by scripts/build-ds4.sh; -# .github/workflows/bump-ds4.yml keeps it current. +# DwarfStar (docs/dwarfstar.md): our fork 1bit-MONSTER/ds4, branch 1bit/main = upstream +# antirez/ds4 main plus our commits (1BP packages, docs/1BP.md there); built by +# scripts/build-ds4.sh; .github/workflows/bump-ds4.yml rebases our commits onto upstream. [submodule "third_party/ds4"] path = third_party/ds4 - url = https://github.com/antirez/ds4.git + url = https://github.com/1bit-MONSTER/ds4.git + branch = 1bit/main shallow = true - branch = main diff --git a/NOTICE b/NOTICE index cfbc95c9..d34f554d 100644 --- a/NOTICE +++ b/NOTICE @@ -84,9 +84,10 @@ ZINC (third_party/zinc) License: MIT DwarfStar (third_party/ds4) - https://github.com/antirez/ds4 + https://github.com/1bit-MONSTER/ds4 + (a fork of https://github.com/antirez/ds4, with 1BP packages) Copyright (c) 2026 The ds4.c authors - License: MIT + License: MIT; the fork's own files (1BP reader, converter and checker) Apache-2.0 Lemonade (third_party/lemonade) https://github.com/1bit-MONSTER/lemonade diff --git a/app/serve.cpp b/app/serve.cpp index cdca8bbb..e1b1e198 100644 --- a/app/serve.cpp +++ b/app/serve.cpp @@ -39,6 +39,7 @@ // long prompt prefixes on HRX0 over one // shared KV cache (docs/hrx.md) // a .gguf, --device zinc -> this build's zinc +// a .1bp (1BP v5 package), --device auto|ds4 -> the same ds4-server (our DwarfStar fork reads 1BP) // a DwarfStar .gguf, --device ds4 -> this build's DwarfStar ds4-server (DeepSeek // V4/V4.1 Flash, GLM 5.x, Qwen3.8-Flash-Next in // DwarfStar's own GGUFs; docs/dwarfstar.md) @@ -91,6 +92,7 @@ #include #include #include +#include #include #include #include @@ -200,6 +202,14 @@ bool hadamard_q4_0(const std::string& model) { gguf_int(model, "onebit.hadamard_q4_0") == 32; } +// A 1BP package (1bit-MONSTER's model format; our DwarfStar fork reads version 5, its +// docs/1BP.md): its first four bytes are "1BP\0". +bool onebp_file(const std::string& model) { + std::ifstream f(model, std::ios::binary); + char magic[4] = {}; + return f.read(magic, 4) && std::memcmp(magic, "1BP\0", 4) == 0; +} + bool fork_only_arch(const std::string& gguf) { // Zyphra ZAYA1 (docs/vulkan.md); OPT, CodeGen, GPT-Neo and GPT-J (llama.cpp #8: upstream has // no model for them, gptj only a name). Keep in step with FORK_ONLY in tools/registry_build.py. @@ -462,7 +472,7 @@ std::string model_id(const Options& o) { fs::path p(o.model); if (!p.has_filename()) p = p.parent_path(); // Only a .gguf loses its extension: "Qwen3-0.6B-4bit" is a name, not a stem. - return p.extension() == ".gguf" ? p.stem().string() : p.filename().string(); + return p.extension() == ".gguf" || p.extension() == ".1bp" ? p.stem().string() : p.filename().string(); } // Forwards an OpenAI POST to the child: the model id becomes the child's (a @@ -847,6 +857,14 @@ int serve_child(const Options& given) { throw std::runtime_error("--laya routes among devices; a Hadamard-rotated file runs on rocm only"); std::fprintf(stderr, "1bit serve: %s is Hadamard-rotated: lean ROCm route, W4A4 prompt processing\n", o.model.c_str()); } + if (onebp_file(o.model)) { + // DwarfStar (our fork) is the engine's 1BP reader + if (o.device == "auto") o.device = "ds4"; + if (o.device != "ds4") + throw std::runtime_error(o.model + " is a 1BP package: it runs on --device ds4 (docs/dwarfstar.md)"); + if (o.laya_auto || !o.laya_model.empty()) + throw std::runtime_error("--laya routes among devices; a 1BP package runs on ds4 only"); + } std::vector backends; // --laya / --laya-model with --device auto (RFC #186): Laya classifies each conversation // (code, prose, short, long_doc; laya/route.h) and the route policy (config/route-policy.json @@ -1462,8 +1480,8 @@ int run_serve(int argc, char** argv) { return run_forward_serve(int(av.size()), av.data()); } #endif - if (fs::path(o.model).extension() == ".gguf") return serve_child(o); - throw std::runtime_error(o.model + ": expected an NPU model directory or a .gguf file"); + if (fs::path(o.model).extension() == ".gguf" || onebp_file(o.model)) return serve_child(o); + throw std::runtime_error(o.model + ": expected an NPU model directory, a .gguf file or a 1BP package"); } } // namespace onebit diff --git a/docs/dwarfstar.md b/docs/dwarfstar.md index 15ea807a..e200d491 100644 --- a/docs/dwarfstar.md +++ b/docs/dwarfstar.md @@ -26,12 +26,17 @@ it loads its own GGUF layouts (`antirez/deepseek-v4-gguf`, llama.cpp cannot load those. `1bit serve --device ds4 -m ` runs its `ds4-server` as the model's -backend, behind the same OpenAI API as every other device. +backend, behind the same OpenAI API as every other device. The engine builds our fork, +[1bit-MONSTER/ds4](https://github.com/1bit-MONSTER/ds4), which also reads 1BP packages +([below](#1bp-packages)). ## Build -`third_party/ds4` pins upstream `antirez/ds4` main; `.github/workflows/bump-ds4.yml` -opens a PR when it moves. +`third_party/ds4` pins our fork's `1bit/main`: upstream `antirez/ds4` main plus the +commits we carry (1BP packages). When upstream main moves, `.github/workflows/bump-ds4.yml` +rebases our commits onto it, tags the old tip `ds4-main-` so the old pin stays +reachable, and opens a PR moving the submodule. A commit that no longer applies stops the bump +for a hand rebase. ``` git submodule update --init --depth 1 third_party/ds4 @@ -73,6 +78,33 @@ Memory: the resident DeepSeek V4 Flash Q2 needs about 81 GiB plus runtime buffer and the GPU must see that much (`amdgpu.gttsize` / `ttm.pages_limit`, DwarfStar's [STRIX_HALO.md](https://github.com/antirez/ds4/blob/main/docs/STRIX_HALO.md)). +## 1BP packages + +Our fork reads 1BP v5, 1bit-MONSTER's model package: one memory-mappable file with the 1BP +header and tensor index, the model's metadata (encoded as GGUF's key/value section) and +64-byte-aligned weights. The format is in +[docs/1BP.md](https://github.com/1bit-MONSTER/ds4/blob/1bit/main/docs/1BP.md) of the fork. +The converter copies a GGUF's metadata byte for byte and carries every tensor in its GGUF +block format, so a package runs exactly as its GGUF: + +``` +python third_party/ds4/gguf-tools/gguf_to_1bp.py hf://antirez/deepseek-v4-gguf@f71f23d5/DeepSeek-V4-Flash-IQ2XXS-w2Q2K-AProjQ8-SExpQ8-OutQ8-chat-v2-imatrix-0731.gguf DeepSeek-V4-Flash-Q2.1bp +1bit serve -m DeepSeek-V4-Flash-Q2.1bp --ctx-size 8192 +``` + +An `hf://` source is read with HTTP range requests, so the GGUF never has to be on disk. +`1bit serve` recognises a package by its first four bytes and serves it on `ds4` (with +`--device auto` too). `gguf-tools/check_1bp.py <1bp>` compares a package with its GGUF. + +Checked on Strix Halo: DeepSeek V4 Flash Q2 (1,328 tensors, 62 metadata keys, 80.76 GiB of +weights), converted from `hf://` at the revision above, gives the same answers and +completion-token counts as its GGUF for three prompts at temperature 0 (256 tokens each). +Qwen3-0.6B Q4_K_M's package passes `check_1bp.py`: all 310 tensors and 32 metadata keys +identical. + +1BP's own tile formats (Q4NX, TQ2 and the others the 1BP header defines) do not run in +DwarfStar yet: a package with one is refused with the tensor's name. + ## Verified (Strix Halo, ds4 `0aaea5a238fb`, TheRock ROCm) | Check | Result | diff --git a/third_party/ds4 b/third_party/ds4 index 0aaea5a2..a7f2e1ea 160000 --- a/third_party/ds4 +++ b/third_party/ds4 @@ -1 +1 @@ -Subproject commit 0aaea5a238fb41a35106a551e73c8409dfb751ac +Subproject commit a7f2e1eaf36c61e9e173fe6879923de01e1ee749