diff --git a/.github/workflows/bump-laya.yml b/.github/workflows/bump-laya.yml new file mode 100644 index 00000000..b06c2a1b --- /dev/null +++ b/.github/workflows/bump-laya.yml @@ -0,0 +1,100 @@ +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Keep Laya current (docs/laya.md): when NandhaKishorM/laya main or the +# convaiinnovations/laya model repo on Hugging Face moves, open a PR that moves +# third_party/laya and config/laya.json's revision together. Before merging, +# run scripts/fetch-laya.sh and the Laya tests against the new revision. +# +# Uses the secret HRX_BUMP_TOKEN (Contents and Pull requests read/write on +# 1bit-MONSTER/engine): a PR opened with the default GITHUB_TOKEN would not +# start CI. +name: bump-laya + +on: + schedule: + - cron: "53 7 * * *" + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: bump-laya + cancel-in-progress: false + +jobs: + bump: + runs-on: ubuntu-latest + steps: + - name: Require the token + env: + HRX_BUMP_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + run: | + if [ -z "$HRX_BUMP_TOKEN" ]; then + echo "::error::secret HRX_BUMP_TOKEN is not set (see the header of this workflow)" + exit 1 + fi + + - uses: actions/checkout@v4 + with: + token: ${{ secrets.HRX_BUMP_TOKEN }} + + - name: Compare with upstream + id: pins + run: | + set -euo pipefail + src=$(git ls-remote https://github.com/NandhaKishorM/laya.git refs/heads/main | awk '{print $1}') + repo=$(python3 -c 'import json; print(json.load(open("config/laya.json"))["repo"])') + rev=$(curl -fsSL "https://huggingface.co/api/models/${repo}" | python3 -c 'import json,sys; print(json.load(sys.stdin)["sha"])') + our_src=$(git ls-tree HEAD third_party/laya | awk '{print $3}') + our_rev=$(python3 -c 'import json; print(json.load(open("config/laya.json"))["revision"])') + echo "source $src (ours $our_src), model $rev (ours $our_rev)" + { + echo "src=$src"; echo "rev=$rev"; echo "our_src=$our_src"; echo "our_rev=$our_rev" + if [ "$src" = "$our_src" ] && [ "$rev" = "$our_rev" ]; then echo "changed=false"; else echo "changed=true"; fi + } >> "$GITHUB_OUTPUT" + + - name: Open the bump PR + if: steps.pins.outputs.changed == 'true' + env: + GH_TOKEN: ${{ secrets.HRX_BUMP_TOKEN }} + SRC: ${{ steps.pins.outputs.src }} + REV: ${{ steps.pins.outputs.rev }} + OUR_SRC: ${{ steps.pins.outputs.our_src }} + OUR_REV: ${{ steps.pins.outputs.our_rev }} + run: | + set -euo pipefail + branch="bump-laya/${SRC:0:8}-${REV:0:8}" + if git ls-remote --exit-code origin "refs/heads/$branch" > /dev/null; then + echo "$branch already exists"; exit 0 + fi + git config user.name "laya-bump" + git config user.email "laya-bump@users.noreply.github.com" + git switch -c "$branch" + git update-index --cacheinfo "160000,$SRC,third_party/laya" + python3 - "$REV" <<'PY' + import json, sys + p = json.load(open("config/laya.json")); p["revision"] = sys.argv[1] + open("config/laya.json", "w").write(json.dumps(p, indent=2) + "\n") + PY + git add config/laya.json + git commit -q -m "Bump Laya: source ${SRC:0:12}, model ${REV:0:12}" + git push -q origin "$branch" + gh pr create --base main --head "$branch" \ + --title "Bump Laya: source ${SRC:0:8}, model ${REV:0:8}" \ + --body "third_party/laya: \`${OUR_SRC:0:12}\` -> \`${SRC:0:12}\` (NandhaKishorM/laya main). config/laya.json revision: \`${OUR_REV:0:12}\` -> \`${REV:0:12}\` (convaiinnovations/laya). + + Before merging: scripts/fetch-laya.sh (verifies every file) and the Laya tests." diff --git a/.gitmodules b/.gitmodules index 30f743fa..8de1a8d2 100644 --- a/.gitmodules +++ b/.gitmodules @@ -41,3 +41,11 @@ path = third_party/linux url = https://github.com/torvalds/linux.git shallow = true +# Laya (docs/laya.md): the step-4 router's upstream source, NandhaKishorM/laya. +# The model checkpoints are pinned by Hugging Face revision in config/laya.json +# and fetched by scripts/fetch-laya.sh; .github/workflows/bump-laya.yml keeps both current. +[submodule "third_party/laya"] + path = third_party/laya + url = https://github.com/NandhaKishorM/laya.git + branch = main + shallow = true diff --git a/config/laya.json b/config/laya.json new file mode 100644 index 00000000..406fe090 --- /dev/null +++ b/config/laya.json @@ -0,0 +1,5 @@ +{ + "repo": "convaiinnovations/laya", + "revision": "aa8c91ca088ec597df95a0d1c76b3063cb2ae5e8", + "note": "Laya's three checkpoints (root, multilingual/, typed-decisions/) at one Hugging Face revision; scripts/fetch-laya.sh downloads exactly this revision and checks every file's hash against the Hub's file list." +} diff --git a/docs/PORTING.md b/docs/PORTING.md index 3076edf3..d855ccf9 100644 --- a/docs/PORTING.md +++ b/docs/PORTING.md @@ -25,7 +25,7 @@ Each step below is one PR (or a short series) that builds and runs on Strix Halo | 2 | **HRX with Vulkan.** llama.cpp with `GGML_HRX=ON` and `GGML_VULKAN=ON` in one build, on AMD's tested pair | `1bit-MONSTER/llama.cpp` `1bit/hrx-vulkan` (AMD's `hrx-graph-develop-v2`) + `ROCm/hrx-system`, pinned from `ROCm/ggml-staging-automation` | **landed** ([docs/hrx.md](hrx.md)): Lemonade serves the same checkpoint on `Vulkan0` and `HRX0`; kept current by `bump-hrx.yml` | | 3 | **NPU engine.** Full ELFs only, and the open 16-tile layer kernel built from source | `engine/npu` (`npu_engine_universal.cpp`, `I8Ctx::init_elf`); ELF dispatch table on `backup/iso-build-elf-native-2026-09-22`; kernel on `bench/fastlane-16tile-corrections-2026-09-22` | **3a–3c landed** ([docs/npu.md](npu.md)). 3a (#8): full ELFs generated in C++, reproducing all 5,153 captured contexts. 3b/3c (#9): the lane runtime (logits bit-identical to the reference lane, 24/24 steps, 11.0 ms/token), tokenizer, `1bit unified`, and Lemonade serving Qwen3-0.6B on the NPU through `onebit`. The XDNA driver and XRT are pinned upstream and built privately (#12). **Open:** the layer kernel and lm-head artifacts are not yet built from source (npu.md, "Open") | | + | **Linux kernel.** The kernel that provides `amdxdna` and `amdgpu`, pinned to upstream | `torvalds/linux` release tags; config from the Strix Halo kernel of 2026-09-23 | **pinned** ([docs/kernel.md](kernel.md)): v7.3-rc4 builds into Debian packages with `amdxdna` in-tree; kept current by `bump-linux.yml`. Installing it on Strix Halo is a separate, deliberate step | -| 4 | **Laya router.** A non-autoregressive scorer that picks where each request runs | `src/laya_scorer.cpp`, `include/laya_scorer.h` on `backup/laya-and-results-2026-09-22`; model at `~/models/laya` | matches its Python reference; routes requests | +| 4 | **Laya router.** A non-autoregressive scorer that picks where each request runs | `src/laya_scorer.cpp`, `include/laya_scorer.h` on `backup/laya-and-results-2026-09-22`; model at `~/models/laya` | **pinned** ([docs/laya.md](laya.md)): source `NandhaKishorM/laya` + the three HF checkpoints at one revision, hash-verified fetch, `bump-laya.yml`. Next: the C++ scorer, gated against the Python reference; then routing | | 5 | **Every HF model, kept current.** The architecture registry (569 tokens mapping 2,030 HF arch strings) and the daily HF census that finds new architectures and proposes mappings | `src/model_registry*.cpp`, `Testing/census_*.py` and `.json`, `.github/workflows/census-{watch,sweep,autopr}.yml` | the census runs daily in CI; docs report *mapped* and *run and checked* counts separately | | 6 | **ZINC (NVIDIA and more).** Upstream `zolotukhin/zinc`, a Zig GGUF engine with Vulkan, ROCm, CUDA and Metal backends; its CUDA backend reaches NVIDIA GPUs (Ada `sm_89`, Blackwell `sm_120`) | not in 1bit-MONSTER; pinned from upstream `main` | **pinned** ([docs/zinc.md](zinc.md)): `scripts/build-zinc.sh` builds it privately; the Vulkan build gives 12095 (" Paris") at 295 tok/s on Strix Halo; the CUDA build answers " Paris." at 167–173 tok/s on an RTX 5090 (Qwen3.5-9B); kept current by `bump-zinc.yml`. `1bit lemonade` serves it as the `zinc` recipe (`-DONEBIT_ZINC=ON`, e2e passes) | diff --git a/docs/laya.md b/docs/laya.md new file mode 100644 index 00000000..ee4d831c --- /dev/null +++ b/docs/laya.md @@ -0,0 +1,49 @@ + +# Laya + +Step 4 of the port (docs/PORTING.md): Laya picks where each request runs. +[Laya](https://github.com/NandhaKishorM/laya) (Apache-2.0) is a +non-autoregressive decision model: a ModernBERT-style encoder plus an RLCD +decision head. It answers typed questions (`choice`, `score`, `noul`) about a +request in one forward pass, with calibrated confidence. It has three +checkpoints and a router that picks between them. + +## The pin + +| | Pin | Kept current by | +|---|---|---| +| Source | `third_party/laya` = `NandhaKishorM/laya` main **`1e28ac20`** | `bump-laya.yml` | +| Checkpoints | `config/laya.json` = Hugging Face `convaiinnovations/laya` at revision **`aa8c91ca`**: root (842 MB), `multilingual/` (644 MB, 34 MB tokenizer), `typed-decisions/` (842 MB) | `bump-laya.yml` (moves the source and the revision together) | + +```sh +scripts/fetch-laya.sh ~/models/laya-pinned +``` + +`fetch-laya.sh` downloads exactly the pinned revision and checks every file +against the Hub's list: sha256 for the LFS weights, the git blob id for the rest. +Files already present and correct are not fetched again. Images and eval plots +are skipped. + +Verified 2026-09-23 on Strix Halo: 20 files, 2.3 GB, all hashes match. + +## Next + +The C++ scorer (1bit-MONSTER `src/laya_scorer.cpp`, on +`backup/laya-and-results-2026-09-22`) is ported next. It is gated against +the Python reference in `third_party/laya` on the pinned checkpoints, then +wired into the router that picks NPU, HRX, Vulkan or ZINC per request. diff --git a/scripts/fetch-laya.sh b/scripts/fetch-laya.sh new file mode 100755 index 00000000..82947854 --- /dev/null +++ b/scripts/fetch-laya.sh @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Copyright 2026 bong-water-water-bong +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# fetch-laya.sh +# +# Downloads the Laya checkpoints pinned in config/laya.json (one Hugging Face +# revision holding all three: root, multilingual/, typed-decisions/) into +# , and checks every file against the Hub's list for that revision: +# sha256 for LFS files (the weights), the git blob id for the small ones. +# Images and the eval plots are skipped. Files already present and correct are +# not downloaded again. +set -euo pipefail +dir=${1:?usage: fetch-laya.sh } +root=$(cd "$(dirname "$0")/.." && pwd) +mkdir -p "$dir" +python3 - "$root/config/laya.json" "$dir" <<'PYEOF' +import hashlib, json, os, sys, urllib.request + +pin = json.load(open(sys.argv[1])) +repo, rev, out = pin["repo"], pin["revision"], sys.argv[2] +api = f"https://huggingface.co/api/models/{repo}/tree/{rev}?recursive=true" +files = [f for f in json.load(urllib.request.urlopen(api)) + if f["type"] == "file" and not f["path"].startswith(("assets/", "eval/"))] + +def digest(path, lfs): + data_h = hashlib.sha256() if lfs else hashlib.sha1() + if not lfs: # git blob id = sha1("blob \0" + contents) + data_h.update(f"blob {os.path.getsize(path)}\0".encode()) + with open(path, "rb") as fh: + for chunk in iter(lambda: fh.read(1 << 20), b""): + data_h.update(chunk) + return data_h.hexdigest() + +for f in files: + lfs = f.get("lfs") + want = lfs["oid"] if lfs else f["oid"] + dst = os.path.join(out, f["path"]) + if os.path.exists(dst) and digest(dst, bool(lfs)) == want: + continue + os.makedirs(os.path.dirname(dst) or ".", exist_ok=True) + url = f"https://huggingface.co/{repo}/resolve/{rev}/{f['path']}" + urllib.request.urlretrieve(url, dst + ".part") + got = digest(dst + ".part", bool(lfs)) + if got != want: + os.remove(dst + ".part") + sys.exit(f"{f['path']}: hash {got} != pinned {want}") + os.replace(dst + ".part", dst) + print(f"fetched {f['path']} ({f['size']} B)") +print(f"Laya {repo}@{rev[:12]}: {len(files)} files verified in {out}") +PYEOF diff --git a/third_party/laya b/third_party/laya new file mode 160000 index 00000000..1e28ac20 --- /dev/null +++ b/third_party/laya @@ -0,0 +1 @@ +Subproject commit 1e28ac20c0896b1c37a744cd11f740eb98f8b178