diff --git a/.github/build_cuda_linux.sh b/.github/build_cuda_linux.sh index de755466e..c988c31b3 100755 --- a/.github/build_cuda_linux.sh +++ b/.github/build_cuda_linux.sh @@ -37,4 +37,7 @@ case "${CUDA_FAST_BUILD:-}" in ;; esac -exec .github/build.sh $@ -DGGML_CUDA=1 -DCMAKE_CUDA_COMPILER=/usr/local/cuda-13.4/bin/nvcc $CUDA_ARCH_ARGS +# CCCL v3.4.3 is fetched instead of the one CUDA 13.4 bundles, as upstream's own CUDA release builds do +# (llama.cpp #29792): ggml's CUB DeviceTopK path needs >= 3.4.3 (an earlier race, NVIDIA/cccl#10627) and +# falls back to a sort below it. Drop the flag once the toolkit is 13.5+, which bundles CCCL 3.5. +exec .github/build.sh $@ -DGGML_CUDA=1 -DCMAKE_CUDA_COMPILER=/usr/local/cuda-13.4/bin/nvcc -DGGML_CUDA_CCCL_VERSION=v3.4.3 $CUDA_ARCH_ARGS diff --git a/.github/buildcheck/tests/test_natives.py b/.github/buildcheck/tests/test_natives.py index 1ba0b4a8f..d1194015d 100644 --- a/.github/buildcheck/tests/test_natives.py +++ b/.github/buildcheck/tests/test_natives.py @@ -225,7 +225,8 @@ def test_cli_prints_the_pom_executions_and_the_targets(self): out = subprocess.run([sys.executable, cli, "fatjar-targets"], capture_output=True, text=True, check=True) self.assertEqual(out.stdout.split(), ["linux-aarch64", "linux-x86-64", "windows-aarch64", "windows-x86-64"]) out = subprocess.run([sys.executable, cli, "pom"], capture_output=True, text=True, check=True) - self.assertEqual(out.stdout.count(""), 26) + rows = natives.rows(natives.read(REPO, ".github/natives.csv")) + self.assertEqual(out.stdout.count(""), len(rows)) self.assertEqual(subprocess.run([sys.executable, cli, "nonsense"], capture_output=True).returncode, 2) diff --git a/.github/natives.csv b/.github/natives.csv index dc667112e..796642e58 100644 --- a/.github/natives.csv +++ b/.github/natives.csv @@ -29,6 +29,7 @@ cuda13-windows-x86-64,Windows/x86_64/cuda13,jllama.dll,no vulkan-linux-x86-64,Linux/x86_64/vulkan,libjllama.so,no vulkan-linux-aarch64,Linux/aarch64/vulkan,libjllama.so,no vulkan-windows-x86-64,Windows/x86_64/vulkan,jllama.dll,no +vulkan-windows-aarch64,Windows/aarch64/vulkan,jllama.dll,no opencl-android-aarch64,Linux-Android/aarch64/opencl,libjllama.so,no opencl-windows-x86-64,Windows/x86_64/opencl,jllama.dll,no opencl-windows-aarch64,Windows/aarch64/opencl,jllama.dll,no diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 6d2505c25..6ad4e21dc 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -1652,8 +1652,10 @@ jobs: # suite is CPU-only and fully covered by the `C++ Tests` job + the CPU Windows # jobs; a GPU-linked jllama_test.exe cannot be discovered/run on a GPU-less # GitHub runner (it errors probing for a CUDA device -> ctest *_NOT_BUILT). + # GGML_CUDA_CCCL_VERSION: fetch CCCL v3.4.3 like upstream's windows-cuda release job and the + # Linux CUDA build (see build_cuda_linux.sh) -- DeviceTopK needs it; drop it with CUDA 13.5+. run: | - .github\build.bat -G "Ninja Multi-Config" -DGGML_CUDA=ON -DOS_NAME=Windows -DOS_ARCH=x86_64 + .github\build.bat -G "Ninja Multi-Config" -DGGML_CUDA=ON -DGGML_CUDA_CCCL_VERSION=v3.4.3 -DOS_NAME=Windows -DOS_ARCH=x86_64 - name: Upload artifacts uses: actions/upload-artifact@v7 with: @@ -2044,6 +2046,57 @@ jobs: path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ if-no-files-found: error + build-windows-arm64-vulkan: + name: Build Windows 11 arm64 Vulkan + needs: [startgate, build-webui] + # Windows-on-ARM Vulkan (Snapdragon X / any Vulkan 1.2+ driver), the natives jar + # vulkan-windows-aarch64 -- the counterpart of upstream's windows arm64 Vulkan release + # (llama.cpp #29954). Same clang-cl + GGML_OPENMP=OFF toolchain as the arm64 CPU and + # OpenCL jobs (ggml refuses MSVC cl.exe on ARM). The SDK is installed the way upstream's + # release job does it: LunarG's (x64) installer with the com.lunarg.vulkan.arm64 component, + # which adds the arm64 import library; its x64 glslc runs under the runner's x64 + # emulation. Build-only like every GPU job (no GPU on the runner). + runs-on: windows-11-arm + env: + SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} + VULKAN_VERSION: 1.4.357.0 + steps: + - uses: actions/checkout@v7 + - name: Download shared WebUI assets + uses: actions/download-artifact@v8 + with: + name: webui-generated + path: ${{ github.workspace }}/llama/webui-generated/ + - name: Set up MSVC developer environment (arm64) + uses: ilammy/msvc-dev-cmd@v1 + with: + arch: arm64 + - name: Install Vulkan SDK (with the arm64 component) + shell: pwsh + # Version and installer arguments as in upstream's release.yml (windows, vulkan arm64). + run: | + curl.exe -fSL -o "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" "https://sdk.lunarg.com/sdk/download/$env:VULKAN_VERSION/windows/vulkansdk-windows-X64-$env:VULKAN_VERSION.exe" + & "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install com.lunarg.vulkan.arm64 + if (-not (Test-Path "C:\VulkanSDK\$env:VULKAN_VERSION")) { throw "Vulkan SDK $env:VULKAN_VERSION was not installed" } + Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\$env:VULKAN_VERSION" + Add-Content $env:GITHUB_PATH "C:\VulkanSDK\$env:VULKAN_VERSION\bin" + - name: Install sccache (shared compiler cache) + if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' + continue-on-error: true + uses: ./.github/actions/install-sccache-windows + - name: Build libraries + shell: cmd + # Build the artifact only (see the CUDA job's note: GPU-less runner can't run a + # GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs). + run: | + .github\build.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DGGML_VULKAN=ON -DOS_NAME=Windows -DOS_ARCH=aarch64 + - name: Upload artifacts + uses: actions/upload-artifact@v7 + with: + name: natives-vulkan-windows-aarch64 + path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ + if-no-files-found: error + build-linux-x86_64-openvino: name: Build Linux x86_64 OpenVINO (Intel) needs: [startgate, build-webui] @@ -2061,7 +2114,7 @@ jobs: with: distribution: 'temurin' java-version-file: .java-version - - name: Install OpenCL dev + Intel OpenVINO 2026.4 (archive) + - name: Install OpenCL dev + Intel OpenVINO 2026.4.1 (archive) run: | # Intel's OpenVINO APT repo only publishes up to ~2025 (the /openvino/2026 path 404s), and # 2025.x has the older ov::Allocator API that breaks ggml-openvino's template compile. So use @@ -2072,11 +2125,11 @@ jobs: # OPENVINO_VERSION_FULL (.github/workflows/release.yml at the pinned GIT_TAG); ggml-openvino # is developed against that pair, so lagging it is what eventually breaks the compile. Both # OpenVINO jobs here (Linux + Windows) use the same two values — bump them together: - # major = 2026.4 full = 2026.4.0.22959.99c81491cc3 + # major = 2026.4.1 full = 2026.4.1.22982.07f9c262b05 # OpenCL headers (incl. the C++ CL/cl2.hpp via opencl-clhpp-headers) come from Ubuntu's own repos. sudo apt-get update sudo apt-get install -y ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd - url="https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/linux/openvino_toolkit_ubuntu24_2026.4.0.22959.99c81491cc3_x86_64.tgz" + url="https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4.1/linux/openvino_toolkit_ubuntu24_2026.4.1.22982.07f9c262b05_x86_64.tgz" sudo mkdir -p /opt/intel/openvino curl -fSL "$url" | sudo tar -xz --strip-components=1 -C /opt/intel/openvino echo "OpenVINO_DIR=/opt/intel/openvino/runtime/cmake" >> "$GITHUB_ENV" @@ -2110,16 +2163,16 @@ jobs: uses: ilammy/msvc-dev-cmd@v1 with: arch: x64 - - name: Install OpenCL headers (vcpkg) + Intel OpenVINO 2026.4 + - name: Install OpenCL headers (vcpkg) + Intel OpenVINO 2026.4.1 shell: pwsh # vcpkg's opencl port ships the full C++ headers incl. CL/cl2.hpp that OpenVINO's # ocl_wrapper.hpp needs (the Khronos OpenCL-Headers dropped cl2.hpp) — same as upstream - # llama.cpp's windows-openvino job. OpenVINO 2026.4 matches ggml-openvino's target API. + # llama.cpp's windows-openvino job. OpenVINO 2026.4.1 matches ggml-openvino's target API. # Keep the version in sync with the Linux OpenVINO job above (and with upstream's # OPENVINO_VERSION_MAJOR / OPENVINO_VERSION_FULL) — see the note there. run: | C:\vcpkg\vcpkg install opencl:x64-windows - $url = "https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/windows/openvino_toolkit_windows_2026.4.0.22959.99c81491cc3_x86_64.zip" + $url = "https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4.1/windows/openvino_toolkit_windows_2026.4.1.22982.07f9c262b05_x86_64.zip" Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\openvino.zip" Expand-Archive -Path "$env:RUNNER_TEMP\openvino.zip" -DestinationPath "C:\openvino" -Force # The archive extracts into a nested versioned folder; point OpenVINO_DIR at its runtime/cmake. @@ -2402,6 +2455,7 @@ jobs: - build-linux-x86_64-sycl-fp32 - build-windows-x86_64-sycl - build-windows-arm64-opencl + - build-windows-arm64-vulkan - build-linux-x86_64-openvino - build-windows-x86_64-openvino - test-cpp-linux-x86_64 diff --git a/CHANGELOG.md b/CHANGELOG.md index 85f36b785..abb6786bd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,33 @@ from version 5.0.0 onward. Pre-fork releases (`1.x`–`4.2.0`) were authored by ## [Unreleased] ### Added +- **Kolibri-1 support** (Aleph Alpha, architecture `kolibri1`, 78B German/English reasoning MoE) ahead of upstream + llama.cpp ([ggml-org/llama.cpp#29922](https://github.com/ggml-org/llama.cpp/issues/29922)), as the carried patch + `0016-model-kolibri1.patch`. It combines the two community ports and, unlike either of them, loads the GGUFs of + both community converters. Guarded by `test_kolibri1.cpp`, which compares tiny random models with an + independent reference written from Aleph Alpha's vLLM implementation. The patch is dropped once upstream adds + the architecture. +- **`GpuSplitMode.TENSOR`** (`--split-mode tensor`, tensor parallelism, EXPERIMENTAL upstream). The mode + existed upstream before; since llama.cpp b11450 (#26610) it also works across RPC servers. +- **Input/output modalities on `RouterModel` and `ModelMeta`** (llama.cpp b11429, #29987): + `getInputModalities()`, `getOutputModalities()` and `isDecisionModel()`, from upstream's new + `architecture` object of `GET /models` (the router computes it offline, so a decision model is + recognisable before its first load) and, for a loaded `LlamaModel`, from the same metadata in + `getModelMeta()`. Empty against a server before b11429. +- **`vulkan-windows-aarch64` natives jar** (Windows on ARM with a Vulkan 1.2+ driver), following + upstream's new Windows arm64 Vulkan release (llama.cpp b11395, #29954). Built natively on + `windows-11-arm` with `clang-cl`; also in the `all-windows-aarch64` fat jar, where the loader tries it + before OpenCL and the CPU. +- **`ModelParameters.setDraftSampling(DraftSampling)`** (`--spec-draft-sampling`, llama.cpp b11368): + `PROBABILISTIC` samples the speculative draft and has the target verify it by rejection sampling, + which accepts more drafted tokens at a temperature above zero; `GREEDY` is upstream's default. Applies + to a draft model and to a model's own MTP heads. +- **Decision models: `LlamaModel.handleSystemOne(String)`**, llama.cpp's TypeSafe-compatible + `/v1/systemone` API (upstream b11361): typed `choice` / `score` / `noul` questions about a state, + answered with probabilities in one forward pass, for the decision models upstream supports (laya, + julia-1, lev, openjev, kev, ...). The JNI method forwards to upstream's own route handler, so the + request and response are exactly the HTTP endpoint's; `NativeServer` serves `POST /v1/systemone` in + classic and attach mode. A model that is not a decision model throws a `LlamaException`. - **`net.ladenthin:llama-atmosphere-agent` on Maven Central**, at the core's version: the agent's thin jar (with `Main-Class`), sources and javadoc, published right after the reactor. Its pom names `llama-platform` as a runtime dependency, so `jbang net.ladenthin:llama-atmosphere-agent:` @@ -18,6 +45,23 @@ from version 5.0.0 onward. Pre-fork releases (`1.x`–`4.2.0`) were authored by core's; `check-natives.py` fails when they differ. ### Changed +- **Upgraded the pinned llama.cpp from b11320 to b11457**, in 21 reviewed steps, each ending at a tag. + Every carried patch that broke was traced to the one upstream commit that broke it, and the step + containing that commit ends at the first tag after it: `0007` at #29818 (b11361) and #29895 (b11401), `0014` at #29895, `0008` + at #29987 (the commit just before b11429), `0015` at #26610 (b11450). Each refresh moved context only, + and all nine patches are still needed. The new + upstream features this binding now exposes are listed under *Added*; the build follows upstream's + CUDA CCCL pin (v3.4.3) and OpenVINO 2026.4.1. Per-step record: + `docs/history/llama-cpp-breaking-changes.md`. +- **RPC protocol 8** (llama.cpp b11450, #26610): `RPC_PROTO_MAJOR_VERSION` 7 → 8. An `RpcServer` or + `--rpc` client of this release talks only to RPC peers of the same protocol -- upgrade the + `rpc-server`s and every JVM using `RpcServer` together. +- **Slot state files from earlier releases no longer restore** (llama.cpp b11411, #28498): upstream + now stores the exact KV-cache rotation in a state file and rejects one restored under a mismatched + rotation, which bumps `LLAMA_SESSION_VERSION` 10 → 11 and `LLAMA_STATE_SEQ_VERSION` 3 → 4. A file + written by `LlamaModel.saveSlot` (or the server's `/slots/{id}?action=save`) with an earlier jar is + rejected by `restoreSlot` with upstream's generic "invalid slot save file" message; regenerate it. + `Session` snapshots taken and restored within one process are not affected. - **`ProcessRunner` rewritten on `ProcessBuilder`** (the helper `OSInfo` runs `uname` with): the timeout is now real -- a command that does not end in time is killed and reported as an `IOException`, where the old timeout overload ignored the result of `waitFor` and then blocked reading the output -- and the diff --git a/CLAUDE.md b/CLAUDE.md index 83028cdfd..0690fa7d3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,13 +6,13 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co Java bindings for [llama.cpp](https://github.com/ggerganov/llama.cpp) via JNI, providing a high-level API for LLM inference in Java. The Java layer communicates with a native C++ library through JNI. -Current llama.cpp pinned version: **b11320** +Current llama.cpp pinned version: **b11457** ## Natives jars: one directory per backend (`.github/natives.csv`) `net.ladenthin:llama` is the **Java classes only**. Every native build ships as its own jar of the same artifact, classifier `--`, holding exactly one directory -`net/ladenthin/llama////` (26 today: `cpu-*` for 8 platforms, `metal-macos-aarch64`, +`net/ladenthin/llama////` (27 today: `cpu-*` for 8 platforms, `metal-macos-aarch64`, `msvc-windows-*`, and the GPU backends). Because the directories never overlap, **any combination of natives jars can share one classpath**, and `LlamaLoader` tries the backends it finds in a fixed order (`BACKEND_PRIORITY`: cuda13, rocm, sycl-fp16, sycl-fp32, sycl, vulkan, opencl, openvino, metal, @@ -44,7 +44,7 @@ it or is checked against it: | `.github/merge-native-artifacts.sh` | reads the list: every listed `natives-*` artifact present, no other, each holding its library and nothing outside its directory, no path claimed twice; writes `jllama-extras.txt` (sibling files loaded before the library, e.g. OpenVINO's `OpenCL.dll` on Windows) | | `.github/package-fatjars.sh` | reads the list: the built natives jars match it, each holds only its directory and the right `Automatic-Module-Name`; merges the all-backends fat jars and fails unless it produced exactly the targets `check-natives.py fatjar-targets` derives | | `.github/verify-native-deps.py` | exact dependency allowlist per CPU directory (`cpu`/`metal`/`msvc`) and for the Android OpenCL build (bionic + `libOpenCL.so`), denylist for the other GPU ones; every Android library 16 KB page-aligned (Google Play). Runs in `package` and on the staged AAR libraries | -| `.github/smoke-natives-jars.sh` (`package` job) | loads the real jars: classes + all 26 natives jars at once, on the classpath **and** the module path (on the GPU-less runner normally ending at `cpu`) | +| `.github/smoke-natives-jars.sh` (`package` job) | loads the real jars: classes + all 27 natives jars at once, on the classpath **and** the module path (on the GPU-less runner normally ending at `cpu`) | **Adding a natives jar:** a row in `natives.csv`, the execution `check-natives.py pom` prints, a build job uploading `natives-`, the backend name in CMake and `BACKEND_PRIORITY` if new, a @@ -53,7 +53,7 @@ targets follow by themselves (a new OS/arch with a GPU backend also needs its sm `check-natives.py` then demands). The checks are Python in `.github/buildcheck/` with unit tests (see "Build checks, shared files and the release gate"). -**Why the build jobs are not spawned from the list as one matrix** (considered and rejected): the 26 +**Why the build jobs are not spawned from the list as one matrix** (considered and rejected): the 27 builds use genuinely different toolchains — dockcross images, the CUDA redist archives, ROCm pip wheels, oneAPI, OpenVINO, `clang-cl` on arm64, qemu for s390x, three macOS variants — so a single matrix job would be a web of `if:` conditions; and `needs:` on a matrix waits for every entry, so @@ -209,7 +209,12 @@ To change the CUDA version, update the following places: (`JLLAMA_BACKEND`), `.github/natives.csv` (the two `cuda13-*` rows), the generated pom executions (`check-natives.py pom`), the build jobs' artifact names, and `LlamaLoader.BACKEND_PRIORITY`. `check-natives.py` fails until they agree. No change for a minor bump. -4. **`CLAUDE.md`** — the "Current CUDA version" line above. +4. **CCCL pin** — both CUDA builds pass `-DGGML_CUDA_CCCL_VERSION=v3.4.3` (`build_cuda_linux.sh` and the + Windows CUDA job), as upstream's own CUDA release jobs do since llama.cpp #29792: ggml's CUB + `DeviceTopK` path needs CCCL >= 3.4.3 and falls back to a sort below it, and CUDA 13.4 bundles an + older 3.4. **Drop both flags once the toolkit is 13.5 or newer** (it bundles CCCL 3.5); follow + upstream's `release.yml` matrix comment, which says the same. +5. **`CLAUDE.md`** — the "Current CUDA version" line above. Available CUDA versions for RHEL8/Manylinux_2_28 can be browsed at: ``` @@ -419,9 +424,9 @@ ctest --test-dir build --output-on-failure .github\build_opencl_windows.bat -G "Ninja Multi-Config" -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=x86_64 ``` -## Linux Vulkan natives + Windows arm64 CPU +## Linux Vulkan natives + Windows arm64 CPU and Vulkan -Three natives jars that extend the matrix toward upstream llama.cpp's release set. +Four natives jars that extend the matrix toward upstream llama.cpp's release set. **Linux Vulkan (`vulkan-linux-x86-64` + `vulkan-linux-aarch64`).** A vendor-neutral GPU jar for Linux (NVIDIA / AMD / Intel) with no CUDA toolkit. The build jobs are `build-linux-x86_64-vulkan` @@ -433,6 +438,15 @@ GPU-less runner — same as the Windows GPU jobs). Glibc floor rises to the ubun aarch64 CPU jar); acceptable for a GPU artifact. GPU runtime `libvulkan.so.1` is supplied by the consumer's driver — nothing is bundled (same policy as every GPU backend). +**Windows arm64 Vulkan (`vulkan-windows-aarch64`).** Added at the llama.cpp b11395 bump, when upstream +put a Windows arm64 Vulkan build into its release set (#29954). `build-windows-arm64-vulkan` uses the +arm64 CPU job's toolchain below (`windows-11-arm`, `clang-cl`, `GGML_OPENMP=OFF`) and installs the SDK +exactly as upstream's release job does: LunarG's x64 installer with the `com.lunarg.vulkan.arm64` +component (the arm64 import library), whose x64 `glslc` runs under the runner's x64 emulation; the SDK +version is upstream's `VULKAN_VERSION`, not the one the x86-64 Vulkan job pins through +`jakoch/install-vulkan-sdk-action`. Build-only like every GPU job, and part of the `all-windows-aarch64` +fat jar (tried before `opencl`, then `cpu`), so the `windows-aarch64` row of `smoke-fatjar` launches it. + **Windows arm64 CPU (`cpu-windows-aarch64`, in `llama-platform`).** `build-windows-arm64` runs natively on GitHub's free `windows-11-arm` runner (`ilammy/msvc-dev-cmd` `arch: arm64`, Ninja Multi-Config, `-DOS_ARCH=aarch64`, build + `ctest`) and writes `Windows/aarch64/cpu/`. No Java change @@ -690,7 +704,7 @@ needs no extra step here, `build-webui` re-reads the tag and rebuilds the matchi ships no UI): ```bash # needs node/npm + network for the asset build; the embed step is plain cmake -P -git clone --depth 1 --branch b11320 https://github.com/ggml-org/llama.cpp /tmp/lc +git clone --depth 1 --branch b11457 https://github.com/ggml-org/llama.cpp /tmp/lc ( cd /tmp/lc/tools/ui && npm ci && npm run build ) mkdir -p webui-generated /tmp/ui-gen cmake -DUI_SOURCE_DIR=/tmp/lc/tools/ui -DUI_BINARY_DIR=/tmp/ui-gen \ @@ -730,7 +744,7 @@ cache lives in **Depot Cache** over sccache's **WebDAV** backend: - `SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}` — a Depot **organization** token, stored as the repo secret **`DEPOT_TOKEN`**. -Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11320`), the +Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11457`), the ~280 upstream object files are byte-identical every run, so a warm cache recompiles only the *changed* files. Depot's cache is **shared across all branches** (unlike GitHub's per-branch `actions/cache`), so every branch builds incrementally; a `b` version bump @@ -934,12 +948,13 @@ Current patches: | `0001-win32-arg-parse-embed-guard.patch` | Windows JNI regression from llama.cpp **#24779** (introduced b9739): on Windows `common_params_parse` re-derived argv from the **process** command line (`GetCommandLineW`) and adopted it, so an embedded/JNI caller (`java.exe`) lost its `--model …` args → "Failed to parse model parameters". b9789 narrowed the unconditional override to a **count-guard** (`if (static_cast(utf8.buf.size()) == argc) { argv = utf8.ptrs.data(); }`), but that is exactly the variant the project already found breaks its Windows server-integration tests (when the embedded argv length coincides with `java.exe`'s). The patch carries the **complete upstream change** (so it can be submitted to llama.cpp verbatim and then dropped here): **(1)** `common_params_parse` parses **exactly the argv it is given** (no `GetCommandLineW` magic) and a new `common_params_parse_main()` wrapper holds the UTF-8 recovery for the standalone tools' `main()` (`common/arg.{cpp,h}`); **(2)** the **~34 standalone `main()` call sites** (every `common_params_parse(argc, argv, …)` across `tools/*`, `examples/*` and the `tests/*` programs) flip to `common_params_parse_main()`; **(3)** a `tests/test-arg-parser.cpp` regression case pins that `common_params_parse` honors a caller-supplied argv. The embedded caller (`jllama.cpp`) keeps calling `common_params_parse` and is never overridden. **Our subproject build compiles only the `arg.{cpp,h}` core** — `LLAMA_BUILD_TOOLS`/`LLAMA_BUILD_TESTS` are OFF for a FetchContent subproject — so the flips + test are applied-but-not-compiled here; they were validated via a one-off `-DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_TESTS=ON` build (the new test compiles and its asserts pass; `test-arg-parser`'s only red there is the live `ggml.ai` download check, which is sandbox-network, not the patch). Because it spans **35 files** it must be refreshed on every llama.cpp bump (the applier fails loud). **Refreshed at the b10679 bump:** upstream rewrote `tests/test-save-load-state.cpp`'s `main()` to take a `--models DIR` option, which it strips itself into a `filtered_argv` before calling `common_params_parse(fargc, filtered_argv.data(), …)`. That call site therefore stopped qualifying for the `_main()` flip — by this patch's own rule a caller that builds its own argv must use `common_params_parse` directly, so its argv is kept — and the hunk was **dropped** rather than refreshed (37 → 36 files). Caveat for whoever submits this upstream: that `main()` now filters a possibly-mojibake Windows argv *before* any UTF-8 recovery, so the fully correct upstream form there is recover-then-filter, not a one-line flip. It is out of scope for the downstream carry because `LLAMA_BUILD_TESTS` is OFF here, so the file is never compiled. **Refreshed the same way at the b11236 bump:** upstream #29426 rewrote `tests/test-recurrent-state-rollback.cpp`'s `main()` into the identical `--models DIR` / `filtered_argv` shape, so that hunk was dropped too (36 → 35 files); the same recover-then-filter caveat applies to it. **Still required at b10679, verified rather than assumed:** `common_params_parse` in pristine `b10679:common/arg.cpp` still carries the `#ifdef _WIN32` count-guarded `argv = utf8.ptrs.data()` override, and `common_params_parse_main` appears nowhere in `b10679:common/arg.h` — upstream has not adopted the fix. The upstream-facing write-up, including a standalone reproducer that makes llama.cpp's own `test-arg-parser` fail on unmodified `master`, lives in [docs/upstream-investigation-win32-argv-substitution.md](docs/upstream-investigation-win32-argv-substitution.md). **Reported upstream as [ggml-org/llama.cpp#26416](https://github.com/ggml-org/llama.cpp/issues/26416)** (2026-08-01, label `bug-unconfirmed`, first bad commit `508a475`); the issue asks which of the two directions the maintainers prefer before a PR is opened, so this patch stays downstream until they answer. | | `0002-server-preserve-caller-load-progress-callback.patch` | Load-progress-callback regression introduced in llama.cpp **b9789**: `server_context::load_model` (`tools/server/server-context.cpp`) now **unconditionally** installs the server's own load-progress reporter on `params_base.load_progress_callback` immediately before `common_init_from_params`, clobbering any callback the embedding caller already set. libjllama's `LoadProgressCallback` feature wires `common_params.load_progress_callback` to a JNI trampoline *before* calling `load_model`, so the bump silently killed it — `LoadProgressCallbackTest` saw zero progress updates and the abort-on-`false` path never threw. The patch guards the assignment with `if (params_base.load_progress_callback == nullptr)`, so the server installs its own reporter **only when the caller hasn't** — a caller-supplied callback survives and fires during load. Standalone `llama-server` (no caller callback, so the field is null) is unaffected. Same JNI-vs-standalone divergence class as `0001`. **The guard is `== nullptr || == load_progress_callback`, and the second disjunct must never be dropped:** `load_progress_text` is a **local** of `load_model()`, and upstream re-assigns both fields on every call so the `user_data` always points at the current frame. `load_model()` runs a **second** time when resuming from the sleeping state (`--sleep-idle-seconds`), and by then `params_base` holds *our own* callback from the first load — a bare nullptr check skips the re-assignment and leaves `user_data` pointing into a **dead stack frame**, which segfaults inside `load_progress_callback()` on the first request after an idle window. That was a latent defect in this patch from the day it was written; only a second `load_model()` can reach it, and nothing exercised sleep until `IdleSleepWakeIntegrationTest` was added. | | `0003-pr22393-server-add-slot-prompt-similarity-getter-setter.patch` | **Upstream-PR carry** of [ggml-org/llama.cpp#22393](https://github.com/ggml-org/llama.cpp/pull/22393) ("server : add slot_prompt_similarity getter/setter"). Purely additive: adds `server_context::get_slot_prompt_similarity()` / `set_slot_prompt_similarity(float)` (`tools/server/server-context.{cpp,h}`) so an embedding/JNI caller can query and tune the slot-selection threshold at runtime without reloading the model. Verbatim copy of the PR, which **upstream closed without merging** (rejected as exposing unsafe internal state — see the patch header). Carried permanently; it will not be droppable via a version bump. | -| `0007-server-attach-http-frontend.patch` | **Adds `llama_server_attach(argc, argv, server_context&)`** so the `NativeServer` *attach mode* can serve an **already-loaded `LlamaModel`** over the upstream HTTP frontend — no second model load, no `start_loop()`; the LlamaModel's worker keeps driving the shared `server_context` and the HTTP routes post tasks to its queue (the queue is the synchronization point). Mechanically: (1) extracts the **pure core route table** (`health` … `slots`) out of `llama_server()` into `static void llama_server_register_common_routes(ctx_http, routes)` (shared, so the two entry points cannot drift on the core endpoint set). **Scope note (narrowed at the b10154 bump):** the helper deliberately carries **only** the stable, state-independent route table — **not** the resumable-streaming routes (their handlers differ between router / non-router), the GCP-compat shim, or the experimental **CORS-proxy / MCP-server / built-in-tools** wiring. b10154 (upstream MCP-server support) moved the streaming routes into the middle of that block and coupled tools/CORS to a per-call `server_mcp mcp_mgr` lifecycle, so the earlier contiguous "route-table + CORS-proxy + tools" extraction is no longer possible; `llama_server()` keeps all of that inline, **byte-identical to upstream b10154** (only the route-table block is factored out). (2) adds `llama_server_attach`, which parses only the HTTP-side argv via `common_params_parse`, starts the stream-session GC + `server_http_context`, registers the common route table, the **non-router** resumable-streaming handlers (upstream b10154 paths `/v1/stream` GET/DEL + `/v1/streams/lookup` POST), the GCP-compat shim, and **403 "disabled" stubs for `/cors-proxy` + `/tools`** (attach mode does not wire the experimental CORS-proxy / MCP / built-in-tools host — those belong to a full `llama-server`, not an embedded model), marks ready immediately (model already loaded), and blocks on the HTTP thread until `llama_server_request_shutdown()` — never calling `common_init()`, backend init, `ctx_server.terminate()` or `llama_backend_free()` (the embedding caller owns those). Applies after `0001`+`0006` (same file); closes the "NativeServer — reuse an already-loaded LlamaModel" TODO. Upstream-submittable ("server: let embedding callers attach the HTTP frontend to an existing server_context"). **Refreshed at the b10519 bump:** upstream #26347 dropped the API key from the `/models` + `/v1/models` public-endpoint set and deleted the two trailing `// public endpoint (no API key check)` comments on those route registrations. Those two lines sit inside this patch's route-table removal block, so `git apply` failed ("patch does not apply", `server.cpp:258`) at **every** tag from b10519 on; the fix was to drop the now-wrong comment from all four affected lines (2 on the `-` side, 2 in the extracted helper on the `+` side), keeping the helper byte-identical to the block it replaces. **This is the invariant to re-check on every bump:** the `+` side of `llama_server_register_common_routes()` must stay a verbatim copy of the route table it factors out of `llama_server()`. **Refreshed at the b11104 bump** (upstream #28690, multi-address `--host`): `server_http_context` lost its single `thread` and `listening_address` members in favour of `join()` and a `listening_addresses` vector, one listener thread per bound address. The patch still *applied* cleanly there — only its own `+` lines named the removed members — so the applier could not see it; `llama_server_attach` now logs every address and blocks in `ctx_http.join()`, exactly as upstream's `llama_server()` does. | -| `0008-server-models-worker-cmd-override.patch` | **Makes router mode usable in-JVM.** The router (`server-models.cpp`) spawns each model worker by re-executing its own binary (`get_server_exec_path()` = `/proc/self/exe` & friends) — inside a JVM that binary is `java`, not a llama-server, so embedded router workers could never start. The patch adds env `LLAMA_SERVER_WORKER_CMD` (whitespace-split; read in `server_model_meta::update_args`) which replaces only the leading binary-path token of the rendered worker args, letting an embedding host relaunch workers through its own bootstrap — e.g. `java -cp app.jar net.ladenthin.llama.server.NativeServer` (each worker is then a fresh JVM running the classic single-model `NativeServer`). Exposed in Java as `NativeServer.setWorkerCommand(String...)` (JNI `setenv`); exercised by `RouterModeIntegrationTest` (Linux CI). Upstream-submittable (also useful for containerized/wrapped deployments). | +| `0007-server-attach-http-frontend.patch` | **Adds `llama_server_attach(argc, argv, server_context&)`** so the `NativeServer` *attach mode* can serve an **already-loaded `LlamaModel`** over the upstream HTTP frontend — no second model load, no `start_loop()`; the LlamaModel's worker keeps driving the shared `server_context` and the HTTP routes post tasks to its queue (the queue is the synchronization point). Mechanically: (1) extracts the **pure core route table** (`health` … `slots`) out of `llama_server()` into `static void llama_server_register_common_routes(ctx_http, routes)` (shared, so the two entry points cannot drift on the core endpoint set). **Scope note (narrowed at the b10154 bump):** the helper deliberately carries **only** the stable, state-independent route table — **not** the resumable-streaming routes (their handlers differ between router / non-router), the GCP-compat shim, or the experimental **CORS-proxy / MCP-server / built-in-tools** wiring. b10154 (upstream MCP-server support) moved the streaming routes into the middle of that block and coupled tools/CORS to a per-call `server_mcp mcp_mgr` lifecycle, so the earlier contiguous "route-table + CORS-proxy + tools" extraction is no longer possible; `llama_server()` keeps all of that inline, **byte-identical to upstream b10154** (only the route-table block is factored out). (2) adds `llama_server_attach`, which parses only the HTTP-side argv via `common_params_parse`, starts the stream-session GC + `server_http_context`, registers the common route table, the **non-router** resumable-streaming handlers (upstream b10154 paths `/v1/stream` GET/DEL + `/v1/streams/lookup` POST), the GCP-compat shim, and **403 "disabled" stubs for `/cors-proxy` + `/tools`** (attach mode does not wire the experimental CORS-proxy / MCP / built-in-tools host — those belong to a full `llama-server`, not an embedded model), marks ready immediately (model already loaded), and blocks on the HTTP thread until `llama_server_request_shutdown()` — never calling `common_init()`, backend init, `ctx_server.terminate()` or `llama_backend_free()` (the embedding caller owns those). Applies after `0001`+`0006` (same file); closes the "NativeServer — reuse an already-loaded LlamaModel" TODO. Upstream-submittable ("server: let embedding callers attach the HTTP frontend to an existing server_context"). **Refreshed at the b10519 bump:** upstream #26347 dropped the API key from the `/models` + `/v1/models` public-endpoint set and deleted the two trailing `// public endpoint (no API key check)` comments on those route registrations. Those two lines sit inside this patch's route-table removal block, so `git apply` failed ("patch does not apply", `server.cpp:258`) at **every** tag from b10519 on; the fix was to drop the now-wrong comment from all four affected lines (2 on the `-` side, 2 in the extracted helper on the `+` side), keeping the helper byte-identical to the block it replaces. **This is the invariant to re-check on every bump:** the `+` side of `llama_server_register_common_routes()` must stay a verbatim copy of the route table it factors out of `llama_server()`. **Refreshed at the b11104 bump** (upstream #28690, multi-address `--host`): `server_http_context` lost its single `thread` and `listening_address` members in favour of `join()` and a `listening_addresses` vector, one listener thread per bound address. The patch still *applied* cleanly there — only its own `+` lines named the removed members — so the applier could not see it; `llama_server_attach` now logs every address and blocks in `ctx_http.join()`, exactly as upstream's `llama_server()` does. **Refreshed at the b11361 bump** (#29818 adds `POST /v1/systemone` to the route table: one line, added on both sides so the helper stays verbatim) **and at b11401** (#29895 adds a `server_child &` overload of `llama_server()` beside the declarations and a `server_child child;` at the top of the argv entry point, the context of `0007`'s first two hunks; only context moved, the `+`/`-` lines are unchanged). Attach mode builds its own `server_routes`, whose sleep callback outlives it -- on file in `TODO.md`. | +| `0008-server-models-worker-cmd-override.patch` | **Makes router mode usable in-JVM.** The router (`server-models.cpp`) spawns each model worker by re-executing its own binary (`get_server_exec_path()` = `/proc/self/exe` & friends) — inside a JVM that binary is `java`, not a llama-server, so embedded router workers could never start. The patch adds env `LLAMA_SERVER_WORKER_CMD` (whitespace-split; read in `server_model_meta::update_args`) which replaces only the leading binary-path token of the rendered worker args, letting an embedding host relaunch workers through its own bootstrap — e.g. `java -cp app.jar net.ladenthin.llama.server.NativeServer` (each worker is then a fresh JVM running the classic single-model `NativeServer`). Exposed in Java as `NativeServer.setWorkerCommand(String...)` (JNI `setenv`); exercised by `RouterModeIntegrationTest` (Linux CI). Upstream-submittable (also useful for containerized/wrapped deployments). **Refreshed at the b11429 bump:** #29987 (untagged `4d60b4d08`, the commit before b11429) moved `server_model_meta::update_args` and changed the function after it (`update_caps(const common_params &)`), the hunk's trailing context; the `+` lines are unchanged. Since b11401 (#29895) a worker is a router *child* whose stdout carries the state commands, which a JVM worker's early `System.out` line now trips (cosmetic, `TODO.md`). | | `0006-server-embed-native-server-jni.patch` | **Makes `server.cpp`'s `llama_server` embeddable in the JVM** so the `NativeServer` JNI bridge can run the full upstream HTTP server (WebUI included) inside `libjllama` — see "Two server modes" below. b9870 already exposes `int llama_server(int, char**)` (non-static; no `main` in the file), so the patch only adds embedded-mode support: (1) a `g_llama_server_embedded` flag + `llama_server_set_embedded()` / `llama_server_request_shutdown()` (declared in the committed `src/main/cpp/native_server_bridge.h`); (2) skips installing the process-wide SIGINT/SIGTERM handlers when embedded (they would hijack the JVM's); (3) in embedded mode parses the **forwarded** argv via `common_params_parse` instead of `common_params_parse_main` (whose `GetCommandLineW` recovery would pick up `java.exe`'s command line — the same Windows class of bug `0001` fixes). `llama_server_request_shutdown()` mirrors the SIGTERM path (invokes the installed `shutdown_handler` → `ctx_server.terminate()` unblocks `start_loop()`), giving JNI an out-of-band stop since `ctx_server` is loop-local. **The handler is guarded, and both guards are load-bearing:** upstream's `shutdown_handler` lambdas capture `llama_server()`'s locals (`ctx_http`, `models_routes`, `mcp_mgr`, `ctx_server`) by reference and are never cleared, while `native_server.cpp` signals *repeatedly* until the worker reports it finished (a stop issued before the handler is installed, or before `start_loop()` begins, would otherwise be lost). A request landing after `llama_server()` returned therefore ran the lambda over destroyed objects — a `SIGSEGV` in `server_http_context::stop()` in `RouterModeIntegrationTest.tearDown` (run 36485159455), with router mode's `clean_up()` (unloading workers) making the window wide. The patch (1) takes `g_shutdown_handler_mutex` for every read, write and invocation of the handler (the caller is on another thread) and (2) declares a `llama_server_shutdown_handler_reset` **after** the captured locals, so the handler is cleared before any of them is destroyed; because invocation holds the same lock, that reset waits for an invocation in progress. `0007`'s `llama_server_attach` carries the same guard. Runnable guard: `src/test/cpp/test_native_server_shutdown.cpp`. Applies **after `0001`** (which flips this call site to `common_params_parse_main`), so its context is the post-`0001` tree; regenerate against `0001`+source on a bump. Only touches `tools/server/server.cpp`. | | `0012-model-guard-zero-split-sum-and-name-the-device-index.patch` | **A GPU that reports zero free memory makes every model load fail with the unactionable `error loading model: vector`.** `llama_model_base::load_tensors` (`src/llama-model.cpp`) weights the per-device layer split by `ggml_backend_dev_memory()`'s `free`, then normalises: `splits[i] /= split_sum`. With a single device reporting `free == 0` that is `0/0` → **NaN** in every split point; NaN compares false against everything, so the `std::upper_bound` below returns the end iterator, `layer_gpu == n_devices()`, and `devices.at(layer_gpu)` throws `std::out_of_range` — whose libc++ `what()` is the bare string `"vector"`, which `llama.cpp`'s `catch (const std::exception &)` prints verbatim. Upstream's `free == 0 && total == 0` host-memory fallback does **not** fire, because `total` is `recommendedMaxWorkingSetSize` and is non-zero. **Reachable since b10618..b10797**: upstream `8c0b9cd04` ("metal : fix memory query under low-memory conditions", [#27701](https://github.com/ggml-org/llama.cpp/pull/27701)) changed `ggml-metal-device.m` to `*free = *total > cur ? *total - cur : 0`; before that clamp an over-committed device (`currentAllocatedSize > recommendedMaxWorkingSetSize`) *underflowed* to a huge `size_t`, which normalised fine, so the same precondition was harmless. That is why the `Java Tests macOS …` jobs went red at the b10792→b10797 step while every Linux/Windows job stayed green — **and why only a GPU build can fail this way at all**: `act_gpu_layers` is `devices.empty() ? 0 : …`, so with no GPU backend `devices` is empty, every layer returns early on `cpu_dev`, and the `.at()` line is unreachable. **Shape:** the two blocks are lifted out of `load_tensors` into free functions declared in `src/llama-model.h`, purely so they can be driven by a test — the failing state needs a real over-committed GPU and cannot be arranged through any public API. `llama_model_splits_normalize()` carries **the fix**: on `split_sum == 0` it `LLAMA_LOG_WARN`s and falls back to an even split (`splits[i] = float(i+1)/splits.size()`), the only neutral choice when no device can be preferred and exactly right for a single device. `llama_model_splits_select_device()` carries **the diagnostic**: it bounds-checks the index and throws a `std::runtime_error` naming the function, the offloaded layer, the device index, the split-point count **and the split points themselves** — with NaN splits that message prints `nan` and names the cause outright, which is precisely what was missing when this had to be diagnosed by reading source. **A second, backend-independent trigger reaches the same line**, found while writing this up and verified against the unfixed library: `--tensor-split` values are parsed with `std::stof` and never range-checked (`common/arg.cpp`), so `-ts 1,-1` cancels out, `split_sum` is 0 again, the split points become `[inf, -nan]`, and every layer maps one past the last device — on CUDA, Vulkan or ROCm just as much as on Metal, with no memory pressure involved. That is what makes this an ordinary upstream defect rather than a Metal edge case, and the warning names both causes rather than only the memory one. Also adds upstream `tests/test-model-split.cpp` (5 cases in upstream's `testing.h` style) + its `llama_build_and_test` registration. Touches `src/llama-model.{cpp,h}`, `tests/test-model-split.cpp` and `tests/CMakeLists.txt` — **none** of which any other patch touches, so it is independent of all of them. Upstream-submittable ("model: fall back to an even split when no device reports free memory"); **not yet filed upstream**. **Runnable guard: `src/test/cpp/test_model_split.cpp`** — a FetchContent subproject builds with `LLAMA_BUILD_TESTS=OFF`, so the upstream test above is applied-but-never-compiled here (same as `0001`'s test). That file drives the same two functions from `jllama_test`, which runs on **every** platform in `C++ Tests`, so a bump that drops this patch fails the build at link time everywhere instead of surfacing as one red macOS Java job. **Verification limit — read before assuming this can be dropped:** the *failing path* still cannot be reached without a GPU backend, so the guard pins the arithmetic (what actually broke), not the end-to-end load; the end-to-end proof is the macOS CI job. On a bump, re-check whether upstream added its own `split_sum == 0` guard (grep `split_sum` in `src/llama-model.cpp`) and **drop this patch rather than refreshing it** if they did — the fail-loud applier detects "does not apply", never "upstream already fixed this". | -| `0014-common-log-callback-sink.patch` | **Gives `common_log` a callback sink: `common_log_set_callback(log, cb, user_data)` (`common/log.{h,cpp}`).** This is what `LlamaModel.setLogger` hooks. Before it, the Java logger was a `llama_log_set()` callback, which has two holes, both found while chasing `slot print_timing` lines interleaving with the Atmosphere agent's streamed answer: **(a)** every model load runs `common_init()`, which re-points `llama_log_set()` at `common_log_default_callback` (`common.cpp:394`), so `setLogger(…)` *before* `new LlamaModel(…)` silently lost the callback; **(b)** the server's own `SRV_*`/`SLT_*` macros are `LOG_INF` and write straight into `common_log`, which `llama_log_set()` never carried, so the per-request `slot …`/`srv …` lines could not be routed to Java at all (the reason `LlamaModelTest#testLogText/JSON` sat `@Disabled` for years). `common_log` upstream offers file, colors, prefix, timestamps, verbosity and JSONL but no hook. The patch adds one: while a callback is set, the worker thread hands every entry to it **instead of** printing to stdout/stderr (a `--log-file` still receives them); the callback gets the bare formatted message (no prefix/timestamp/colors) with the `ggml_log_callback` signature; swapping the callback pauses the worker first, so queued entries reach the *previous* sink (which is what makes `setLogger(format, null)` a synchronous drain). With the sink, `common_init()`'s `llama_log_set()` reset is harmless — it points at the default callback that feeds `common_log`, i.e. exactly the path into the sink — so the ordering problem (a) disappears without any re-install logic, and the `srv`/`slot` lines (b) arrive because they are `common_log` entries. `jllama.cpp`'s `setLogger` therefore sets `common_log_set_callback(common_log_main(), trampoline)` **plus** `llama_log_set(common_log_default_callback)` (so llama/ggml lines feed `common_log` even before the first load). Two consequences to know: the callback runs on `common_log`'s **worker thread**, a plain `std::thread` llama.cpp re-creates on every pause/resume and never attaches to the JVM — the trampoline attaches per call and detaches again (`get_jni_env_attaching`; a thread that exits while attached leaks a `JavaThread`, and this thread is not ours — the leak-free choice, not the cheapest: each attach creates a `java.lang.Thread` object, so a `thread_local` guard that detaches once at thread exit is the optimisation on file in `TODO.md`), `setLogger` must call `common_log_set_callback` **outside** `g_log_mutex`, because the pause joins the worker, which needs that mutex to read the callback, and `setLogger` callers are serialized by a **separate** `g_set_logger_mutex`: two unserialized swaps race on the worker's `std::thread` (one joins it while the other assigns a fresh thread over the still-joinable object = `std::terminate`, the whole JVM), which `LlamaLoggerTest#concurrentSetLoggerCallsDoNotRaceOnTheLogWorker` reproduced before the mutex existed. Two caveats the Javadoc carries: a caller must not hold a lock the *previous* callback needs (the drain runs it on the worker while the caller waits), and the verbosity threshold is process-wide and reset by **every** load (`common_params_parse` ends with `common_log_set_verbosity_thold(params.verbosity)`, default 3), so a load without `-lv` puts it back to 3 — a review assumed the opposite, and `LlamaLoggerTest#verbosityThresholdIsProcessWideAndEveryLoadSetsIt` now pins the measured behaviour. And the verbosity threshold applies *before* the sink: at the default (`3`) the Java logger sees errors, warnings and the server's INFO lines, while llama/ggml INFO lines (`common_log_get_verbosity` maps them to TRACE = 4) arrive only from `setLogVerbosity(4)` on — the same filtering the console gets, and a behaviour change for consumers who captured the unfiltered `llama_log_set()` stream before. **Runnable guards:** `src/test/cpp/test_common_log_callback.cpp` (6 tests over a private `common_log_init()` instance: delivery, bare text under prefix+timestamps, clear, swap-drains-to-old-sink, file kept, levels pass through) links the function on every platform, so a bump that drops the patch reds `C++ Tests` at link time; `LlamaLoggerTest` (model-free, needs only `libjllama`: a logger set before a deliberately failing load on a non-GGUF file sees the `srv … loading model` INFO line and llama's ERROR line, in TEXT and JSON) and the re-enabled `LlamaModelTest#testLogText/testLogJSON` plus `#testLoggerSetBeforeLoadSurvivesTheLoad` (vocab-only load) cover the Java side. Upstream-submittable ("common : add a callback sink to common_log for embedding hosts"); **not yet filed upstream**. Touches only `common/log.{h,cpp}`, which no other patch touches. **On a bump, check whether upstream added a hook of its own (grep `callback` in `common/log.h`) and, if so, DROP this patch and port `setLogger` to theirs rather than refreshing it.** | -| `0015-rpc-embeddable-client-and-stoppable-server.patch` | **Makes ggml-rpc usable inside a JVM** (see "RPC backend" below). Upstream treats every client-side RPC problem as `GGML_ABORT` — a malformed endpoint, a server that is not running, a failed handshake — which in a JVM kills the application over a typo in `--rpc`. **(1)** The *registration* path reports failure instead: `rpc_dispatcher::start()` returns `false`, `try_get_dispatcher()` `nullptr`, `ggml_backend_rpc_get_device_count()` `0`, `ggml_backend_rpc_add_server()` `nullptr`, and `add_rpc_devices()` (`common/arg.cpp`) throws `std::invalid_argument` naming the server where it used to register `nullptr` and silently run without it. Every path *after* registration keeps the original contract through `get_dispatcher()`, which still aborts — a server that vanishes mid-inference stays fatal (TODO). **(2)** `add_server()`'s cache hit is re-checked (a stopped server used to pass registration and abort on the first tensor upload), and `ggml_backend_rpc_get_device_memory()` of a gone server returns 0/0 instead of aborting — RPC devices stay in ggml's process-wide registry forever, and `common_init()` queries the memory of **every** registered device at `-lv 4`, so a stopped server would otherwise take down an unrelated later load. **(3)** The server becomes stoppable: `ggml_backend_rpc_stop_server()` (disconnects the client being served, wakes `accept()` with a loopback connection — the one portable way) and `ggml_backend_rpc_server_listening()`; the loop now reaches its cleanup and frees its backends on every early return, but deliberately **not** `rpc_transport_shutdown()`, whose `WSACleanup()` would invalidate every other RPC socket in the process. **(4)** transport: `MSG_NOSIGNAL`/`SO_NOSIGPIPE`, `socket_t::shutdown()`, and the fds `connect()`/`create_server()`/`accept()` leaked on their error paths are closed. Touches `ggml/include/ggml-rpc.h`, `ggml/src/ggml-rpc/{ggml-rpc.cpp,transport.cpp,transport.h}` and one function in `common/arg.cpp` (`add_rpc_devices`, away from `0001`'s `common_params_parse` hunks). **Runnable guard: `src/test/cpp/test_rpc.cpp`** — links both new functions (a bump that drops the patch fails `C++ Tests` at link time on every platform) and drives registration, stop and the memory query over loopback. Upstream-submittable; **not yet filed upstream**. On a bump, check whether upstream grew its own stop function or non-aborting registration (grep `stop_server` / `GGML_ABORT("Failed to connect` in `ggml-rpc.cpp`). | +| `0014-common-log-callback-sink.patch` | **Gives `common_log` a callback sink: `common_log_set_callback(log, cb, user_data)` (`common/log.{h,cpp}`).** This is what `LlamaModel.setLogger` hooks. Before it, the Java logger was a `llama_log_set()` callback, which has two holes, both found while chasing `slot print_timing` lines interleaving with the Atmosphere agent's streamed answer: **(a)** every model load runs `common_init()`, which re-points `llama_log_set()` at `common_log_default_callback` (`common.cpp:394`), so `setLogger(…)` *before* `new LlamaModel(…)` silently lost the callback; **(b)** the server's own `SRV_*`/`SLT_*` macros are `LOG_INF` and write straight into `common_log`, which `llama_log_set()` never carried, so the per-request `slot …`/`srv …` lines could not be routed to Java at all (the reason `LlamaModelTest#testLogText/JSON` sat `@Disabled` for years). `common_log` upstream offers file, colors, prefix, timestamps, verbosity and JSONL but no hook. The patch adds one: while a callback is set, the worker thread hands every entry to it **instead of** printing to stdout/stderr (a `--log-file` still receives them); the callback gets the bare formatted message (no prefix/timestamp/colors) with the `ggml_log_callback` signature; swapping the callback pauses the worker first, so queued entries reach the *previous* sink (which is what makes `setLogger(format, null)` a synchronous drain). With the sink, `common_init()`'s `llama_log_set()` reset is harmless — it points at the default callback that feeds `common_log`, i.e. exactly the path into the sink — so the ordering problem (a) disappears without any re-install logic, and the `srv`/`slot` lines (b) arrive because they are `common_log` entries. `jllama.cpp`'s `setLogger` therefore sets `common_log_set_callback(common_log_main(), trampoline)` **plus** `llama_log_set(common_log_default_callback)` (so llama/ggml lines feed `common_log` even before the first load). Two consequences to know: the callback runs on `common_log`'s **worker thread**, a plain `std::thread` llama.cpp re-creates on every pause/resume and never attaches to the JVM — the trampoline attaches per call and detaches again (`get_jni_env_attaching`; a thread that exits while attached leaks a `JavaThread`, and this thread is not ours — the leak-free choice, not the cheapest: each attach creates a `java.lang.Thread` object, so a `thread_local` guard that detaches once at thread exit is the optimisation on file in `TODO.md`), `setLogger` must call `common_log_set_callback` **outside** `g_log_mutex`, because the pause joins the worker, which needs that mutex to read the callback, and `setLogger` callers are serialized by a **separate** `g_set_logger_mutex`: two unserialized swaps race on the worker's `std::thread` (one joins it while the other assigns a fresh thread over the still-joinable object = `std::terminate`, the whole JVM), which `LlamaLoggerTest#concurrentSetLoggerCallsDoNotRaceOnTheLogWorker` reproduced before the mutex existed. Two caveats the Javadoc carries: a caller must not hold a lock the *previous* callback needs (the drain runs it on the worker while the caller waits), and the verbosity threshold is process-wide and reset by **every** load (`common_params_parse` ends with `common_log_set_verbosity_thold(params.verbosity)`, default 3), so a load without `-lv` puts it back to 3 — a review assumed the opposite, and `LlamaLoggerTest#verbosityThresholdIsProcessWideAndEveryLoadSetsIt` now pins the measured behaviour. And the verbosity threshold applies *before* the sink: at the default (`3`) the Java logger sees errors, warnings and the server's INFO lines, while llama/ggml INFO lines (`common_log_get_verbosity` maps them to TRACE = 4) arrive only from `setLogVerbosity(4)` on — the same filtering the console gets, and a behaviour change for consumers who captured the unfiltered `llama_log_set()` stream before. **Runnable guards:** `src/test/cpp/test_common_log_callback.cpp` (6 tests over a private `common_log_init()` instance: delivery, bare text under prefix+timestamps, clear, swap-drains-to-old-sink, file kept, levels pass through) links the function on every platform, so a bump that drops the patch reds `C++ Tests` at link time; `LlamaLoggerTest` (model-free, needs only `libjllama`: a logger set before a deliberately failing load on a non-GGUF file sees the `srv … loading model` INFO line and llama's ERROR line, in TEXT and JSON) and the re-enabled `LlamaModelTest#testLogText/testLogJSON` plus `#testLoggerSetBeforeLoadSurvivesTheLoad` (vocab-only load) cover the Java side. Upstream-submittable ("common : add a callback sink to common_log for embedding hosts"); **not yet filed upstream**. Touches only `common/log.{h,cpp}`, which no other patch touches. **Refreshed at the b11401 bump** (#29895 adds a `colors` member next to the sink's members and a `get_colors()` next to `set_callback()`; only context moved, the `+`/`-` lines are unchanged, and upstream still has no hook -- it added `common_log_get_colors` only). **On a bump, check whether upstream added a hook of its own (grep `callback` in `common/log.h`) and, if so, DROP this patch and port `setLogger` to theirs rather than refreshing it.** | +| `0015-rpc-embeddable-client-and-stoppable-server.patch` | **Makes ggml-rpc usable inside a JVM** (see "RPC backend" below). Upstream treats every client-side RPC problem as `GGML_ABORT` — a malformed endpoint, a server that is not running, a failed handshake — which in a JVM kills the application over a typo in `--rpc`. **(1)** The *registration* path reports failure instead: `rpc_dispatcher::start()` returns `false`, `try_get_dispatcher()` `nullptr`, `ggml_backend_rpc_get_device_count()` `0`, `ggml_backend_rpc_add_server()` `nullptr`, and `add_rpc_devices()` (`common/arg.cpp`) throws `std::invalid_argument` naming the server where it used to register `nullptr` and silently run without it. Every path *after* registration keeps the original contract through `get_dispatcher()`, which still aborts — a server that vanishes mid-inference stays fatal (TODO). **(2)** `add_server()`'s cache hit is re-checked (a stopped server used to pass registration and abort on the first tensor upload), and `ggml_backend_rpc_get_device_memory()` of a gone server returns 0/0 instead of aborting — RPC devices stay in ggml's process-wide registry forever, and `common_init()` queries the memory of **every** registered device at `-lv 4`, so a stopped server would otherwise take down an unrelated later load. **(3)** The server becomes stoppable: `ggml_backend_rpc_stop_server()` (disconnects the client being served, wakes `accept()` with a loopback connection — the one portable way) and `ggml_backend_rpc_server_listening()`; the loop now reaches its cleanup and frees its backends on every early return, but deliberately **not** `rpc_transport_shutdown()`, whose `WSACleanup()` would invalidate every other RPC socket in the process. **(4)** transport: `MSG_NOSIGNAL`/`SO_NOSIGPIPE`, `socket_t::shutdown()`, and the fds `connect()`/`create_server()`/`accept()` leaked on their error paths are closed. Touches `ggml/include/ggml-rpc.h`, `ggml/src/ggml-rpc/{ggml-rpc.cpp,transport.cpp,transport.h}` and one function in `common/arg.cpp` (`add_rpc_devices`, away from `0001`'s `common_params_parse` hunks). **Runnable guard: `src/test/cpp/test_rpc.cpp`** — links both new functions (a bump that drops the patch fails `C++ Tests` at link time on every platform) and drives registration, stop and the memory query over loopback. Upstream-submittable; **not yet filed upstream**. On a bump, check whether upstream grew its own stop function or non-aborting registration (grep `stop_server` / `GGML_ABORT("Failed to connect` in `ggml-rpc.cpp`). **Refreshed at the b11450 bump** (#26610, `-sm tensor` over RPC, ~800 lines in `ggml-rpc.cpp`, protocol 7 → 8): three hunks lost their context -- new dispatcher members beside `start()`, and new `get_proc_address` names before `return NULL` (the two new names now sit after `ggml_backend_comm_allreduce_tensor`); the `+`/`-` lines are unchanged. The server-to-server comm #26610 adds (a listener on `0.0.0.0`, an `accept()` a stop does not wake) is on file in `TODO.md`. | +| `0016-model-kolibri1.patch` | **Adds Aleph Alpha's Kolibri-1 (architecture `kolibri1`) ahead of upstream** -- a 78B German/English reasoning MoE (3.46B active) that upstream does not support yet ([ggml-org/llama.cpp#29922](https://github.com/ggml-org/llama.cpp/issues/29922), no PR at the time). A **temporary carry, not a fix**: drop it the moment upstream registers the architecture (`git grep -n kolibri src/llama-arch.cpp` at the new tag), and keep `test_kolibri1.cpp` -- its numerical comparisons must then pass against upstream's implementation, while its GGUF-format rows follow whatever format upstream's converter fixes (see the test-table row). **Architecture** (Aleph Alpha's vLLM reference, `aleph_alpha_inference/kolibri1.py` at `049a6a7`): GQA with per-head q/k RMSNorm; `layer_types` interleaves sliding-window layers (NEOX RoPE) with full-attention layers that have **no** positional encoding; sandwich norms around attention and MoE; every layer routed MoE + one ungated shared expert; the router selects top-k on **`logits + expert_bias`** and weights by the **unbiased `sigmoid(logits)`** -- not DeepSeek-V3's sigmoid router (selection on `sigmoid(logits) + bias`), which picks other experts as soon as the bias is non-zero. The graph therefore builds the selection itself and passes it to `build_moe_ffn` (`selected_experts_in`), leaving upstream's shared MoE code untouched. **Derived from two community ports, which are incompatible with each other's GGUFs** -- Eliasfpv28's AFMoE-based one (base b11378, `expert_gating_func` 2, pre-tokenizer `qwen2`) and the Qwen3-MoE-based one behind Hob-forge's GGUFs (base b11381, as carried in [mjwsolo/localcode#101](https://github.com/mjwsolo/localcode/pull/101): gating 5, pre-tokenizer `kolibri1`, rejects gating 2). This patch **loads both dialects**: gating 2, 5 or absent (all mean the one router), `kolibri1` mapped to the Qwen2 pre-tokenizer in `llama-vocab.cpp`, `expert_weights_norm` / `expert_shared_count` / `output.weight` optional. Every source, with license, is listed in the patch header; `REUSE.toml` annotates the file `MIT AND Apache-2.0` (Eliasfpv28's additions are Apache-2.0). Touches `src/llama-arch.{h,cpp}`, `src/llama-model.cpp` (mapping + NEOX rope list), `src/llama-vocab.cpp` (one name), `src/models/models.h` and the new `src/models/kolibri1.cpp`; no converter. **Runnable guard: `src/test/cpp/test_kolibri1.cpp`** (writes tiny random GGUFs in both dialects and compares every logit with an independent double-precision reference). **Not verified here:** the real 78B model (HuggingFace is unreachable from the sessions this was written in, and the smallest GGUF is 28.6 GB) and GPU backends -- the community ports report real-model checks (top-1 67/67 against a float64 reference for Q8_0, tool calls, German text). | **Dropped patches** (`0004`, `0005`, `0009`, `0010`, `0011`, `0013` -- upstream fixed each defect itself) are recorded in [`docs/history/dropped-llama-patches.md`](docs/history/dropped-llama-patches.md): @@ -1237,7 +1252,7 @@ For the full record of upstream API breaks across version ranges (b5022 → ### Java (Maven) ```bash -mvn compile # Compiles Java and generates JNI headers +mvn compile # Compiles Java (jllama.h is maintained by hand, see "The JNI exception boundary") mvn test # Run all tests (requires native library and model files) mvn package # Build JAR mvn -P assembly package # Also build the fat jar-with-dependencies uber JAR (library + Java deps + native libs); CI builds it and uploads it in the `llama-jars` artifact @@ -1301,7 +1316,7 @@ claims came from. End-to-end local workflow for running Java tests: ```bash -# 1. Generate JNI headers (one-time per Java API change) +# 1. Compile the Java classes (a new native method also needs its line in jllama.h, by hand) mvn -q compile # 2. Configure + build the native library for the current host @@ -1345,6 +1360,7 @@ below covers the model bindings: | `net.ladenthin.llama.tts.model` | `TtsIntegrationTest` | `Qwen3-TTS-12Hz-1.7B-Base-Q4_K_M.gguf` (any Qwen3-TTS-family model works) | | `net.ladenthin.llama.tts.mmproj` | `TtsIntegrationTest` | `mmproj-Qwen3-TTS-12Hz-1.7B-Base-Q8_0.gguf` | | `net.ladenthin.llama.train.model` | `LlamaTrainerIntegrationTest` | `stories260K.gguf` (must be **F32**) | +| `net.ladenthin.llama.decision.model` | `SystemOneIntegrationTest` (decision-model half) | none (not in CI's set) — e.g. upstream's `ggml-org/tinylaya-for-testing-gguf` | | `net.ladenthin.llama.audio.model` | `AudioInputIntegrationTest` (llama.cpp discussion #13759) | none (not in CI's set) — e.g. `ultravox-v0_5-llama-3_2-1b.gguf` | | `net.ladenthin.llama.audio.mmproj` | `AudioInputIntegrationTest` | none — e.g. `mmproj-ultravox-v0_5-llama-3_2-1b-f16.gguf` | | `net.ladenthin.llama.audio.input` | `AudioInputIntegrationTest` | committed `src/test/resources/audios/sample.wav`; any `.wav`/`.mp3` | @@ -1427,7 +1443,7 @@ pip install "clang-format==23.1.3" clang-format -i src/main/cpp/*.cpp src/main/cpp/*.hpp src/test/cpp/*.cpp # Format C++ code ``` -The generated JNI header `src/main/cpp/jllama.h` (produced by `javac -h`) is intentionally excluded. +The JNI header `src/main/cpp/jllama.h` (originally `javac -h` output, now maintained by hand) is intentionally excluded. To bump the enforced version, update the pin in **both** the workflow (`CLANG_FORMAT_VERSION`) and this line, then reformat the whole tree with the new version in the same commit. @@ -1506,7 +1522,7 @@ If the local check passes (`BUILD SUCCESS`), the `mvn package` job in - The `server` package is a dedicated top layer in the ArchUnit `layeredArchitecture` rule (the only layer allowed to access the root `Api`); `noInternalJdkImports` carries an explicit exception for the supported `com.sun.net.httpserver` (the exported `jdk.httpserver` module, which `module-info.java` `requires`). See README "OpenAI-compatible HTTP server". **Native layer** (`src/main/cpp/`): -- `jllama.cpp` — JNI implementation bridging Java calls to llama.cpp. ~1,900 lines; 34 native methods (30 `LlamaModel` + 3 `TextToSpeech` + 1 `LlamaQuantizer`) plus `JNI_OnLoad`/`JNI_OnUnload`. +- `jllama.cpp` — JNI implementation bridging Java calls to llama.cpp. ~1,900 lines; 35 native methods (31 `LlamaModel` + 3 `TextToSpeech` + 1 `LlamaQuantizer`) plus `JNI_OnLoad`/`JNI_OnUnload`. - `utils.hpp` — Helper utilities (format helpers, argv stripping, token-piece serialisation). - `json_helpers.hpp` — Pure JSON transformation helpers (no JNI, no llama state). Independently unit-testable. - `jni_helpers.hpp` — JNI bridge helpers (handle management + server orchestration). Includes `json_helpers.hpp`. @@ -1563,7 +1579,7 @@ The project C++ helpers follow a strict semantic split: Functions: `get_result_error_message`, `results_to_json`, `rerank_results_to_json`, `parse_encoding_format`, `extract_embedding_prompt`, `is_infill_request`, `parse_slot_prompt_similarity`, `parse_positive_int_config`, `wrap_stream_chunk`, -`server_metrics_to_json`. +`server_metrics_to_json`, `route_error_message`. **`log_helpers.hpp`** — Pure log-formatting transforms. - Input: `ggml_log_level`, message text (`const char*`), an explicit `std::time_t` timestamp. @@ -1627,8 +1643,8 @@ Functions with `_impl` suffix are called directly from `jllama.cpp`. An exception that escapes a native method and unwinds across the JNI boundary is **undefined behaviour and aborts the JVM** on most implementations. **Every `Java_*` entry point must therefore -convert anything that escapes into a Java exception**, and there are 44 of them across four TUs — -`jllama.cpp` (34), `native_server.cpp` (5), `rpc_bridge.cpp` (4), `train_engine.cpp` (1). +convert anything that escapes into a Java exception**, and there are 45 of them across four TUs — +`jllama.cpp` (35), `native_server.cpp` (5), `rpc_bridge.cpp` (4), `train_engine.cpp` (1). The mechanism is `jni_guard_impl(env, exception_class, [&]() -> Ret { … })` (`jni_helpers.hpp`, Layer A). It is **additive**: an entry point that already converts `std::exception` itself keeps @@ -1655,6 +1671,13 @@ contract. **When you add a native method, wrap it.** The guard is not enforced by a test — a new unguarded entry point is invisible until something throws through it in production. +**And declare it in `src/main/cpp/jllama.h` by hand.** That header is committed and maintained +manually -- nothing regenerates it, `mvn compile` included, whatever older notes here say -- and it +is what gives the `LlamaModel` entry points in `jllama.cpp` their C linkage (they are not inside an +`extern "C"` block there). A method missing from it compiles and links, but under its C++-mangled +name, so the JVM cannot find it and the first call throws `UnsatisfiedLinkError`. Caught this way +for `handleSystemOne` at the b11361 bump, before the first build. + ### Parameter Flow Java parameters are serialized to JSON strings and passed to native code, which deserializes them using nlohmann/json. This avoids complex JNI field mapping for the many llama.cpp parameters. @@ -1765,7 +1788,8 @@ properties. That also closed a gap this file used to describe: `LlamaTrainerInte to `stories260K.gguf` and runs on every Java test job, so the Java → JNI → native trainer round trip has a runnable guard. **One** class still self-skips everywhere: `AudioInputIntegrationTest` — its prompt clip is committed (`src/test/resources/audios/sample.wav`), but the audio model + mmproj have no -CI download. +CI download. Half of `SystemOneIntegrationTest` does too (the `/v1/systemone` answers need a decision model, +which is not in the CI set either); its rejection case runs everywhere with the draft model. The model set has a **single source of truth: `.github/models.csv`** (one `filename,url` row per model; `#` comments). Everything derives from it: the **`download-models`** job (ubuntu, `needs: startgate`) is the only place models are fetched from HuggingFace (one manifest-driven @@ -1825,24 +1849,25 @@ ctest --test-dir build --output-on-failure -R "ResultsToJson" |------|-------|-------| | `src/test/cpp/test_utils.cpp` | 168 | Upstream helpers: `server_tokens`, `server_grammar_trigger`, `gen_tool_call_id`, `json_value`, `json_get_nested_values`, UTF-8 helpers, `format_response_rerank`, `format_embeddings_response_oaicompat`, `oaicompat_completion_params_parse`, `oaicompat_chat_params_parse`, `are_lora_equal`, `strip_flag_from_argv`, `token_piece_value`, `json_is_array_and_contains_numbers`, `format_oai_sse`, `format_oai_resp_sse`, `format_anthropic_sse`, `parse_lora_request`, `common_chat_parse` over malformed UTF-8 (the `ContentOnlyParseUtf8` guard, which pins upstream #29161's one-U+FFFD-per-invalid-run contract — formerly the guard for the dropped `patches/0011`) | | `src/test/cpp/test_server.cpp` | 206 | Upstream result types: `server_slot_stats` (the `timings` JSON payload; replaced `result_timings` in b10408), `task_params::to_json()` (incl. `dry_sequence_breakers`, `preserved_tokens`, `timings_per_token`), `completion_token_output`, `server_task_result_cmpl_partial` (non-oaicompat + `to_json_oaicompat` + logprobs + `to_json_oaicompat_chat` + `to_json_anthropic` + dispatcher), `server_task_result_cmpl_final` (non-oaicompat + `to_json_oaicompat` + `to_json_oaicompat_chat` + `to_json_oaicompat_chat_stream` + `to_json_anthropic` + `to_json_anthropic_stream` + tool_calls + dispatcher), `server_task_result_embd`, `server_task_result_rerank`, `server_task_result_metrics` (`to_metrics()` = the `/metrics` Prometheus exposition text; its `to_json()` has been unused since b10519 and returns `json{}` = JSON null), `server_task_result_slots` (`to_json()` = the `/slots` array, fed by the b10519 `SERVER_TASK_TYPE_SLOT_GET` task), `server_task_result_slot_save_load`, `server_task_result_slot_erase`, `server_task_result_apply_lora`, `server_task_result_get_lora`, `server_task_result_error`, `format_error_response`, `server_task::need_sampling()`, `server_task::n_tokens()`, `server_schema::eval_llama_cmpl_schema()` (parsing pipeline + grammar routing + error paths + per-request `dry_*` and `sse_ping_interval` field round-trips incl. hard-limit + server-default inheritance), `response_fields` projection | -| `src/test/cpp/test_json_helpers.cpp` | 64 | All functions in `json_helpers.hpp`: `get_result_error_message`, `results_to_json`, `rerank_results_to_json` (incl. missing/out-of-range `index` rejection), `parse_encoding_format`, `extract_embedding_prompt`, `is_infill_request`, `parse_slot_prompt_similarity`, `parse_positive_int_config`, `wrap_stream_chunk`, `server_metrics_to_json` | +| `src/test/cpp/test_json_helpers.cpp` | 67 | All functions in `json_helpers.hpp`: `get_result_error_message`, `results_to_json`, `rerank_results_to_json` (incl. missing/out-of-range `index` rejection), `parse_encoding_format`, `extract_embedding_prompt`, `is_infill_request`, `parse_slot_prompt_similarity`, `parse_positive_int_config`, `wrap_stream_chunk`, `server_metrics_to_json`, `route_error_message` | | `src/test/cpp/test_log_helpers.cpp` | 13 | All functions in `log_helpers.hpp`: `log_level_name`, `format_log_as_json` | | `src/test/cpp/test_common_log_callback.cpp` | 6 | **The runnable guard for `patches/0014`**: `common_log_set_callback()` on a private `common_log_init()` instance (never `common_log_main()`, so the process-wide logger the other tests print through is untouched) — delivery of level + bare text, no prefix/timestamp even when both are on (what `common_init()` does), clearing stops delivery, a swap drains queued entries to the *previous* sink (the property behind `LlamaModel.setLogger(format, null)` being a synchronous flush), a `--log-file` keeps being written alongside the sink, and every `ggml_log_level` passes through unchanged. The Java half (`LlamaLoggerTest`, model-free) proves the JNI trampoline on top of it. | | `src/test/cpp/test_jni_helpers.cpp` | 70 | All functions in `jni_helpers.hpp` using a zero-filled `JNINativeInterface_` mock (incl. the `utf8_to_jstring_impl` byte-array string path: emoji byte-preservation, truncated-UTF-8 replace-not-throw). Seven of them pin `jni_guard_impl` — the JNI exception boundary every `Java_*` entry point runs inside — including the `catch (...)` arm that is the only backstop for a non-`std::exception` type, and its two refusals (never `ThrowNew` over a pending Java exception, never with a null class). | | `src/test/cpp/test_tts_wav.cpp` | 2 | The in-memory WAV writer `pcm_to_wav16_bytes` in `tts_wav.hpp` (WAV header/payload + little-endian clamping) — our own code, not upstream. The Qwen3-TTS pipeline it pairs with (`mtmd_helper::gen_audio`) is entirely upstream-owned (no project-side DSP to unit-test here). The load path is additionally covered by `test_tts_params.cpp` (3 tests over `tts_params.hpp`'s `build_tts_params`, plus 2 pinning the upstream `-1` default it depends on), which pins the CPU-thread resolution whose absence used to crash the JVM on every platform — see the `TODO.md` entry for the mechanism. End-to-end coverage is `TtsIntegrationTest`, which is model-gated. | | `src/test/cpp/test_tts_params.cpp` | 13 | The **three** builders every hand-assembled `common_params` goes through: `build_tts_params` (`tts_params.hpp`), `build_train_params` (`train_params.hpp`) and the shared `jllama::resolve_cpu_params` (`cpu_params.hpp`). Each builder is guarded separately on purpose — testing the resolver alone does **not** cover its call sites, because `train_engine.cpp` is compiled into `jllama` only, never into `jllama_test`, and `LlamaTrainerIntegrationTest` is gated on `net.ladenthin.llama.train.model`, which no CI job sets. Without these the JVM-abort bug could regress in the trainer on every platform, unseen. | | `src/test/cpp/test_model_split.cpp` | 7 | The two `load_tensors()` split helpers that `patches/0012` extracts out of llama.cpp's `src/llama-model.cpp` — `llama_model_splits_normalize` (proportional split, single device, and the zero-sum case that used to produce NaN, **and the cancelling `--tensor-split` case** — `-ts 1,-1` reaches the identical line on any backend with no GPU memory pressure at all) and `llama_model_splits_select_device` (every layer maps to a real device index; malformed split points throw a message that names the function, the layer, the index and the split values instead of libc++'s bare `"vector"`). **This is the runnable guard for `0012`**: the patch also ships an upstream `tests/test-model-split.cpp`, but a FetchContent subproject builds with `LLAMA_BUILD_TESTS=OFF`, so that one is applied-but-never-compiled here. This file is the only place the two functions are linked in CI, on every platform — so a bump that drops the patch fails the `C++ Tests` build outright rather than resurfacing as one red macOS Java job. It is the one test file that includes an **internal** upstream header (`llama-model.h`, via the `${llama.cpp_SOURCE_DIR}/src` include dir added for it), which is deliberate: a signature drift should fail loudly at compile time. | +| `src/test/cpp/test_kolibri1.cpp` | 5 | **The runnable guard for `patches/0016`** (Kolibri-1). Writes tiny random `kolibri1` GGUFs with the public `gguf` API, loads them through the real library on the CPU and compares every logit -- one batch, and token by token through the iSWA KV cache -- with an independent double-precision reference written from Aleph Alpha's vLLM implementation, not from the patch. The layer pattern puts a full-attention (NoPE) layer between sliding ones, the sequence is longer than the window, and the expert biases are large enough that DeepSeek-V3's router would pick other experts (one test asserts that, so the comparison cannot pass vacuously). Covers both published GGUF dialects (gating 2 + `qwen2` + `output.weight`; gating 5 + `kolibri1` + tied output), renormalized routing, and the rejection of another gating function. **What happens when upstream supports Kolibri-1 and `0016` is dropped** -- the file compiles unchanged (public `llama.h`/`gguf.h`/`ggml.h` only, nothing from the patch), but its tests split in two. The **numerical comparison** against the reference must stay green, and `TheDeepSeekRouterWouldGiveDifferentLogits` never touches the library at all; a red comparison means upstream computes something else than Aleph Alpha's reference (router, window boundary, positional encoding) and is a finding, not a test to adjust. The **GGUF format** each test writes -- architecture name, key and tensor names, `expert_gating_func`, pre-tokenizer, optional `output.weight` -- is upstream's converter's decision, not ours: `AfmoeDialectMatchesTheReference` (gating 2) is the likeliest to go red if upstream follows the gating-5 port, `RenormalizedRoutingMatchesTheReference` writes no gating key at all, and `AnotherGatingFunctionIsRejected` assumes a rejection upstream need not make. A red format row is a real signal as well -- the published GGUFs of that dialect stop loading without the patch -- so decide it deliberately (a small compatibility patch, or document that those files must be reconverted) and only then move the row to upstream's format: one `{gating, pre, ...}` entry per test, with the reference untouched. | | `src/test/cpp/test_model_flags.cpp` | 4 | **The contract between the Java CLI-flag registries and llama.cpp's server argument parser.** CMake reads `ModelFlag.java` + `ModelOption.java` (`cmake/extract-java-wire-names.cmake` → a generated header of `{name, contract}` pairs), and this file asserts every `SERVER_PARSER` name is in `common_params_parser_init(params, LLAMA_EXAMPLE_SERVER).options`. It exists because **no Java test can catch this class**: `ModelFlagTest`/`ModelParametersExtendedTest` pin the *string mapping* (`hasKey("--mlock")`), never that llama.cpp still accepts the string, so they stay green forever while the flag is dead — and `common_params_parse` treats an unregistered option as a hard error, so the affected builder method makes the model **unloadable**, not merely ineffective. **A grep over `arg.cpp` is not a substitute**: `--grp-attn-n`/`-w` are present there at every pinned tag but `set_examples()`-scoped to `LLAMA_EXAMPLE_COMPLETION`/`PASSKEY`, so the server parser rejects them exactly like a deleted flag — only the real option table sees that. `--vocab-only` is the one exemption, and it declares itself `CliContract.PROJECT_PSEUDO` on its own constant rather than appearing in a list inside this file; the test asserts such a name is **still unknown** to the parser (an exemption upstream later registers would be hiding a real check) and that the exempt set is non-empty. | | `src/test/cpp/test_wire_contracts.cpp` | 6 | **The same contract for the two quieter surfaces.** `RequestField` against `server_schema::make_llama_cmpl_schema(...)` (5 tests) and `TrainingField` against `jllama_train::config_keys()` (1 test). Both receivers *silently ignore* an unknown key — the schema skips it, `train_engine.cpp` reads with `j.value(key, default)` and falls back — so a dead field produces no error anywhere and every string-mapping test keeps passing. `OAI_LAYER`-declared keys (consumed by `oaicompat_*_params_parse` before the schema) are exempt from the schema check, and are checked **both** ways: still unknown to the schema (the inverted check), and read by at least one upstream reader-shaped site (the configure-time sweep — this is what caught `chat_template`, a key a public builder wrote and nothing read). See [`docs/history/parameter-wire-surface.md`](docs/history/parameter-wire-surface.md). | | `src/test/cpp/test_rpc.cpp` | 28 | **The runnable guard for `patches/0015` and for `rpc_support.hpp`.** The device-selection rules (pure, literal inputs: `--rpc` endpoints accumulate across repeated options, an explicit `--device`/`-dev` is never overridden, stale RPC devices are replaced by the local GPUs chosen the way llama.cpp's default does — duplicates by device id dropped, iGPUs only without a discrete GPU, `none` on a CPU-only host). Then the **real** ggml-rpc client and server over loopback, no model: a `mul_mat` graph computed on the RPC backend equals the local CPU result; `ggml_backend_rpc_stop_server()` ends the server, frees the port and disconnects a client that is still connected; an unreachable or malformed endpoint is `nullptr`/`std::invalid_argument` instead of an abort; a registered server that went away is re-checked on the next registration; its device reports 0/0 memory instead of aborting; and a stale server is kept out of a later load that did not ask for it. The loopback server picks a free port from a range so parallel jobs do not collide, and serves the **CPU** by name — a served GPU that lacks an op aborts the server (see the RPC section). Also: server devices chosen by name (case-insensitive, deduplicated), an unknown name listing the available devices, and a registered RPC device refused; the mmproj device pinned the way clip would choose it minus the stale server; and `exclude_stale_devices()`, the params-level guard for TTS and the trainer. | | `src/test/cpp/test_native_server_shutdown.cpp` | 3 | **The runnable guard for the shutdown-handler guard in `patches/0006`/`0007`.** Runs the real `llama_server()` in router mode over an empty `--models-dir` on an ephemeral loopback port (no model, no worker) and stops it the way `native_server.cpp` does. Pins a clean stop (exit code 0), that `llama_server_request_shutdown()` **after** the server returned is a no-op (the deterministic form of the CI `SIGSEGV` — it crashed every run before the fix), and that requests hammered from another thread while the server tears down are safe (5 rounds). Compiled on non-Android only, like `server.cpp` itself. | -**Current total: 590 tests (all passing).** +**Current total: 598 tests (all passing).** #### Upstream source location (in CMake build tree) -llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11320`. +llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11457`. **GoogleTest** is a separate `BUILD_TESTING`-only FetchContent (`GIT_TAG v1.18.0`), used solely by the `jllama_test` C++ unit-test binary — not by the shipped library, and not coupled to the @@ -2052,7 +2077,7 @@ exceeds `--max-major`: ``` Paths may be jars or directories (searched recursively for `*.jar`), so one invocation covers a whole -artifact set — here the classes jar, all 26 natives jars and every `all--` fat jar. `module-info.class` +artifact set — here the classes jar, all 27 natives jars and every `all--` fat jar. `module-info.class` and `META-INF/versions/**` are skipped unconditionally: a classpath JVM never loads either, which is why a `release 9` `module-info` is fine. `--allow` is a repeatable glob matched against `:` for anything else that must be tolerated. Exit codes: 0 clean, diff --git a/README.md b/README.md index a513b8a7e..57d1ad34f 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ **Build:** ![Java 8+](https://img.shields.io/badge/Java-8%2B-informational) ![Platform](https://img.shields.io/badge/Platform-Linux%20%7C%20macOS%20%7C%20Windows%20%7C%20Android-lightgrey) -[![llama.cpp b11320](https://img.shields.io/badge/llama.cpp-%23b11320-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11320) +[![llama.cpp b11457](https://img.shields.io/badge/llama.cpp-%23b11457-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11457) [![JPMS](https://img.shields.io/badge/JPMS-modular%20JAR-25A162)](https://openjdk.org/projects/jigsaw/) ![JUnit](https://img.shields.io/badge/tested%20with-JUnit6-25A162) [![JSpecify](https://img.shields.io/badge/JSpecify-1.0.0%20%40NullMarked-25A162)](https://jspecify.dev) @@ -111,6 +111,8 @@ Inference of Meta's LLaMA model (and others) in pure C/C++. - **Runtime LoRA adapter control** — list the loaded adapters and change their scales at runtime without reloading the model (`getLoraAdapters()` / `setLoraAdapters(Map)`), the typed counterpart of the upstream `GET`/`POST /lora-adapters` endpoints. - **Text-to-speech** (`TextToSpeech`) over llama.cpp's Qwen3-TTS pipeline (`mtmd_helper::gen_audio`), returning WAV audio. - **In-JVM GGUF quantization** (`LlamaQuantizer`) over llama.cpp's `llama_model_quantize` — convert a GGUF to another quantization scheme without shelling out to `llama-quantize`. +- **Kolibri-1** (Aleph Alpha's German/English reasoning MoE, architecture `kolibri1`) before upstream llama.cpp supports it, through a carried patch that loads the GGUFs of both community converters. See [Kolibri-1](#kolibri-1-aleph-alpha). +- **Decision models** (`handleSystemOne`) — llama.cpp's TypeSafe-compatible `/v1/systemone` API: typed `choice` / `score` / `noul` questions about a state, answered with probabilities in one forward pass, no token generated (laya, julia-1, lev, openjev, kev and the later decision models). See [Decision models](#decision-models-v1systemone). - **Infilling** (fill-in-the-middle) for code models. - **Tokenize / detokenize** and **JSON-schema → grammar** conversion. - **Raw JSON endpoint handlers** mirroring the upstream llama.cpp HTTP server (`/completions`, `/v1/completions`, `/embeddings`, `/infill`, `/tokenize`, `/detokenize`). @@ -232,6 +234,7 @@ platforms you target, e.g. `cpu-linux-x86-64`. | `vulkan-linux-x86-64` | Vulkan | Linux x86-64 with a Vulkan 1.2+ GPU (NVIDIA / AMD / Intel) | A Vulkan runtime (`libvulkan.so.1`), which current GPU drivers install. The most portable Linux GPU option. glibc ≈ 2.39 (built on `ubuntu-latest`). | | `vulkan-linux-aarch64` | Vulkan | Linux aarch64 with a Vulkan 1.2+ GPU | A Vulkan runtime (`libvulkan.so.1`). glibc ≥ 2.39. | | `vulkan-windows-x86-64` | Vulkan | Windows x86-64 with a Vulkan 1.2+ GPU | A Vulkan runtime (`vulkan-1.dll`), which current GPU drivers install. The most portable Windows GPU option. | +| `vulkan-windows-aarch64` | Vulkan | Windows on ARM (Snapdragon X) with a Vulkan 1.2+ GPU | A Vulkan runtime (`vulkan-1.dll`), which current GPU drivers install. Built natively on `windows-11-arm` with `clang-cl`, like upstream's Windows arm64 Vulkan release. | | `opencl-windows-x86-64` | OpenCL | Windows x86-64 with an OpenCL 2.0+ GPU | A vendor OpenCL ICD (`OpenCL.dll`). The GGML OpenCL backend is Adreno-tuned; on desktop GPUs CUDA or Vulkan are better supported. | | `opencl-windows-aarch64` | OpenCL (Adreno) | Windows on ARM (Snapdragon X) | The Adreno driver's OpenCL ICD (`OpenCL.dll`). | | `opencl-android-aarch64` | OpenCL (Adreno) | Android aarch64 with Adreno GPU | A device OpenCL ICD (`libOpenCL.so`); see also the `llama-android-opencl` AAR. | @@ -361,6 +364,7 @@ Every `net.ladenthin.llama.*` system property recognised by the library, deep-sc | `net.ladenthin.llama.vision.model` | `models/SmolVLM-500M-Instruct-Q8_0.gguf` (test self-skips if missing) | test | `MultimodalIntegrationTest` | Path to a vision-capable model GGUF. Any vision-capable GGUF works; CI default is `SmolVLM-500M-Instruct-Q8_0.gguf`. | | `net.ladenthin.llama.vision.mmproj` | `models/mmproj-SmolVLM-500M-Instruct-Q8_0.gguf` (test self-skips if missing) | test | `MultimodalIntegrationTest` | Matching mmproj GGUF for the vision model. | | `net.ladenthin.llama.vision.image` | `llama/src/test/resources/images/test-image.jpg` (a CC-BY-4.0 / MIT-granted photo committed to the repo) | test | `MultimodalIntegrationTest` | Visual prompt image. Any png/jpeg/webp/gif works; the extension drives MIME detection. | +| `net.ladenthin.llama.decision.model` | unset (test self-skips) | test | `SystemOneIntegrationTest` | Path to a decision model GGUF (laya, julia-1, lev, openjev, kev, ...) for the `/v1/systemone` tests of `LlamaModel.handleSystemOne`; upstream tests with `ggml-org/tinylaya-for-testing-gguf`. Without it only the rejection of a non-decision model runs. | | `net.ladenthin.llama.audio.model` | unset (test self-skips) | test | `AudioInputIntegrationTest` (llama.cpp discussion #13759) | Path to an audio-input model GGUF (e.g. Ultravox, Qwen2.5-Omni). | | `net.ladenthin.llama.audio.mmproj` | unset (test self-skips) | test | `AudioInputIntegrationTest` | Matching audio mmproj (encoder) GGUF. | | `net.ladenthin.llama.audio.input` | `src/test/resources/audios/sample.wav` (committed) | test | `AudioInputIntegrationTest` | `.wav`/`.mp3` audio prompt clip; the extension drives format detection. | @@ -613,6 +617,56 @@ try (LlamaModel model = new LlamaModel(modelParams)) { } ``` +### Kolibri-1 (Aleph Alpha) + +[Kolibri-1](https://huggingface.co/Aleph-Alpha/Kolibri-1) is Aleph Alpha's 78B-parameter Mixture-of-Experts +reasoning model for German and English (about 3.5B parameters active per token, Apache-2.0). Upstream llama.cpp +does not support its architecture yet ([ggml-org/llama.cpp#29922](https://github.com/ggml-org/llama.cpp/issues/29922)), +so this library carries it as a patch (`llama/patches/0016-model-kolibri1.patch`) until upstream does. It loads the +community GGUFs of both published converters -- e.g. [Hob-forge/Kolibri-1-GGUF](https://huggingface.co/Hob-forge/Kolibri-1-GGUF) +and [Eliasfpv28/Kolibri-1-Q3_K_S-GGUF](https://huggingface.co/Eliasfpv28/Kolibri-1-Q3_K_S-GGUF) -- which the two +community llama.cpp patches cannot each load from the other. The embedded chat template handles reasoning and +Hermes-style tool calls; for a split GGUF, point the model path at the first part. + +```java +ModelParameters params = new ModelParameters() + .setModel("/models/Kolibri-1-Q4_K_M.gguf") + .setCtxSize(32768) + .enableJinja(); +try (LlamaModel model = new LlamaModel(params)) { + ChatResponse answer = model.chat(ChatRequest.empty() + .appendMessage("user", "Warum ist der Himmel blau?")); +} +``` + +The model is large (the Q4_K_M file is 47.5 GB) and runs from system RAM on the CPU, or partly offloaded to a +GPU. The architecture is checked numerically against Aleph Alpha's reference on tiny random models in every C++ +test run; the real model was not run in this project's CI, and the GPU backends are untested for it. + +### Decision models (`/v1/systemone`) + +A decision model answers typed questions about a state without generating text: each question is +evaluated in one forward pass and returns probabilities. `handleSystemOne` takes and returns +llama.cpp's TypeSafe-compatible `/v1/systemone` JSON, served by the upstream handler itself (the full +request/response description is in upstream's `tools/server/README.md`). The native server serves the +same endpoint at `POST /v1/systemone`, in attach mode too. + +```java +try (LlamaModel model = new LlamaModel(new ModelParameters().setModel("/path/to/laya.gguf"))) { + String answers = model.handleSystemOne("{" + + "\"state\": \"I was charged twice for my order and nobody replied.\"," + + "\"questions\": {" + + " \"route\": {\"type\": \"choice\", \"instructions\": \"Which team?\"," + + " \"criteria\": {\"billing\": null, \"shipping\": null}}," + + " \"urgency\": {\"type\": \"score\", \"instructions\": \"How urgent?\"," + + " \"criteria\": [\"can wait\", \"today\", \"right now\"]}," + + " \"angry\": {\"type\": \"noul\", \"instructions\": \"Is the customer angry?\"}}}"); + // {"answers": {"route": {"choice": "billing", "probabilities": {...}, ...}, ...}, "usage": {...}} +} +``` + +A model that is not a decision model throws a `LlamaException` ("This model is not a decision model"). + ### Runtime LoRA adapter control Adapters loaded at model-load time (`addLoraAdapter(...)` / `addLoraScaledAdapter(...)`, optionally @@ -969,9 +1023,12 @@ client.unloadModel("Qwen3-0.6B-Q4_K_M"); // POST /models/unload ``` `RouterModel` carries the identifier, the lifecycle status -(`UNLOADED`/`LOADING`/`LOADED`/`SLEEPING`/`DOWNLOADING`/`DOWNLOADED`), and the router's -failed-worker marker. Chat requests then select a model per request via the standard -`"model"` field on `POST /v1/chat/completions`. +(`UNLOADED`/`LOADING`/`LOADED`/`SLEEPING`/`DOWNLOADING`/`DOWNLOADED`), the router's +failed-worker marker, and the model's input/output modalities (`getInputModalities()`, +`getOutputModalities()`), which the router computes without loading the model: `isDecisionModel()` +picks out a [decision model](#decision-models-v1systemone) for `/v1/systemone` before its first load. +A loaded `LlamaModel` reports the same through `getModelMeta()`. Chat requests then select a model per +request via the standard `"model"` field on `POST /v1/chat/completions`. Against a router started with `--api-key`, pass the key — it is sent as `Authorization: Bearer ` on every call. All of them need it: `/models/load` and diff --git a/REUSE.toml b/REUSE.toml index 794b8d129..d713a79ee 100644 --- a/REUSE.toml +++ b/REUSE.toml @@ -75,6 +75,19 @@ path = "llama/patches/**" SPDX-FileCopyrightText = "2026 Bernard Ladenthin " SPDX-License-Identifier = "MIT" +# The Kolibri-1 patch derives from two community ports (see its header): the llama.cpp-derived +# parts are MIT, Eliasfpv28's additions Apache-2.0. Aleph Alpha's Apache-2.0 reference served as +# the specification; no code of it is copied. +[[annotations]] +path = "llama/patches/0016-model-kolibri1.patch" +precedence = "override" +SPDX-FileCopyrightText = [ + "2026 Bernard Ladenthin ", + "2023-2026 The ggml authors", + "2026 Eliasfpv28 (Kolibri1 llama.cpp port)", +] +SPDX-License-Identifier = "MIT AND Apache-2.0" + # Project logo / branding assets (binary + generated SVG, cannot carry inline SPDX # without corrupting the base64-embedded font). Same MIT OR Apache-2.0 dual license # the author ships them under in the workspace repo (the embedded Martian Mono font diff --git a/TODO.md b/TODO.md index 55d66b1ad..45b0508da 100644 --- a/TODO.md +++ b/TODO.md @@ -60,6 +60,16 @@ so everything below is genuinely still open. tunnel. Only worth doing if it lands upstream. - **RDMA transport** (`GGML_RPC_RDMA`) as its own classifier, since it needs `libibverbs` at runtime. - **Several clients at once.** Upstream's server serves one connection at a time. +- **Server-to-server comm (`-sm tensor`, llama.cpp b11450) widens what a client can make the server do.** + #26610 lets a client tell two RPC servers to form a pair (`RPC_CMD_COMM_INIT`): the rank-0 server + then *listens on `0.0.0.0`* on a port the client names (`socket_t::create_server("0.0.0.0", port)` + in `rpc_server::comm_init`) and blocks in `accept()` until the rank-1 server connects. Two + consequences for the in-JVM `RpcServer`: (1) `startLocal` binds loopback only, but a local client can + still open a listener on every interface; (2) `close()` cannot end that wait -- + `ggml_backend_rpc_stop_server()` shuts down the *client* socket and wakes the *main* listener, not + the comm listener -- so if the peer never connects, the server thread and `close()` hang. Not + reproduced; found reading the diff at the bump. Fix candidates for `0015`: bind the comm listener to + the server's own host and register it with the stop machinery so a stop also wakes it. ### Logging sink (`patches/0014`) — follow-ups @@ -104,6 +114,32 @@ answered, read→write→read loop changed the file). Still open: - **Model recommendation table** for the agent (which local GGUFs actually complete an edit→build→test loop) — needs a GPU host, not CI. +### NativeServer attach mode leaves a sleep callback behind (found at the b11361 bump, not reproduced) + +`llama_server_attach` (`patches/0007`) builds a `server_routes` on its own stack frame over the +`LlamaModel`'s `server_context`. Its constructor registers a sleeping-state callback on the model's +queue (`server_queue::on_sleeping_state` only appends, there is no unregister), and that callback +captures the `server_routes`. When the attached `NativeServer` is closed, the frame returns and the +object is gone, but the callback stays in the queue of the model, which lives on. The next time that +model enters idle sleep, the callback runs on a destroyed object. Reachable only with a model loaded +with `--sleep-idle-seconds` that was served by an attached `NativeServer` and then kept in use after +the server closed. Since b11361 `LlamaModel` holds a `server_routes` of its own for its whole lifetime +(`jllama_context::routes`, for `handleSystemOne`); the natural fix is to let attach mode serve +*that* object instead of building a second one, which changes `llama_server_attach`'s signature in +`0007` and `native_server.cpp`. Needs a test with sleep enabled (`IdleSleepWakeIntegrationTest` is the +template) before the fix, to show it red first. + +### Router workers print the backend line onto the router's command pipe (cosmetic, since b11401) + +Since llama.cpp b11401 (#29895) a router child keeps its stdout for the state commands to the router +and redirects everything else written to stdout to stderr -- but only once `llama_server()` starts. +A JVM worker (`NativeServer.setWorkerCommand`, `patches/0008`) prints `LlamaLoader`'s +`[jllama] using native backend '...'` line to `System.out` before that, so the router logs it as +`unexpected output on the command pipe`. Harmless (the router warns and goes on), but misleading. +Moving the line to `System.err` would fix it; three smoke scripts grep for it +(`smoke-test-fatjar.sh` reads both streams, `smoke-rpc-fatjar.sh` and `smoke-natives-jars.sh` need +checking first), so it is not a one-line change. + ### LlamaLoader extraction-directory isolation (optional follow-up, low priority) Left over from the 2026-06-20 code audit (18/18 findings fixed in PRs #258/#260, regression tests in @@ -135,6 +171,12 @@ round-trips — see CLAUDE.md "Two server modes"). **Owner priority: the native- `/infill` applies the model's FIM tokens server-side, so low value. - **Multi-model registry (Java transport).** The native surface has this via router mode + `RouterClient`; the Java `OpenAiCompatServer` still advertises/serves a single model id. +- **400 vs. 500 for an invalid request body.** Since llama.cpp b11337 (#29060) upstream's server + answers a malformed or empty embedding `"prompt"` (and any `common_json_error`) with 400. The JNI + layer throws a plain `LlamaException` for both, and `LlamaModelBackend` does not translate it into + the `IllegalArgumentException` that `completeNonStreaming` maps to 400, so `OpenAiCompatServer` + answers 500. A fix needs a typed signal from native (an invalid-request exception subclass, or the + `throw_invalid_request` JSON shape parsed on the Java side) rather than message matching. - **Manual real-client validation.** Server-side round-trips exist for every surface; what remains is pointing the actual editor clients (Copilot Ollama provider / Custom Endpoint, Claude Code, a Responses client) at a running server, since round-trips confirm wire shapes but not each client's @@ -189,6 +231,19 @@ from the bump checklist. The exception is **`0003`**, a carry of upstream PR #22 **closed without merging** — it is permanent and will never be droppable via a bump. (`0003` used to be described here as "drops automatically when that merges"; it will not.) +**`0016` (Kolibri-1) is not a submission candidate but a temporary carry**: upstream will add the +architecture itself (request [ggml-org/llama.cpp#29922](https://github.com/ggml-org/llama.cpp/issues/29922)). +Drop it on the first bump whose tag registers `kolibri1` (`git grep -n kolibri src/llama-arch.cpp`), +keep `src/test/cpp/test_kolibri1.cpp` (it compiles without the patch). Its **numerical comparisons** +must stay green -- red there means upstream computes something else than Aleph Alpha's reference, a +finding to report, not a test to adjust. Its **GGUF-format rows** (gating function 2 *and* 5, no gating +key, pre-tokenizer `qwen2` *and* `kolibri1`, rejection of gating 1) follow whatever format upstream's +converter fixes: a red row means the published GGUFs of that dialect stop loading without the patch. +Decide that deliberately -- keep a small compatibility patch, or document that those files must be +reconverted (and say so upstream rather than lose them silently) -- and only then move the row's +`{gating, pre, ...}` entry to upstream's format. Open verification gaps of the carry: +no run of the real 78B model and no GPU backend from here (see the patch header). + - **`0001` Windows arg-parse embed guard** (against #24779): `common_params_parse` trusts the caller's argv; `common_params_parse_main()` keeps the standalone tools' UTF-8 recovery. Ship with the standalone-safe repro (synthetic argv discarded on Windows because `GetCommandLineW()` returns the diff --git a/docs/history/llama-cpp-breaking-changes.md b/docs/history/llama-cpp-breaking-changes.md index 1454fc832..8db2a9660 100644 --- a/docs/history/llama-cpp-breaking-changes.md +++ b/docs/history/llama-cpp-breaking-changes.md @@ -786,3 +786,45 @@ Used during `llama.cpp` version bumps: when upgrading, scan this file from the r | b11303–b11313 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11313 with `git apply`. Patch-target file in the range: `common/arg.cpp` (`0001`, `0015`; #28977 adds three lines in `common_models_handler_init`, away from both). Drop-checks against pristine b11313: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override and `common_params_parse_main` is absent (`0001` needed); `load_progress_callback` still assigned unconditionally (`0002` needed); no `split_sum == 0` guard in `load_tensors` (`0012` needed); no callback hook in `common/log.h` (`0014` needed); `add_server` still `GGML_ABORT`s on a failed connect and there is no stop function (`0015` needed); no upstream counterpart to `llama_server_attach`, `LLAMA_SERVER_WORKER_CMD` or the embedded-shutdown hook (`0006`-`0008` needed). `tools/server/` is unchanged in the range, so the request-schema, bound and response-key sets are identical. | | b11313–b11320 | Seven commits, 18 files. LLM-jp-4.1 Harmony dialect handler (#29681, `common/parsers/*` + a chat template), PLaMo-2/3 honour the BOS/EOS settings (#29734), Metal bf16 math for mxfp4 mul-mat (#29770), `llama-bench` fixes. The +16k/-9k line count is `docs/ops/CPU.csv` (#29666, a regenerated ops matrix, ~4 MB of diff) -- which is why `llama-next-version.sh`'s byte threshold reads this range as one 4.8 MB step; excluding `docs/ops` it is 53 KiB. **No project source change.** | | b11313–b11320 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11320 with `git apply`. No patch-target file in the range. Drop-checks against pristine b11320: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override and `common_params_parse_main` is absent (`0001` needed); `load_progress_callback` still assigned unconditionally (`0002` needed); no `split_sum == 0` guard in `load_tensors` (`0012` needed); no callback hook in `common/log.h` (`0014` needed); `add_server` still `GGML_ABORT`s on a failed connect and there is no stop function (`0015` needed); no upstream counterpart to `llama_server_attach`, `LLAMA_SERVER_WORKER_CMD` or the embedded-shutdown hook (`0006`-`0008` needed). `tools/server/` is unchanged in the range, so the request-schema, bound and response-key sets are identical. | +| b11320–b11327 | Seven commits, 19 files, ~296/52 lines. **#29773** (mtmd) changes the return type of `mtmd_get_memory_usage()` (`tools/mtmd/mtmd.h`) from a bare `std::map` to `struct mtmd_memory_usage { backend_mem_usage, image_max_tokens, use_non_causal }`, and the server now *always* measures an mmproj (no longer only under `fit_params`) to cap `image_max_tokens` to `n_ubatch` for a non-causal vision encoder, warning "increase n_ubatch (-ub) to increase vision token budget". The function is upstream's "unstable, used internally by fit_params" API and the project never calls it -- `MultimodalIntegrationTest`'s server path gets the cap for free. Rest: Hexagon work-queue race, HIP CDNA FA guard, direct-io mmap copy, jinja loop-scope copy, meta AllReduce FILL, AOCL-BLAS label. **No project source change.** | +| b11320–b11327 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11327 with `git apply`. Patch-target file in the range: `tools/server/server-context.cpp` (`0002`, `0003`; #29773 edits `load_model`'s mmproj block, away from both hunks) and `tools/mtmd/mtmd-cli.cpp` (`0001`, unchanged call site). Drop-checks against pristine b11327: `0001` win32 override present, `common_params_parse_main` absent; `load_progress_callback` still assigned unconditionally (`0002`); no slot-similarity getter (`0003`); no attach / worker-command / embedded hook (`0006`-`0008`); no `split_sum == 0` guard (`0012`); `common/log.h` has no callback hook (`0014`); no `stop_server`, connect still `GGML_ABORT`s (`0015`). Request-schema fields, bounds and response keys unchanged. | +| b11327–b11337 | Ten commits, 37 files, ~796/667 lines, nearly all backend/model. **#29060** (server) makes `tokenize_input_subprompt` / `tokenize_input_prompts` (`tools/server/server-common.cpp`) throw `std::invalid_argument` instead of `std::runtime_error` for a malformed or empty `"prompt"`, and `server.cpp`'s `ex_wrapper` now maps `common_json_error` to 400 as well -- upstream's `/embeddings` answers 400 instead of 500 for both. `jllama.cpp` calls `tokenize_input_prompts` in three places (completion, infill, embeddings) and catches `std::exception` in each, so the JNI surface is unchanged (`LlamaException` with the same message); `OpenAiCompatServer` still answers such a request with 500, since `LlamaModelBackend` does not translate a `LlamaException` into the `IllegalArgumentException` its 400 path expects (on file in `TODO.md`). Also: Qwen4Exp MTP heads (#29761, speculative decoding for that architecture -- `common/speculative.cpp` one line) and a qwen4exp fix (#29751), CUDA CCCL made configurable and pinned to 3.4.3 in upstream's CUDA release jobs (#29792: `GGML_CUDA_CCCL_VERSION` replaces `GGML_CUDA_CUB_3DOT2`; CUB `DeviceTopK` needs >= 3.4.3 and falls back to a sort below -- adopted for both CUDA builds here in a follow-up commit of the same bump, after the b11374 OpenVINO check turned it up), NVFP4 cuBLAS compute type, a recurrent-memory assert fix (#29799), webgpu bf16 mat-mul, Metal transfer-buffer release, CUDA sm70 MMVQ table. **No project source change.** | +| b11327–b11337 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11337 with `git apply`. Patch-target files in the range: `tools/server/server.cpp` (`0006`, `0007`; #29060 adds a `catch` arm to `ex_wrapper`, above the route table `0007` extracts) and `tools/server/server-common.cpp` (no patch hunk). Drop-checks against pristine b11337 unchanged from b11327: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11337–b11355 | Eighteen commits, 58 files, ~2301/575 lines, backend-only for this project. **#23671** adds `alloc_buffer_n` to the buffer-type interface (`ggml_backend_buft_alloc_buffer_n`, `ggml/include/ggml-backend.h` + `ggml-alloc.h`; a `NULL` entry in every existing buffer type, including `ggml-rpc.cpp`'s), which the project neither implements nor calls. **#29816** adds `common_create_directories()` (`common/common.h`) to work around old libstdc++ not following a symlink in `create_directories` (GCC bug 101510) and uses it for the cache directories in `common` and `tools/rpc/rpc-server.cpp`; `RpcServer` creates its tensor-cache directory on the Java side (`Files.createDirectories`, which follows symlinks) and passes it to `ggml_backend_rpc_start_server`, so it is not affected. Rest: qwen4exp mask/test work, Hexagon q2_k/q3_k + strided DMA, CUDA Volta FA, Vulkan pipeline logging and a Samsung 32 KB tile guard, SYCL oneDNN/FA work, the k-pool re-pool clamp (#29805). **No project source change.** | +| b11337–b11355 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11355 with `git apply`. Patch-target files in the range: `ggml/src/ggml-rpc/ggml-rpc.cpp` (`0015`; #23671 adds the `alloc_buffer_n` slot to the RPC buffer type, away from the hunks) and `src/llama-model.cpp` (`0012`; a comment-only change by #29074). Drop-checks against pristine b11355 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11355–b11361 | Six commits, 45 files, ~2339/37 lines. **#29818 adds the `/v1/systemone` API** (TypeSafe-compatible decision models: laya, julia-1, lev, openjev, kev): new `tools/server/server-decision.{cpp,h}` (`server_decision_context` -- question parsing, per-model prompt layout, answer formatting), `SERVER_TASK_TYPE_DECISION` + `server_task::decision` + `server_task_result_decision` (`to_json()` emits the new response key `scores` with `index` and `tokens_evaluated`), a `server_context_impl::decision` member initialised on every load (`decision.init(model_tgt)`, a load failure if the metadata is malformed), `server_routes::post_systemone`, the route `POST /v1/systemone` in `server.cpp` (and its router-mode proxy), `common_decision_type` + `common_get_decision_type()` in `common/common.h`, and `handle_media()` made non-static in `server-common.h`. **Three project changes:** (1) `server-decision.cpp` added to `jllama`'s and `jllama_test`'s sources (`server-context.cpp` holds the member, so leaving it out is an undefined reference -- latent on Linux, fatal on macOS/Windows); (2) `0007` refreshed (below); (3) the feature itself, as `LlamaModel.handleSystemOne(String)`: the decision context is private to `server_context_impl`, so instead of re-implementing the handler the JNI method calls upstream's `server_routes::post_systemone` directly. `jllama_context` now holds a `server_routes` for its lifetime, constructed **before** `load_model()` as upstream's `llama_server()` does (its constructor registers a sleeping-state callback that must run before the server's own, which frees the model) and given its metadata with `update_meta()` after the load; a non-200 answer becomes a `LlamaException` with upstream's message via the new pure helper `route_error_message` (`json_helpers.hpp`, 3 C++ tests). `SystemOneIntegrationTest` pins the rejection of a non-decision model with the CI draft model; its decision-model cases need `-Dnet.ladenthin.llama.decision.model` (upstream tests with `ggml-org/tinylaya-for-testing-gguf`, which is not in `models.csv`). `NativeServer` serves the route in classic mode and, through `0007`'s shared route table, in attach mode. Found while wiring it and put in `TODO.md`: attach mode's own `server_routes` leaves its sleep callback in the model's queue when the server closes. Also in the range: #29840 turns `llama_load_mode_from_str()`'s `throw` on an unknown mode into a `GGML_ABORT` -- nothing in `common` or this project calls it (`--load-mode` is parsed by `arg.cpp` itself, and `LoadMode` is an enum); #29841 removes `fs_open_ifstream()` (unused here). | +| b11355–b11361 | patches + upstream verification | **`0007` breaks at exactly #29818 (`a4cb4c61f`, = b11361)** -- every patch applies to its parent `70849ee82`, and `0007` alone fails on the commit, at `tools/server/server.cpp:305`. The cause is the one line #29818 adds to the route table `0007` moves out of `llama_server()`: `ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone));` after `/v1/reranking`. **Refreshed** by adding that line on both sides -- the removal from `llama_server()` and the extracted `llama_server_register_common_routes()` -- so the helper stays a verbatim copy of the table it replaces (checked mechanically: the 42 helper lines equal the 42 table lines of the patched-up-to-`0006` b11361 tree). The regenerated patch differs from the old one in exactly those two lines. The other eight apply unchanged. Drop-checks against pristine b11361 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields and bounds unchanged; response keys gain `scores` (the decision result). | +| b11361–b11368 | Seven commits, 25 files, ~984/31 lines. **#27694 adds probabilistic draft sampling** for the simple-draft and MTP speculative paths: a new server option `--spec-draft-sampling {greedy,probabilistic}` (`LLAMA_ARG_SPEC_DRAFT_SAMPLING`, default `greedy`) sets the new `common_params_speculative::draft.probabilistic`; with it the drafter samples and fills a per-token candidate distribution (`common_speculative_draft_params::result_q`, plus the target's `temp`/`seed`), and the server verifies with the new `common_sampler_sample_and_accept_n_rejection()` whenever the request's temperature is above zero (a replay after a checkpoint restore re-accepts the drafted tokens instead of verifying again). No request-schema field: the mode is per server. **Exposed as `ModelParameters.setDraftSampling(DraftSampling)`** -- a new `args.DraftSampling` enum (`GREEDY`, `PROBABILISTIC`) and the `ModelOption.SPEC_DRAFT_SAMPLING` constant, whose `SERVER_PARSER` contract `test_model_flags.cpp` checks against the real server option table; `setDraftSampling` joins the `OCP_OVERLY_CONCRETE_PARAMETER` design-intent list in `spotbugs-exclude.xml` beside `setLazyMode`. #29844 adds the nimble decision model (`COMMON_DECISION_TYPE_NIMBLE`, a prompt listing every question of the request) to the `/v1/systemone` code -- served by `handleSystemOne` without a change. Rest: Metal tensor-API FA kernel for F16 KV, CPU `soft_max_back` aliasing fix, qkx3 scale rounding, a CUDA transposed-copy fix. The project calls none of the speculative/sampling functions whose parameters grew. Also fixed here: the README badge's label had stayed at `b11320` since the b11327 step (the pin edit matched `b11320` at a word boundary, and `%23b11320` has none); it now names b11368 like its link. | +| b11361–b11368 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11368 with `git apply`. Patch-target files in the range: `common/arg.cpp` (`0001`, `0015`; #27694 adds the option far from both) and `tools/server/server-context.cpp` (`0002`, `0003`; the speculative and decision changes are away from their hunks). Drop-checks against pristine b11368 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11368–b11371 | Three commits, 30 files, ~1499/31 lines. **#29831 adds the clef decision model** (text-only): `COMMON_DECISION_TYPE_CLEF`, a *joint* head that answers every question of a request in one prompt (`server_decision_context::is_joint()` / `fill_task_joint()`, `server_task::decision::order` + `n_scores`, an internal `decision_order` per batch token), plus the model graph in `src/`. Reached through `/v1/systemone`, so `LlamaModel.handleSystemOne` serves it unchanged. Rest: CUDA shared-expert fusion into MMVQ, a CI runner change. **No project source change.** | +| b11368–b11371 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11371 with `git apply`. Patch-target files in the range: `src/llama-model.cpp` (`0012`; the clef architecture is added away from `load_tensors`' split block) and `tools/server/server-context.cpp` (`0002`, `0003`; decision changes away from their hunks). Drop-checks against pristine b11371 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11371–b11374 | Three commits, 64 files, ~3673/864 lines. **#29852 moves ggml-openvino to OpenVINO 2026.4.1** (MoE/GDN fusions, a compiled-model disk cache, device listing in which only the device selected by `GGML_OPENVINO_DEVICE` reports as GPU and the others as IGPU so llama.cpp does not offload to them, the allocation limit reported to ggml) and changes upstream's `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` from `2026.4` / `2026.4.0.22959.99c81491cc3` to `2026.4.1` / `2026.4.1.22982.07f9c262b05`. **Both OpenVINO build jobs follow** (`build-linux-x86_64-openvino`, `build-windows-x86_64-openvino`: step names, the `KEEP IN SYNC WITH UPSTREAM` comment and the archive URLs, built from upstream's own `linux-setup-openvino` / `windows-setup-openvino` URL templates); the runbook now lists the vendor settings that track upstream's CI. The backend's CMake adds `-Wno-pedantic` / `-Wno-gnu-zero-variadic-macro-arguments` and links `psapi` on Windows (a system DLL; the OpenVINO directory is held only to the GPU denylist by `verify-native-deps.py`). #29862 registers the LFM2.5 encoder models (conversion only); #29825 halves qwen4exp's indexer memory. **No project source change.** | +| b11371–b11374 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11374 with `git apply`. No patch-target file in the range. Drop-checks against pristine b11374 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11374–b11382 | Eight commits, 17 files, ~536/211 lines. **#29903** (`common/common.cpp`): with embeddings enabled, `common_init_from_params` now caps `n_batch` to `n_ubatch` (warning "embeddings enabled: setting n_batch = n_ubatch") -- an embedding needs its whole input in one ubatch, and the check `server.cpp`'s `main()` made for `--embedding` ran before the load and never covered this layer, which loads through `server_context::load_model` directly. A model loaded with `enableEmbedding()` / `enableReranking()` therefore now runs with `n_batch == n_ubatch` (512 by default) instead of 2048; the input limit is `n_ubatch` either way, so only the logical batch shrinks. **#29886 updates the vendored cpp-httplib from 0.58.0 to 0.59.0** (compiled into `jllama` for the native server; no new include or link dependency). #29860 adds `common_is_tty()` (`common/common.h`) and moves `common/log.cpp`'s Windows `isatty` shims there; #29813 honours `json_schema` in the Ling 3.0 chat parser; #29856 gathers the recurrent states once per graph reserve; #29863 an mtmd `strdup` warning. **No project source change.** | +| b11374–b11382 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11382 with `git apply`. Patch-target file in the range: `common/log.cpp` (`0014`; #29860 removes the Windows `isatty`/`fileno` block at the top of the file, away from the worker and sink hunks). Drop-checks against pristine b11382 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook (`common/log.h` still has only `common_log_default_callback`), no `0015` stop function. `tools/server/` unchanged in the range. | +| b11382–b11388 | Five commits plus two CI ones, 15 files, ~1654/763 lines. #29938 fixes router presets: the allow-list named the dead `LLAMA_ARG_HF_REPO_FILE` key instead of `LLAMA_ARG_HF_FILE`, so `hf-file` in a preset was dropped (`tools/server/server-models.cpp`). #29674 rewrites `load_from_models_dir()` (`common/preset.cpp`) and removes `fs_list()` / `common_file_info` from `common/common.h` -- the project uses neither. #29924 fixes n-gram drafts being rejected at temperature > 0 after a truncation (the candidate distribution was not truncated with the draft; a follow-up to #27694). #14891 adds activation-based statistics to GGUF imatrices: `--nextn` (imatrix-only, `common_params::load_mtp`), `common_speculative_are_compatible()` declared in `common/speculative.h`, the new `common/imatrix-loader.{h,cpp}`. **No project source change.** | +| b11382–b11388 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11388 with `git apply`. Patch-target files in the range: `common/arg.cpp` (`0001`, `0015`; #14891 adds `--nextn` among the imatrix options), `tools/imatrix/imatrix.cpp` (`0001`'s `main()` flip, still in place) and `tools/server/server-models.cpp` (`0008`; the one-line allow-list fix is away from its hunk). Drop-checks against pristine b11388 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11388–b11395 | Seven commits, 62 files, ~333/106 lines, mostly upstream CI. **#29954 adds a Windows arm64 Vulkan build to upstream's release set** (`release.yml` / `build-vulkan.yml`: the x64 LunarG installer with the `com.lunarg.vulkan.arm64` component, `cmake/arm64-windows-llvm.cmake`, `GGML_VULKAN=ON`). This project ships no such natives jar yet -- **added as `vulkan-windows-aarch64` in the next commit**. #29942 clears `current_tool` when the pending tool call of the PEG chat parser is reset (`common/chat-peg-parser.cpp`, one line -- a stale tool name could leak into the next streamed tool call). Rest: a CUDA MMQ memory fault with `n_expert >> n_ubatch` (#29941), RDNA4 Vulkan mat-vec tuning, CUDA refactors, CI permissions, an `AGENTS.md` revamp. **No project source change.** | +| b11388–b11395 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11395 with `git apply`. No patch-target file in the range. Drop-checks against pristine b11395 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11395–b11400 | Five commits plus a CI one, 36 files, ~772/202 lines, all inside `src/` and `ggml/`. **#29622 lets one `llama_batch` mix embedding rows and raw tokens** for the architectures that support it (`llm_arch_supports_mixed_batch`, a dedicated mixed graph input, the m-rope position handling consolidated) -- an internal change behind an unchanged `include/llama.h`; the multimodal prompt path (`mtmd` image embeddings followed by text) is its consumer. Rest: tinyBLAS BF16/FP16/FP32 K tails on x86, CUDA swizzle refactor and two `move ... to where it is used` cleanups, the Windows LLVM build on Ninja Multi-Config in upstream CI. **No project source change.** | +| b11395–b11400 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11400 with `git apply`. No patch-target file in the range (`src/llama-model.cpp` untouched). Drop-checks against pristine b11400 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11400–b11401 | One commit, 9 files, ~194/73 lines: **#29895, "log, server: self contained colors, split child commands from logs in router mode"**. (1) `common_log` writes the colour reset *before* a line's trailing newlines, enables ANSI on a Windows console through the new `tty_enable_ansi()` (`common/common.h`) and keeps colours off where a console cannot render them, and gains `common_log_get_colors()` (`common/log.h`) -- **no callback hook**, so `0014` stays. (2) Router mode separates a child's state commands from its logs: `server_child` gets a constructor that, in a child (`LLAMA_SERVER_ROUTER_PORT` set), reserves a duplicate of stdout for the commands and `dup2`s stderr over stdout (`server_reserve_stdout()`, `server-common.{h,cpp}`); the router reads both pipes (`server_subproc::read_output(stream, ...)`, `output_closed()`), forwards stderr as the log and warns about any other line on the command pipe; the router passes its colour setting to its children. `llama_server()` gains a `server_child &` overload and the argv entry point now constructs the `server_child` first. **Effects on this project:** an embedded `NativeServer` is not a router child, so nothing is redirected in the host JVM; a router *worker* JVM (`0008`) is one, and from the moment `llama_server()` starts its `System.out` goes to stderr -- which the router logs, as before. The one visible change: `LlamaLoader`'s `[jllama] using native backend` line is printed before that, so the router now reports it as "unexpected output on the command pipe" (cosmetic; on file in `TODO.md`). **No project source change.** | +| b11400–b11401 | patches + upstream verification | **`0007` and `0014` break at exactly #29895 (`a7fb71fab`, = b11401)** -- the full stack applies to its parent b11400. **`0007`:** the two hunks that add the `llama_server_attach` declaration and the extracted route-table helper lost their context (a new `server_child &` overload among the declarations; `server_child child;` at the top of `llama_server(int, char **)`); re-applied at the same places -- the declaration after the new overload, the helper before the argv entry point -- and the helper checked again to be a verbatim copy of the 42-line table it replaces. **`0014`:** the hunk adding the sink's two members and the one adding `set_callback()` lost theirs to the new `colors` member and `get_colors()`; re-applied next to them. For both patches the `+`/`-` lines of the regenerated diff are identical to the old ones (checked mechanically); `0014` keeps its commit-message header. The other seven apply unchanged, and the whole stack applies through b11418. Drop-checks against pristine b11401: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, **no `0014` hook** (`common/log.h` adds `common_log_get_colors` only), no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11401–b11414 | Thirteen commits, 46 files, ~2871/383 lines. **#28498 is a state-file format break:** a state file now records the exact KV-cache rotation and a restore under a mismatched rotation is rejected, which bumps `LLAMA_SESSION_VERSION` 10 → 11 and `LLAMA_STATE_SEQ_VERSION` 3 → 4 (`include/llama.h`; the two constants are its whole diff there). As at b10642, a slot file written by `LlamaModel.saveSlot` (or `/slots/{id}?action=save`) with an earlier jar no longer restores and has to be regenerated; `saveSlot`'s Javadoc now names both bumps, and the CHANGELOG says so. No signature changed. #29958 fixes an unexpected graph reallocation in the k-pool models. Rest: Metal few-row MMA mat-mul (a new `kernels/mul_mv_mma.metal` in the Metal CMake list), CUDA FA scheduling and thin f16/bf16 MMVF, CUDA lightning-indexer tiling, webgpu MMVQ types, a SpacemiT Q8_0 kernel, Vulkan sparse FA for quantized K/V and a stale-prealloc fix, CI. **No project source change** beyond the Javadoc. | +| b11401–b11414 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11414 with `git apply`. Patch-target files in the range: `tests/test-state-restore-fragmented.cpp` (`0001`'s `main()` flip, still applying -- #28498 moves its rotation test into `test-save-load-state.cpp`, the file `0001` already leaves alone) and `tests/CMakeLists.txt` (`0012`'s test registration, away from the changed lines). Drop-checks against pristine b11414 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11414–b11418 | Three commits plus a CI one, 9 files, ~289/86 lines. **#29969 adds vision input for the clef decision model** (`/v1/systemone` with `images`; `server-decision.{h,cpp}` and the clef prompt in `server-context.cpp`). `LlamaModel.handleSystemOne` serves it unchanged: the upstream handler checks `meta->has_inp_image`, which the `server_routes` the context holds takes from the metadata of the load, so a clef model loaded with `--mmproj` accepts images and one without answers upstream's 501 text as a `LlamaException`. **#24076** fixes the prompt-cache cut check for multimodal prompts (`server_tokens::keep_first` now requires the cut to sit *at* a chunk boundary, `find_chunk(n)` instead of `find_chunk(n - 1)`), so a cached prefix can no longer be truncated in the middle of an image chunk. Rest: CUDA NVFP4 mmq accumulation. **No project source change.** | +| b11414–b11418 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11418 with `git apply`. Patch-target file in the range: `tools/server/server-context.cpp` (`0002`, `0003`; the clef vision prompt is away from their hunks). Drop-checks against pristine b11418 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11418–b11429 | Nine commits, 22 files, ~306/94 lines. **#29987 (`4d60b4d08`, untagged) reports each model's input and output modalities in `GET /models`**: a new `architecture` object `{input_modalities, output_modalities}` per entry, from `server_model_architecture_json()` / `server_model_output_modalities()` (`server-common.{h,cpp}`) -- `["decisions"]` for a native decision model, `["text"]` otherwise; `server_context_meta::model_output_modalities`; and in router mode computed **offline** (`server_model_meta::update_caps(const common_params &)`, `common_get_decision_type(fname)` reading the GGUF without loading it), cached across sleep and unload. The key is emitted from `server-common.cpp`, which is why the response-key sweep over `server-task.cpp` does not list it. **Exposed in Java:** `RouterModel` gains `getInputModalities()` / `getOutputModalities()` / `isDecisionModel()` (a 7-argument constructor; the 5-argument one stays and reports none), parsed from `architecture` by `RouterModelsResponseParser` (absent = empty, as upstream asks of clients); `getModelMeta()` carries `input_modalities` / `output_modalities` (built natively with the same upstream helper, flattened because `architecture` already names the GGUF architecture there) and `ModelMeta` the same three getters plus the constant `OUTPUT_MODALITY_DECISIONS`. Tests: `RouterModelTest`, `RouterModelsResponseParserTest`, `ModelMetaTest` (model-free) and `LlamaModelTest#testGetModelMeta` (CodeLlama reports `["text"]` / `["text"]`). Also in the range: `common/arg.cpp` clears only the model's path and repos for router mode instead of the whole `common_params_model`; ggml 0.26.0 and llama.cpp **0.6.0** (`LLAMA_VERSION_MINOR` 5 → 6 in the root `CMakeLists.txt`; `b11429` is also tagged `v0.6.0` -- the build-info string `LlamaCppVersion` is checked against still reads `b-`); Vulkan FA shmem and a reverted `mul_mat_id` tile selection; CUDA alloc-deps and MUSA indexer fixes; a WebUI number-format change (picked up by `build-webui`). | +| b11418–b11429 | patches + upstream verification | **`0008` breaks at exactly #29987 (`4d60b4d08`)**, the untagged commit right before b11429 -- the full stack applies to its parent `c06f84160`, and `0008` alone fails on it, at `tools/server/server-models.cpp:215`. The commit moves `server_model_meta::update_args` (to ~line 530) and renames the function after it to `update_caps(const common_params &)`, i.e. the hunk's trailing context. **Refreshed** by re-applying the hunk at the end of `update_args`, after the `LLAMA_APP_CMD` re-injection as before; the `+` lines are unchanged (checked mechanically), and the patch is now a full `git diff` (it was a bare `---`/`+++` diff). Since b11429 is the first tag after the break, the pin moves straight to it. The other eight apply unchanged, the stack through b11449. Drop-checks against pristine b11429 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart (still no `LLAMA_SERVER_WORKER_CMD` upstream), no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields and bounds unchanged. | +| b11429–b11440 | Eleven commits plus CI, 44 files, ~2562/994 lines. #29943 moves the scheduler's selective expert copying out of ggml into llama via a new, additive `ggml_backend_sched_set_copy_callback()` (`ggml/include/ggml-backend.h`; the project does not drive a scheduler itself). #29994 fixes a data race in the k-pool scatter on shared sequences; #30020 re-reserves the scheduler when the nextn extraction flags change. #30019 bumps the vendored LibreSSL to 4.3.3 in `vendor/cpp-httplib/CMakeLists.txt` -- not built here (Windows builds BoringSSL via `LLAMA_BUILD_BORINGSSL`, the rest no SSL). Rest: Hexagon matmul/FA scalability, pool op and ssm-conv; OpenVINO CI and GPU regression fixes on 2026.4.1 (#30037); a debug-only HIP `-O0` for host code; `test-llama-archs` backend init. **No project source change.** | +| b11429–b11440 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11440 with `git apply`. No patch-target file in the range. Drop-checks against pristine b11440 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11440–b11447 | Seven commits plus CI, 36 files, ~205/95 lines. **#30044 adds the pplx-decider decision model** (`COMMON_DECISION_TYPE_PPLX_DECIDER`: openjev-style labels with codes of one or two letters, a shared prompt prefix, image input): `server-decision.{h,cpp}`, and `server_model_output_modalities()` reports it as `["decisions"]`. `LlamaModel.handleSystemOne`, `ModelMeta.isDecisionModel()` and `RouterModel.isDecisionModel()` cover it without a change -- all three take the type from upstream. Rest: Metal excess threadgroup memory in quantized FA (#29340), a null `vkEnumerateInstanceVersion` check in Vulkan (#29872), the nextn row-cropping helpers, transformers-5.18 conversion fallback, `apiabi` checks narrowed to libllama/libmtmd. **No project source change.** | +| b11440–b11447 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11447 with `git apply`. No patch-target file in the range. Drop-checks against pristine b11447 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, no `0015` stop function. Request-schema fields, bounds and response keys unchanged. | +| b11447–b11449 | Two commits, 4 files, ~101/36 lines: CUDA BF16/FP16 → F32 conversion in chunks (#29442) and `CLAMP` on non-contiguous views on CPU and CUDA (#29517). **No project source change.** | +| b11447–b11449 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11449 with `git apply` -- the last tag before `0015` breaks. No patch-target file in the range; nothing under `common/`, `tools/`, `include/` or `ggml/include/` changed, so the drop-checks are those of b11447. | +| b11449–b11450 | One commit, 3 files, ~755/48 lines: **#26610, "RPC: add `-sm tensor`"**. Tensor parallelism (`--split-mode tensor`) across RPC servers: graphs are stored per uid on the server and computed asynchronously (`ggml_backend_graph_compute_async`, with a `sync_all_backends()` before every buffer read or write), 2-D tensor get/set, a busy-spinning dispatcher while commands keep coming, and a **server-to-server comm**: `RPC_CMD_COMM_INIT` pairs two servers (rank 0 listens on `0.0.0.0` on a client-named port, rank 1 connects with retries) and `RPC_CMD_COMM_ALLREDUCE` reduces partial results between them (in bf16 above 32k elements). **`RPC_PROTO_MAJOR_VERSION` 7 → 8** (`ggml/include/ggml-rpc.h`): peers of different major versions no longer talk -- `rpc-server`s and `RpcServer` JVMs have to be upgraded together (CHANGELOG). **Exposed in Java:** `GpuSplitMode.TENSOR` (`--split-mode tensor`), which upstream accepted for longer but the enum lacked and which now spans RPC servers too. **Found and put in `TODO.md`:** a client can make an in-JVM `RpcServer` started with `startLocal` listen on every interface (the rank-0 comm listener binds `0.0.0.0`), and `close()` cannot wake that listener's `accept()`. | +| b11449–b11450 | patches + upstream verification | **`0015` breaks at exactly #26610 (`a46709b68`, = b11450)**; the full stack applies to b11449. Three of its hunks lost their context: the `start()` declaration (new dispatcher members `busy_spin_acquire/release`, `graph_compute` around it), the `start()` body (offset only -- the body itself is unchanged upstream, so the same `GGML_ABORT` → `return false` edits apply), and the two `get_proc_address` names, which now follow the new `ggml_backend_comm_*` entries before `return NULL`. Re-applied (`patch -F3`, each placement checked by reading it); the `+`/`-` lines of the regenerated patch are identical to the old ones and its commit-message header is kept. Checked against the new code paths: the comm and graph-uid state is per client (`rpc_serve_client`'s local `rpc_server`), destroyed before `0015`'s `free_backends()`, so the cleanup order still holds; new client paths call `get_dispatcher()` after registration, which keeps upstream's abort by design. The other eight apply unchanged, the stack through b11457. Drop-checks against pristine b11450: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard, no `0014` hook, **no `stop_server` and a `GGML_ABORT("Failed to connect` still present** (`0015` needed). `tools/server/` unchanged in the range. | +| b11450–b11454 | Four commits, 35 files, ~2576/126 lines. **#29535 adds K2 Horizon** (dense and MoVA) with its own chat-output parser (`common/parsers/k2-horizon.cpp`, registered in `common/chat.cpp` and `common/jinja/caps.cpp`) -- tool calls and reasoning of that template work through the existing chat path without a change. **#30054 adds embeddinggemma2** (a new `gemma-embedding2` graph; the conversion handles its vision and audio towers). #30042 removes the gather path of the glm5-next sparse attention; #30041 fixes an out-of-bounds read in the Adreno xmem GEMM of the OpenCL backend (the `opencl-*` natives jars pick it up). **No project source change.** | +| b11450–b11454 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11454 with `git apply`. Patch-target files in the range: `src/llama-model.{cpp,h}` (`0012`; the two new architectures are registered away from `load_tensors`' split block and the helper declarations). Drop-checks against pristine b11454 unchanged: `0001` override present and `common_params_parse_main` absent, `0002` assignment unconditional, no `0003` getter, no `0006`-`0008` counterpart, no `0012` guard (`split_sum` still unguarded), no `0014` hook, no `0015` stop function. `tools/server/` unchanged in the range. | +| b11454–b11457 | Three commits, 9 files, ~122/27 lines. **#30045 adds the PLaMo-3 tokenizer pre-segmentation**, a new `LLAMA_VOCAB_TYPE_PLAMO3 = 8` (`include/llama.h`, appended to the enum, so no existing value moves). `ModelMeta.getVocabType()` returns the number as is (`8` for such a model); the project has no Java enum of vocab types to extend. #28782 uses a per-thread CUDA stream for the buffer-init padding memset; #29955 adds BF16 to CUDA's `XIELU`. **No project source change.** This is the target of the bump: b11457 was the newest llama.cpp release when it was done (b11458 had no release assets). | +| b11454–b11457 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11457 with `git apply`. No patch-target file in the range; nothing under `common/` or `tools/server/` changed. Drop-checks against pristine b11457 -- the final state of the bump: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override and `common_params_parse_main` is absent (`0001` needed); `load_progress_callback` still assigned unconditionally (`0002`); no slot-similarity getter (`0003`); no `llama_server_attach`, `LLAMA_SERVER_WORKER_CMD` or embedded-shutdown counterpart (`0006`-`0008`); no `split_sum == 0` guard (`0012`); no callback hook in `common/log.h` (`0014`); no `stop_server` and a `GGML_ABORT("Failed to connect` still present (`0015`). **Every carried patch is still needed.** Across the whole bump (b11320 → b11457) four patches broke, each at exactly one upstream commit: `0007` at #29818 (b11361) and #29895 (b11401), `0014` at #29895 (b11401), `0008` at #29987 (`4d60b4d08`, before b11429), `0015` at #26610 (b11450); every refresh moved context only. | diff --git a/docs/upgrade/llama-cpp-version-bump.md b/docs/upgrade/llama-cpp-version-bump.md index 9896d44e9..019626c99 100644 --- a/docs/upgrade/llama-cpp-version-bump.md +++ b/docs/upgrade/llama-cpp-version-bump.md @@ -101,6 +101,19 @@ historically caused breaks (`common.h`, `chat.h`, `speculative.h`, `mtmd.h`, `ll `llama.h`, `download.h`), plus the project `CMakeLists.txt` for renamed link targets. Note any new API surface worth wiring through the Java layer (e.g. a new completion param or model-metadata getter). +Three vendor toolchain settings in the CI follow upstream's own CI rather than a version of their +own, so a bump has to carry them along when upstream moves them -- check the chunk's `.github/` diff +for each: + +- **OpenVINO** -- `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` in upstream's + `.github/workflows/release.yml`. Both OpenVINO build jobs (Linux archive, Windows zip) use the pair in + their download URL and in the `KEEP IN SYNC WITH UPSTREAM` comment; ggml-openvino is developed against + it, so lagging behind is what eventually breaks the compile. +- **CUDA's CCCL** -- `GGML_CUDA_CCCL_VERSION` in upstream's CUDA release jobs (`v3.4.3` since #29792); + both CUDA builds pass the same value, see CLAUDE.md, "Upgrading CUDA Version". +- **ROCm** -- the TheRock wheel version of upstream's `ubuntu-rocm` / `windows-rocm` release jobs (see + CLAUDE.md, "Additional GPU-backend natives"). + --- ## Applying a bump diff --git a/llama/CMakeLists.txt b/llama/CMakeLists.txt index 6a8f3a75a..410d1a8e4 100644 --- a/llama/CMakeLists.txt +++ b/llama/CMakeLists.txt @@ -182,7 +182,7 @@ set(LLAMA_BUILD_APP OFF CACHE BOOL "" FORCE) FetchContent_Declare( llama.cpp GIT_REPOSITORY https://github.com/ggerganov/llama.cpp.git - GIT_TAG b11320 + GIT_TAG b11457 PATCH_COMMAND ${CMAKE_COMMAND} -DPATCH_DIR=${CMAKE_CURRENT_SOURCE_DIR}/patches -DLLAMA_SRC= @@ -376,12 +376,21 @@ add_library(jllama SHARED # Android too and stays outside the server-models Android guard below. jllama # wires its own routes and never calls g_stream_sessions.start_gc() (only the # standalone server.cpp main does), so the GC thread stays dormant here. +# +# server-decision.cpp (added upstream in b11361, #29818) implements the typed +# decision models behind POST /v1/systemone (server_decision_context, the +# SERVER_TASK_TYPE_DECISION prompts). server-context.cpp holds one as a member +# and calls into it on every load, so it is mandatory for the same reason as +# server-stream.cpp -- a missing unit is an undefined reference, latent on Linux +# but a hard link error on macOS/ld64 and Windows/MSVC. Platform-neutral, so it +# stays outside the Android guard too. target_sources(jllama PRIVATE ${llama.cpp_SOURCE_DIR}/tools/server/server-context.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-queue.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-task.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-schema.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-stream.cpp + ${llama.cpp_SOURCE_DIR}/tools/server/server-decision.cpp ) if(NOT ANDROID_ABI AND NOT OS_NAME MATCHES "Android") target_sources(jllama PRIVATE @@ -597,6 +606,7 @@ if(BUILD_TESTING) src/test/cpp/test_tts_wav.cpp src/test/cpp/test_tts_params.cpp src/test/cpp/test_model_split.cpp + src/test/cpp/test_kolibri1.cpp src/test/cpp/test_model_flags.cpp src/test/cpp/test_wire_contracts.cpp src/test/cpp/test_rpc.cpp @@ -608,6 +618,7 @@ if(BUILD_TESTING) ${llama.cpp_SOURCE_DIR}/tools/server/server-schema.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-models.cpp ${llama.cpp_SOURCE_DIR}/tools/server/server-stream.cpp + ${llama.cpp_SOURCE_DIR}/tools/server/server-decision.cpp ) target_include_directories(jllama_test PRIVATE diff --git a/llama/patches/0007-server-attach-http-frontend.patch b/llama/patches/0007-server-attach-http-frontend.patch index ee77e402a..a700ba2f3 100644 --- a/llama/patches/0007-server-attach-http-frontend.patch +++ b/llama/patches/0007-server-attach-http-frontend.patch @@ -1,16 +1,16 @@ diff --git a/tools/server/server.cpp b/tools/server/server.cpp -index 6a1dc82db..ac4c45f42 100644 +index 4c6052d78..cb03f239a 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp -@@ -89,6 +89,7 @@ int llama_server(int argc, char ** argv); - +@@ -90,6 +90,7 @@ int llama_server(int argc, char ** argv); // to be used via CLI (argc / argv are used by router mode only) int llama_server(common_params & params, int argc, char ** argv); + int llama_server(common_params & params, int argc, char ** argv, server_child & child); +int llama_server_attach(int argc, char ** argv, server_context & ctx_server); void llama_server_terminate(); void llama_server_terminate() { if (shutdown_handler) { -@@ -133,6 +134,57 @@ static server_http_context::handler_t ex_wrapper(server_http_context::handler_t +@@ -138,6 +139,58 @@ static server_http_context::handler_t ex_wrapper(server_http_context::handler_t }; } @@ -48,6 +48,7 @@ index 6a1dc82db..ac4c45f42 100644 + ctx_http.post("/reranking", ex_wrapper(routes.post_rerank)); + ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank)); + ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank)); ++ ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone)); + ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize)); + ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize)); + ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template)); @@ -66,9 +67,9 @@ index 6a1dc82db..ac4c45f42 100644 +} + int llama_server(int argc, char ** argv) { - std::setlocale(LC_NUMERIC, "C"); + server_child child; -@@ -305,47 +357,7 @@ int llama_server(common_params & params, int argc, char ** argv) { +@@ -317,48 +370,7 @@ int llama_server(common_params & params, int argc, char ** argv, server_child & ctx_http.del ("/models", ex_wrapper(models_routes->del_router_models)); } @@ -98,6 +99,7 @@ index 6a1dc82db..ac4c45f42 100644 - ctx_http.post("/reranking", ex_wrapper(routes.post_rerank)); - ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank)); - ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank)); +- ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone)); - ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize)); - ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize)); - ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template)); @@ -117,7 +119,7 @@ index 6a1dc82db..ac4c45f42 100644 // resumable streaming: a child binds the local session factories, the router binds // proxies that resolve the owning child, see server-stream.h -@@ -615,3 +627,91 @@ int llama_server(common_params & params, int argc, char ** argv) { +@@ -628,3 +640,91 @@ int llama_server(common_params & params, int argc, char ** argv, server_child & return 0; } diff --git a/llama/patches/0008-server-models-worker-cmd-override.patch b/llama/patches/0008-server-models-worker-cmd-override.patch index b8f2a3258..b8a1b948a 100644 --- a/llama/patches/0008-server-models-worker-cmd-override.patch +++ b/llama/patches/0008-server-models-worker-cmd-override.patch @@ -1,6 +1,8 @@ +diff --git a/tools/server/server-models.cpp b/tools/server/server-models.cpp +index 35d21c876..e8ecb539f 100644 --- a/tools/server/server-models.cpp +++ b/tools/server/server-models.cpp -@@ -215,6 +215,29 @@ +@@ -535,6 +535,29 @@ void server_model_meta::update_args(common_preset_context & ctx_preset, std::str if (app_cmd != nullptr && app_cmd[0] != '\0' && !bin_path.empty()) { args.insert(args.begin() + 1, app_cmd); } @@ -29,4 +31,4 @@ + } } - void server_model_meta::update_caps() { + void server_model_meta::update_caps(const common_params & base) { diff --git a/llama/patches/0014-common-log-callback-sink.patch b/llama/patches/0014-common-log-callback-sink.patch index f3d77140e..8d1d4aa7a 100644 --- a/llama/patches/0014-common-log-callback-sink.patch +++ b/llama/patches/0014-common-log-callback-sink.patch @@ -17,10 +17,10 @@ Upstream-submittable ("common : add a callback sink to common_log for embedding hosts"); not yet filed. Guarded by src/test/cpp/test_common_log_callback.cpp. diff --git a/common/log.cpp b/common/log.cpp -index 0a9a4eb..8c8ed23 100644 +index 86c98e8e6..79c8d8a2a 100644 --- a/common/log.cpp +++ b/common/log.cpp -@@ -176,6 +176,9 @@ struct common_log { +@@ -168,6 +168,9 @@ struct common_log { running = false; t_start = t_us(); @@ -30,17 +30,17 @@ index 0a9a4eb..8c8ed23 100644 queue.resize(capacity, common_log_entry(256)); head = 0; tail = 0; -@@ -198,6 +201,9 @@ private: +@@ -190,6 +193,9 @@ private: FILE * file; + common_log_callback callback; + void * callback_user_data; + + bool colors; bool prefix; bool timestamps; - bool running; -@@ -212,7 +218,12 @@ private: +@@ -205,7 +211,12 @@ private: bool print_entry(const common_log_entry & e) const { if (e.is_end) return true; @@ -54,7 +54,7 @@ index 0a9a4eb..8c8ed23 100644 if (file) { e.print(file); } -@@ -407,6 +418,17 @@ public: +@@ -400,6 +411,17 @@ public: resume(); } @@ -69,10 +69,10 @@ index 0a9a4eb..8c8ed23 100644 + resume(); + } + - void set_colors(bool colors) { - pause(); - -@@ -498,6 +520,10 @@ void common_log_set_file(struct common_log * log, const char * file) { + bool get_colors() const { + return colors; + } +@@ -497,6 +519,10 @@ void common_log_set_file(struct common_log * log, const char * file) { log->set_file(file); } @@ -84,10 +84,10 @@ index 0a9a4eb..8c8ed23 100644 if (colors == LOG_COLORS_AUTO) { log->set_colors(tty_can_use_colors()); diff --git a/common/log.h b/common/log.h -index e36b094..65a0261 100644 +index e9e1f1761..5d6ede9aa 100644 --- a/common/log.h +++ b/common/log.h -@@ -97,6 +97,18 @@ void common_log_set_prefix (struct common_log * log, bool prefix); // w +@@ -98,6 +98,18 @@ void common_log_set_prefix (struct common_log * log, bool prefix); // w void common_log_set_timestamps(struct common_log * log, bool timestamps); // whether to output timestamps in the prefix void common_log_flush (struct common_log * log); // flush all pending log messages diff --git a/llama/patches/0015-rpc-embeddable-client-and-stoppable-server.patch b/llama/patches/0015-rpc-embeddable-client-and-stoppable-server.patch index f7ecfde63..365009d3e 100644 --- a/llama/patches/0015-rpc-embeddable-client-and-stoppable-server.patch +++ b/llama/patches/0015-rpc-embeddable-client-and-stoppable-server.patch @@ -46,10 +46,10 @@ src/test/cpp/test_rpc.cpp: it links both new functions and exercises the registr stop paths over loopback. diff --git a/common/arg.cpp b/common/arg.cpp -index 42bbc5601..7fe6b0ce2 100644 +index e15db3315..a9882aaad 100644 --- a/common/arg.cpp +++ b/common/arg.cpp -@@ -1177,6 +1177,11 @@ static void add_rpc_devices(const std::string & servers) { +@@ -1183,6 +1183,11 @@ static void add_rpc_devices(const std::string & servers) { } for (const auto & server : rpc_servers) { auto reg = ggml_backend_rpc_add_server_fn(server.c_str()); @@ -62,7 +62,7 @@ index 42bbc5601..7fe6b0ce2 100644 } } diff --git a/ggml/include/ggml-rpc.h b/ggml/include/ggml-rpc.h -index 1f8cb7906..a8d50ee01 100644 +index 482bd3666..f7fc68956 100644 --- a/ggml/include/ggml-rpc.h +++ b/ggml/include/ggml-rpc.h @@ -26,6 +26,11 @@ GGML_BACKEND_API void ggml_backend_rpc_get_device_memory(const char * endpoint, @@ -78,10 +78,10 @@ index 1f8cb7906..a8d50ee01 100644 GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void); GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint); diff --git a/ggml/src/ggml-rpc/ggml-rpc.cpp b/ggml/src/ggml-rpc/ggml-rpc.cpp -index 353b79b07..abfe8b37d 100644 +index e9af71228..5a6c30065 100644 --- a/ggml/src/ggml-rpc/ggml-rpc.cpp +++ b/ggml/src/ggml-rpc/ggml-rpc.cpp -@@ -351,7 +351,12 @@ static bool negotiate_hello(const std::shared_ptr & sock) { +@@ -402,7 +402,12 @@ static bool negotiate_hello(const std::shared_ptr & sock) { sock->get_caps(request.conn_caps); bool status = send_rpc_cmd(sock, RPC_CMD_HELLO, &request, sizeof(request), &response, sizeof(response)); @@ -95,17 +95,17 @@ index 353b79b07..abfe8b37d 100644 if (response.major != RPC_PROTO_MAJOR_VERSION || response.minor > RPC_PROTO_MINOR_VERSION) { GGML_LOG_ERROR("RPC server version mismatch: %d.%d.%d\n", -@@ -419,7 +424,7 @@ public: - void event_record(ggml_backend_event_t event); - void synchronize(); +@@ -483,7 +488,7 @@ public: + void busy_spin_release(); + void graph_compute(uint32_t device, const ggml_cgraph * cgraph); - void start(const std::string & endpoint); + bool start(const std::string & endpoint); void work(); ~rpc_dispatcher(); -@@ -532,26 +537,32 @@ void rpc_dispatcher::synchronize() { - msg->completion.get_future().wait(); +@@ -608,26 +613,32 @@ void rpc_dispatcher::busy_spin_release() { + GGML_ASSERT(previous > 0); } -void rpc_dispatcher::start(const std::string & endpoint) { @@ -142,7 +142,7 @@ index 353b79b07..abfe8b37d 100644 } void rpc_dispatcher::work() { -@@ -582,7 +593,8 @@ rpc_dispatcher::~rpc_dispatcher() { +@@ -668,7 +679,8 @@ rpc_dispatcher::~rpc_dispatcher() { } } @@ -152,7 +152,7 @@ index 353b79b07..abfe8b37d 100644 static std::mutex mutex; std::lock_guard lock(mutex); static std::unordered_map> dispatchers; -@@ -595,11 +607,23 @@ static std::shared_ptr get_dispatcher(const std::string & endpoi +@@ -681,11 +693,23 @@ static std::shared_ptr get_dispatcher(const std::string & endpoi } auto dispatcher = std::make_shared(); @@ -177,7 +177,7 @@ index 353b79b07..abfe8b37d 100644 static void ggml_backend_rpc_buffer_free_buffer(ggml_backend_buffer_t buffer) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; auto request = std::make_shared(); -@@ -1125,6 +1149,10 @@ ggml_backend_t ggml_backend_rpc_init(const char * endpoint, uint32_t device) { +@@ -1277,6 +1301,10 @@ ggml_backend_t ggml_backend_rpc_init(const char * endpoint, uint32_t device) { /* .name = */ dev_name, }; auto reg = ggml_backend_rpc_add_server(endpoint); @@ -188,7 +188,7 @@ index 353b79b07..abfe8b37d 100644 ggml_backend_t backend = new ggml_backend { /* .guid = */ ggml_backend_rpc_guid(), /* .iface = */ ggml_backend_rpc_interface, -@@ -1139,7 +1167,15 @@ bool ggml_backend_is_rpc(ggml_backend_t backend) { +@@ -1291,7 +1319,15 @@ bool ggml_backend_is_rpc(ggml_backend_t backend) { } void ggml_backend_rpc_get_device_memory(const char * endpoint, uint32_t device, size_t * free, size_t * total) { @@ -205,7 +205,7 @@ index 353b79b07..abfe8b37d 100644 auto request = std::make_shared(); request->device = device; rpc_msg_get_device_memory_rsp response; -@@ -2085,6 +2121,41 @@ static void rpc_serve_client(const std::vector & backends, const +@@ -2650,6 +2686,41 @@ static void rpc_serve_client(const std::vector & backends, const } } @@ -247,7 +247,7 @@ index 353b79b07..abfe8b37d 100644 void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir, size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices) { if (n_devices == 0 || devices == nullptr) { -@@ -2092,6 +2163,12 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir +@@ -2657,6 +2728,12 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir return; } std::vector backends; @@ -260,7 +260,7 @@ index 353b79b07..abfe8b37d 100644 printf("Starting RPC server v%d.%d.%d\n", RPC_PROTO_MAJOR_VERSION, RPC_PROTO_MINOR_VERSION, -@@ -2108,6 +2185,7 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir +@@ -2673,6 +2750,7 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir auto backend = ggml_backend_dev_init(dev, nullptr); if (!backend) { fprintf(stderr, "Failed to create backend for device %s\n", dev->iface.get_name(dev)); @@ -268,7 +268,7 @@ index 353b79b07..abfe8b37d 100644 return; } backends.push_back(backend); -@@ -2123,6 +2201,7 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir +@@ -2688,6 +2766,7 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir std::string host; int port; if (!parse_endpoint(endpoint, host, port)) { @@ -276,7 +276,7 @@ index 353b79b07..abfe8b37d 100644 return; } -@@ -2133,29 +2212,57 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir +@@ -2698,29 +2777,57 @@ void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir #endif // GGML_RPC_RDMA if (!rpc_transport_init()) { fprintf(stderr, "Failed to initialize RPC transport\n"); @@ -340,9 +340,9 @@ index 353b79b07..abfe8b37d 100644 } static const char * ggml_backend_rpc_device_get_name(ggml_backend_dev_t dev) { -@@ -2299,6 +2406,12 @@ static void * ggml_backend_rpc_get_proc_address(ggml_backend_reg_t reg, const ch - if (std::strcmp(name, "ggml_backend_rpc_start_server") == 0) { - return (void *)ggml_backend_rpc_start_server; +@@ -3004,6 +3111,12 @@ static void * ggml_backend_rpc_get_proc_address(ggml_backend_reg_t reg, const ch + if (std::strcmp(name, "ggml_backend_comm_allreduce_tensor") == 0) { + return (void *)ggml_backend_rpc_comm_allreduce_tensor; } + if (std::strcmp(name, "ggml_backend_rpc_stop_server") == 0) { + return (void *)ggml_backend_rpc_stop_server; @@ -353,7 +353,7 @@ index 353b79b07..abfe8b37d 100644 return NULL; GGML_UNUSED(reg); -@@ -2321,8 +2434,13 @@ ggml_backend_reg_t ggml_backend_rpc_reg(void) { +@@ -3026,8 +3139,13 @@ ggml_backend_reg_t ggml_backend_rpc_reg(void) { return &ggml_backend_rpc_reg; } @@ -368,7 +368,7 @@ index 353b79b07..abfe8b37d 100644 rpc_msg_device_count_rsp response; dispatcher->send(RPC_CMD_DEVICE_COUNT, nullptr, 0, &response, sizeof(response)); return response.device_count; -@@ -2341,6 +2459,11 @@ ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint) { +@@ -3046,6 +3164,11 @@ ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint) { static uint32_t dev_id = 0; std::lock_guard lock(mutex); if (reg_map.find(endpoint) != reg_map.end()) { diff --git a/llama/patches/0016-model-kolibri1.patch b/llama/patches/0016-model-kolibri1.patch new file mode 100644 index 000000000..fa1362062 --- /dev/null +++ b/llama/patches/0016-model-kolibri1.patch @@ -0,0 +1,385 @@ +model: add Kolibri-1 (kolibri1) -- carried until upstream supports it + +Aleph Alpha's Kolibri-1 is a 78B-parameter German/English reasoning MoE (3.46B active per +token, Apache-2.0). Upstream llama.cpp has no support yet; the request is +https://github.com/ggml-org/llama.cpp/issues/29922 (open, no PR at the time of writing). +DROP THIS PATCH as soon as upstream registers the architecture -- check on every bump with +`git grep -n kolibri src/llama-arch.cpp` in the new llama.cpp tag. + +Architecture (from Aleph Alpha's own vLLM implementation, the reference used here): + - GQA attention with per-head q/k RMSNorm. config.layer_types interleaves sliding-window + layers, which use (NEOX) RoPE, with full-attention layers that have no positional encoding. + - Sandwich norms: input norm -> attention -> post-attention norm -> residual, then + pre-FFN norm -> routed MoE + one ungated shared expert -> post-FFN norm -> residual. + - Router: top-k selected on (logits + expert_bias), experts weighted by the unbiased + sigmoid(logits), renormalized only if norm_topk_prob (off in Kolibri-1). This is NOT + DeepSeek-V3's sigmoid router (selection on sigmoid(logits) + bias), which picks other + experts as soon as the bias is non-zero -- so the graph builds the selection itself and + hands it to build_moe_ffn (selected_experts_in); build_moe_ffn itself is unchanged. + +What the patch touches: the arch name "kolibri1" (llama-arch.{h,cpp}), the model mapping and +NEOX rope type (llama-model.cpp), the model struct (models/models.h), the new +src/models/kolibri1.cpp, and the "kolibri1" pre-tokenizer name mapped to Qwen2's +(llama-vocab.cpp). No converter or gguf-py change: this project only loads GGUFs. + +Both published GGUF dialects load. The two community converters disagree on metadata, and +each community runtime rejects the other's files: + - expert_gating_func: 2 (sigmoid) from the AFMoE-based converter, 5 ("sigmoid_logit_add", + a value only the Qwen3-MoE-based converter's patch defines) from the other one, or + absent. All three mean the same router, which is a property of the architecture. + - tokenizer.ggml.pre: "qwen2" or "kolibri1"; the Qwen3-MoE-based runtime registers the + latter, the AFMoE-based one does not (llama.cpp rejects an unknown pre-tokenizer). + - expert_weights_norm, expert_shared_count and output.weight are optional here. + +Known places (all read while writing this patch, 2026-10-06): + - Request: https://github.com/ggml-org/llama.cpp/issues/29922 + - Model: https://huggingface.co/Aleph-Alpha/Kolibri-1 (and Kolibri-1-BF16) + - Reference: https://github.com/Aleph-Alpha/aleph-alpha-inference, aleph_alpha_inference/ + kolibri1.py at 049a6a7 (Apache-2.0, Copyright 2026 Aleph Alpha GmbH) -- the + specification this patch follows; its tests/test_kolibri1.py pins the router. + - AFMoE-based port by Eliasfpv28, base llama.cpp edd6e2bbd (b11378): + https://huggingface.co/Eliasfpv28/Kolibri-1-Q3_K_S-GGUF, runtime-source/ + kolibri1-runtime.patch (new additions Apache-2.0, llama.cpp-derived parts MIT). + Source of the graph-side router (selection built outside build_moe_ffn). + Its runtime is also what https://huggingface.co/InsidiousFiddler/Kolibri-1-GGUF + points to. + - Qwen3-MoE-based port, base llama.cpp 836d571 (b11381), as carried in + https://github.com/mjwsolo/localcode/pull/101 (patches/0006-kolibri1.patch, + "community patch"). https://huggingface.co/Hob-forge/Kolibri-1-GGUF publishes + GGUFs and a kolibri1-llama.cpp.patch (MIT) for the same base commit whose model + card describes the same design (a new gating mode); that file itself could not be + fetched from here, so it is not verified to be byte-identical. Source of the + loader (optional keys, tied-output fallback, swa rope base) and of the graph layout. + - Independent effort under the name "kolibri" (not "kolibri1"), whose GGUFs this patch + does not load: https://github.com/CWBudde/kolibri-llama-cpp, + https://github.com/CWBudde/llama.cpp/pull/7 + +Verification in java-llama.cpp: src/test/cpp/test_kolibri1.cpp builds tiny random Kolibri-1 +GGUFs in both dialects, runs them through the real library and compares every logit with an +independent float reference written from the vLLM implementation above. + +diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp +index eea2ef589..638d73c4b 100644 +--- a/src/llama-arch.cpp ++++ b/src/llama-arch.cpp +@@ -117,6 +117,7 @@ static const std::map LLM_ARCH_NAMES = { + { LLM_ARCH_DOTS3NOTE, "dots3note" }, + { LLM_ARCH_ARCEE, "arcee" }, + { LLM_ARCH_AFMOE, "afmoe" }, ++ { LLM_ARCH_KOLIBRI1, "kolibri1" }, + { LLM_ARCH_LAGUNA, "laguna" }, + { LLM_ARCH_ERNIE4_5, "ernie4_5" }, + { LLM_ARCH_ERNIE4_5_MOE, "ernie4_5-moe" }, +diff --git a/src/llama-arch.h b/src/llama-arch.h +index c8abee234..6a235d2b1 100644 +--- a/src/llama-arch.h ++++ b/src/llama-arch.h +@@ -122,6 +122,7 @@ enum llm_arch { + LLM_ARCH_DOTS3NOTE, + LLM_ARCH_ARCEE, + LLM_ARCH_AFMOE, ++ LLM_ARCH_KOLIBRI1, + LLM_ARCH_LAGUNA, + LLM_ARCH_ERNIE4_5, + LLM_ARCH_ERNIE4_5_MOE, +diff --git a/src/llama-model.cpp b/src/llama-model.cpp +index 608285d5c..03150375d 100644 +--- a/src/llama-model.cpp ++++ b/src/llama-model.cpp +@@ -276,6 +276,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params + return new llama_model_arcee(params); + case LLM_ARCH_AFMOE: + return new llama_model_afmoe(params); ++ case LLM_ARCH_KOLIBRI1: ++ return new llama_model_kolibri1(params); + case LLM_ARCH_LAGUNA: + return new llama_model_laguna(params); + case LLM_ARCH_ERNIE4_5: +@@ -3254,6 +3256,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { + case LLM_ARCH_COGVLM: + case LLM_ARCH_PANGU_EMBED: + case LLM_ARCH_AFMOE: ++ case LLM_ARCH_KOLIBRI1: + case LLM_ARCH_LAGUNA: + case LLM_ARCH_QWEN3NEXT: + case LLM_ARCH_MIMO2: +diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp +index de20c757f..5b54a31bf 100644 +--- a/src/llama-vocab.cpp ++++ b/src/llama-vocab.cpp +@@ -2376,7 +2376,8 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) { + tokenizer_pre == "qwen2" || + tokenizer_pre == "deepseek-r1-qwen" || + tokenizer_pre == "kormo" || +- tokenizer_pre == "f2llmv2") { ++ tokenizer_pre == "f2llmv2" || ++ tokenizer_pre == "kolibri1") { + pre_type = LLAMA_VOCAB_PRE_TYPE_QWEN2; + clean_spaces = false; + } else if ( +diff --git a/src/models/kolibri1.cpp b/src/models/kolibri1.cpp +new file mode 100644 +index 000000000..fb52a7a4b +--- /dev/null ++++ b/src/models/kolibri1.cpp +@@ -0,0 +1,237 @@ ++#include "models.h" ++ ++// Aleph Alpha Kolibri 1 (https://huggingface.co/Aleph-Alpha/Kolibri-1), a German/English reasoning MoE. ++// Reference: aleph_alpha_inference/kolibri1.py (https://github.com/Aleph-Alpha/aleph-alpha-inference, 049a6a7). ++// ++// - attention: GQA with per-head q/k RMSNorm; layer_types interleave sliding-window layers, which use ++// (NEOX) RoPE, with full-attention layers, which use no positional encoding at all (RNoPE) ++// - sandwich norms: input norm -> attention -> post-attention norm -> residual, ++// pre-FFN norm -> MoE + shared expert -> post-FFN norm -> residual ++// - every layer is MoE plus one ungated shared expert, added to the routed output ++// - router: top-k is selected on (logits + expert_bias) and the experts are weighted by the *unbiased* ++// sigmoid(logits), renormalized only if norm_topk_prob is set (Kolibri 1 ships it off). This is not ++// DeepSeek-V3's sigmoid router, which selects on sigmoid(logits) + bias -- that one picks different ++// experts as soon as the bias is non-zero, so the selection is built here and handed to build_moe_ffn. ++// ++// Two community converters produced the published GGUFs, and they disagree on metadata; both load: ++// - expert_gating_func: 2 (sigmoid) from the AFMoE-based converter, 5 ("sigmoid_logit_add", a value that ++// exists only in the other converter's patch) from the Qwen3-MoE-based one, or absent. The routing is a ++// property of the architecture, so the value is only checked, never used to pick another router. ++// - tokenizer.ggml.pre: "qwen2" or "kolibri1" (llama-vocab.cpp maps both to the Qwen2 pre-tokenizer). ++// - expert_weights_norm and output.weight are optional (defaults: no renormalization, tied embeddings). ++ ++void llama_model_kolibri1::load_arch_hparams(llama_model_loader & ml) { ++ ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps); ++ ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all); ++ ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp); ++ ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa); ++ ++ hparams.n_expert_shared = 1; ++ ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared, false); ++ ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM, hparams.expert_weights_norm, false); ++ ++ uint32_t gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID; ++ ml.get_key(LLM_KV_EXPERT_GATING_FUNC, gating_func, false); ++ if (gating_func != LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID && gating_func != 5) { ++ throw std::runtime_error(format("kolibri1: unexpected expert_gating_func %u (expected 2 or 5)", gating_func)); ++ } ++ // what print_info reports; the selection on logits + bias is built in the graph ++ hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID; ++ ++ if (hparams.n_swa == 0) { ++ throw std::runtime_error("kolibri1: sliding_window must be > 0"); ++ } ++ if (hparams.n_expert_shared != 1 || hparams.n_ff_shexp == 0) { ++ throw std::runtime_error(format("kolibri1: expected one shared expert, got %u of length %u", ++ hparams.n_expert_shared, hparams.n_ff_shexp)); ++ } ++ ++ // the converters write the exact per-layer pattern from config.layer_types; 4 sliding + 1 full otherwise ++ hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; ++ load_swa_pattern(ml, 5); ++ ++ // only the sliding-window layers use RoPE, with the model's single rope_theta ++ hparams.rope_freq_base_train_swa = hparams.rope_freq_base_train; ++ hparams.rope_freq_scale_train_swa = hparams.rope_freq_scale_train; ++ ml.get_key(LLM_KV_ROPE_FREQ_BASE_SWA, hparams.rope_freq_base_train_swa, false); ++ ++ type = LLM_TYPE_UNKNOWN; ++} ++ ++void llama_model_kolibri1::load_arch_tensors(llama_model_loader &) { ++ LLAMA_LOAD_LOCALS; ++ ++ if (n_expert == 0 || n_expert_used == 0) { ++ throw std::runtime_error("kolibri1: n_expert and n_expert_used must be > 0"); ++ } ++ ++ tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); ++ ++ output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0); ++ output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED); ++ if (output == NULL) { ++ output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED); ++ } ++ ++ const int64_t n_ff_exp = hparams.n_ff_exp(); ++ const int64_t n_ff_shexp = hparams.n_ff_shexp; ++ ++ for (int i = 0; i < n_layer; ++i) { ++ auto & layer = layers[i]; ++ ++ layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0); ++ layer.attn_post_norm = create_tensor(tn(LLM_TENSOR_ATTN_POST_NORM, "weight", i), {n_embd}, 0); ++ ++ create_tensor_qkv(layer, i, n_embd, n_embd_head_k * n_head, n_embd_k_gqa, n_embd_v_gqa, 0); ++ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, 0); ++ ++ layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {n_embd_head_k}, 0); ++ layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {n_embd_head_k}, 0); ++ ++ layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0); ++ layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0); ++ ++ layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", i), {n_embd, n_expert}, 0); ++ layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", i), {n_expert}, 0); ++ ++ layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), { n_embd, n_ff_exp, n_expert}, 0); ++ layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd, n_expert}, 0); ++ layer.ffn_up_exps = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i), { n_embd, n_ff_exp, n_expert}, 0); ++ ++ layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), { n_embd, n_ff_shexp}, 0); ++ layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_shexp, n_embd}, 0); ++ layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", i), { n_embd, n_ff_shexp}, 0); ++ } ++} ++ ++std::unique_ptr llama_model_kolibri1::build_arch_graph(const llm_graph_params & params) const { ++ return std::make_unique(*this, params); ++} ++ ++llama_model_kolibri1::graph::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) { ++ const int64_t n_embd_head = hparams.n_embd_head_v(); ++ GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); ++ ++ ggml_tensor * cur; ++ ggml_tensor * inpL = build_inp_embd(model.tok_embd); ++ ++ ggml_tensor * inp_pos = build_inp_pos(); ++ auto * inp_attn = build_attn_inp_kv_iswa(); ++ ggml_tensor * inp_out_ids = build_inp_out_ids(); ++ ++ const float kq_scale = 1.0f/sqrtf(float(n_embd_head)); ++ ++ for (int il = 0; il < n_layer; ++il) { ++ const auto & layer = model.layers[il]; ++ ++ ggml_tensor * inpSA = inpL; ++ ++ cur = build_norm(inpL, layer.attn_norm, NULL, LLM_NORM_RMS, il); ++ cb(cur, "attn_norm", il); ++ ++ // self-attention ++ { ++ auto [Qcur, Kcur, Vcur] = build_qkv(layer, cur, n_embd_head, n_head, n_head_kv, il); ++ ++ Qcur = build_norm(Qcur, layer.attn_q_norm, NULL, LLM_NORM_RMS, il); ++ Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il); ++ cb(Qcur, "Qcur_normed", il); ++ cb(Kcur, "Kcur_normed", il); ++ ++ // sliding-window layers: RoPE; full-attention layers: no positional encoding ++ if (hparams.is_swa(il)) { ++ const float freq_base_l = model.get_rope_freq_base (cparams, il); ++ const float freq_scale_l = model.get_rope_freq_scale(cparams, il); ++ ++ Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, nullptr, ++ n_rot, rope_type, n_ctx_orig, freq_base_l, freq_scale_l, ++ ext_factor, attn_factor, beta_fast, beta_slow); ++ Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, nullptr, ++ n_rot, rope_type, n_ctx_orig, freq_base_l, freq_scale_l, ++ ext_factor, attn_factor, beta_fast, beta_slow); ++ cb(Qcur, "Qcur_rope", il); ++ cb(Kcur, "Kcur_rope", il); ++ } ++ ++ cur = build_attn(inp_attn, ++ layer.wo, NULL, layer.wo_s, ++ Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il); ++ cb(cur, "attn_out", il); ++ } ++ ++ cur = build_norm(cur, layer.attn_post_norm, NULL, LLM_NORM_RMS, il); ++ cb(cur, "attn_post_norm", il); ++ ++ if (il == n_layer - 1 && inp_out_ids) { ++ cur = ggml_get_rows(ctx0, cur, inp_out_ids); ++ inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids); ++ } ++ ++ ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA); ++ cb(ffn_inp, "ffn_inp", il); ++ ++ // HF "post_attention_layernorm" is the pre-FFN norm here ++ cur = build_norm(ffn_inp, layer.ffn_norm, NULL, LLM_NORM_RMS, il); ++ cb(cur, "ffn_norm", il); ++ ++ // router: select on logits + expert_bias, weight by the unbiased sigmoid(logits). The reference ++ // computes the logits in fp32 (GateLinear out_dtype=float32), and the selection is sensitive to it: ++ // the biases reach ~20, so keep both the accumulation and the activations at F32 on every backend. ++ ggml_tensor * logits = build_lora_mm(layer.ffn_gate_inp, cur); // [n_expert, n_tokens] ++ ggml_prec_set_acc(logits, GGML_PREC_F32); ++ ggml_prec_set_src(logits, GGML_PREC_F32, 1); ++ cb(logits, "ffn_moe_logits", il); ++ ++ ggml_tensor * selection = ggml_add(ctx0, logits, layer.ffn_exp_probs_b); ++ cb(selection, "ffn_moe_logits_biased", il); ++ ++ ggml_tensor * selected = ggml_argsort_top_k(ctx0, selection, n_expert_used); // [n_expert_used, n_tokens] ++ ++ ggml_tensor * moe_out = build_moe_ffn(cur, ++ layer.ffn_gate_inp, ++ layer.ffn_up_exps, ++ layer.ffn_gate_exps, ++ layer.ffn_down_exps, ++ nullptr, // the bias already went into the selection above ++ n_expert, n_expert_used, ++ LLM_FFN_SILU, ++ hparams.expert_weights_norm, ++ 0.0f, // no routed scaling factor ++ LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID, ++ il, ++ logits, ++ nullptr, nullptr, nullptr, nullptr, ++ selected); ++ cb(moe_out, "ffn_moe_out", il); ++ ++ ggml_tensor * ffn_shexp = build_ffn(cur, ++ layer.ffn_up_shexp, NULL, NULL, ++ layer.ffn_gate_shexp, NULL, NULL, ++ layer.ffn_down_shexp, NULL, NULL, ++ NULL, ++ LLM_FFN_SILU, LLM_FFN_PAR, il); ++ cb(ffn_shexp, "ffn_shexp", il); ++ ++ cur = ggml_add(ctx0, moe_out, ffn_shexp); ++ cb(cur, "ffn_out", il); ++ ++ cur = build_norm(cur, layer.ffn_post_norm, NULL, LLM_NORM_RMS, il); ++ cb(cur, "ffn_post_norm", il); ++ ++ cur = ggml_add(ctx0, cur, ffn_inp); ++ cur = build_cvec(cur, il); ++ cb(cur, "l_out", il); ++ ++ inpL = cur; ++ } ++ ++ cur = build_norm(inpL, model.output_norm, NULL, LLM_NORM_RMS, -1); ++ cb(cur, "result_norm", -1); ++ res->t_embd = cur; ++ ++ cur = build_lora_mm(model.output, cur, model.output_s); ++ cb(cur, "result_output", -1); ++ res->t_logits = cur; ++ ++ ggml_build_forward_expand(gf, cur); ++} +diff --git a/src/models/models.h b/src/models/models.h +index 023ed3021..84eff7e97 100644 +--- a/src/models/models.h ++++ b/src/models/models.h +@@ -1972,6 +1972,18 @@ struct llama_model_afmoe : public llama_model_base { + }; + + ++struct llama_model_kolibri1 : public llama_model_base { ++ llama_model_kolibri1(const struct llama_model_params & params) : llama_model_base(params) {} ++ void load_arch_hparams(llama_model_loader & ml) override; ++ void load_arch_tensors(llama_model_loader & ml) override; ++ ++ struct graph : public llm_graph_context { ++ graph(const llama_model & model, const llm_graph_params & params); ++ }; ++ ++ std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; ++}; ++ + struct llama_model_laguna : public llama_model_base { + llama_model_laguna(const struct llama_model_params & params) : llama_model_base(params) {} + void load_arch_hparams(llama_model_loader & ml) override; diff --git a/llama/pom.xml b/llama/pom.xml index ae7542441..ce0f118f3 100644 --- a/llama/pom.xml +++ b/llama/pom.xml @@ -1227,6 +1227,26 @@ SPDX-License-Identifier: MIT + + natives-vulkan-windows-aarch64 + package + + jar + + + vulkan-windows-aarch64 + ${project.basedir}/src/main/natives + + net/ladenthin/llama/Windows/aarch64/vulkan/** + + + false + + net.ladenthin.llama.natives.vulkan_windows_aarch64 + + + + natives-opencl-android-aarch64 package diff --git a/llama/spotbugs-exclude.xml b/llama/spotbugs-exclude.xml index 2f1568cb4..d5e13bb8b 100644 --- a/llama/spotbugs-exclude.xml +++ b/llama/spotbugs-exclude.xml @@ -79,6 +79,7 @@ SPDX-License-Identifier: MIT + diff --git a/llama/src/main/cpp/jllama.cpp b/llama/src/main/cpp/jllama.cpp index ace4ff30f..c67587c84 100644 --- a/llama/src/main/cpp/jllama.cpp +++ b/llama/src/main/cpp/jllama.cpp @@ -1011,12 +1011,18 @@ static void load_model_impl(JNIEnv *env, jobject obj, jobjectArray jparams, jobj params.load_progress_callback_user_data = &progress_ud; } + // Before load_model(), as upstream's llama_server() does -- see jllama_context::routes. It keeps a + // reference to jctx->params, which lives as long as it does. + jctx->routes = std::make_unique(jctx->params, jctx->server); + if (!jctx->server.load_model(params)) { fail_load("could not load model from given file path"); return; } jctx->vocab = llama_model_get_vocab(llama_get_model(jctx->server.get_llama_context())); + // The handlers read the model's metadata from this copy; taken once, as upstream does after a load. + jctx->routes->update_meta(jctx->server); LOG_INF("%s: model loaded\n", __func__); @@ -1120,6 +1126,15 @@ JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_getModelMetaJson(J {"architecture", arch}, {"ftype", m.model_ftype}, }; + // The input/output modalities of upstream's GET /models `architecture` object (b11429, #29987), + // built by the same helper; flattened to the top level because "architecture" above already + // names the GGUF architecture string. A native decision model reports ["decisions"]. + { + const json modalities = server_model_architecture_json(m.has_inp_image, m.has_inp_audio, m.has_inp_video, + m.model_output_modalities); + j["input_modalities"] = modalities.at("input_modalities"); + j["output_modalities"] = modalities.at("output_modalities"); + } // Resolved default chat template (Jinja); empty when the model ships none. const char *chat_tmpl = mdl != nullptr ? llama_model_chat_template(mdl, /*name*/ nullptr) : nullptr; j["chat_template"] = chat_tmpl != nullptr ? std::string(chat_tmpl) : std::string(); @@ -1377,6 +1392,35 @@ JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_handleRerank(JNIEn }); } +/** + * `POST /v1/systemone` (llama.cpp b11361, #29818): answers typed questions about a state with a + * decision model. Forwarded to the upstream route handler rather than re-implemented -- the request + * parsing, the per-model prompt layout, the shared-prefix grouping and the answer formatting all live + * in upstream's `server_decision_context`, which is private to `server_context_impl` and reachable + * only through `server_routes`. The handler waits for the server to leave sleep itself + * (`create_response()`), and every failure it reports as an error body becomes a LlamaException + * carrying upstream's message ("This model is not a decision model", a malformed question, ...). + */ +JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_handleSystemOne(JNIEnv *env, jobject obj, + jstring jrequest) { + return jni_guard_impl(env, c_llama_error, [&]() -> jstring { + REQUIRE_SERVER_CONTEXT(nullptr); + if (!jctx->routes) { + env->ThrowNew(c_llama_error, "systemone is not available in vocab-only mode"); + return nullptr; + } + + const std::function should_stop = [jctx] { return jctx->closing.load(); }; + const server_http_req req{{}, {}, "/v1/systemone", "", parse_jstring(env, jrequest), {}, should_stop}; + const server_http_res_ptr res = jctx->routes->post_systemone(req); + if (res->status != 200) { + env->ThrowNew(c_llama_error, route_error_message(res->data).c_str()); + return nullptr; + } + return utf8_to_jstring(env, res->data); + }); +} + JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_applyTemplate(JNIEnv *env, jobject obj, jstring jparams) { return jni_guard_impl(env, c_llama_error, [&]() -> jstring { REQUIRE_SERVER_CONTEXT(nullptr); diff --git a/llama/src/main/cpp/jllama.h b/llama/src/main/cpp/jllama.h index bc57752f2..6df2ef537 100644 --- a/llama/src/main/cpp/jllama.h +++ b/llama/src/main/cpp/jllama.h @@ -112,6 +112,13 @@ JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_nativeLlamaCppBuil */ JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_handleRerank(JNIEnv *, jobject, jstring, jobjectArray); +/* + * Class: net_ladenthin_llama_LlamaModel + * Method: handleSystemOne + * Signature: (Ljava/lang/String;)Ljava/lang/String; + */ +JNIEXPORT jstring JNICALL Java_net_ladenthin_llama_LlamaModel_handleSystemOne(JNIEnv *, jobject, jstring); + /* * Class: net_ladenthin_llama_LlamaModel * Method: applyTemplate diff --git a/llama/src/main/cpp/jni_helpers.hpp b/llama/src/main/cpp/jni_helpers.hpp index 07b8b8dd9..0f0303122 100644 --- a/llama/src/main/cpp/jni_helpers.hpp +++ b/llama/src/main/cpp/jni_helpers.hpp @@ -52,6 +52,12 @@ struct server_response_reader; // --------------------------------------------------------------------------- struct jllama_context { server_context server; // value member (pimpl inside) + // The upstream HTTP route handlers over `server`, for endpoints this layer forwards to upstream + // instead of re-implementing them (`handleSystemOne` -> `post_systemone`). Constructed before + // `server.load_model()`, as upstream does: its constructor registers a sleeping-state callback, + // and upstream relies on that callback running before the server's own, which frees the model. + // Declared after `server`, so it is destroyed first. Null in vocab-only mode. + std::unique_ptr routes; std::thread worker; bool vocab_only = false; std::atomic worker_ready{false}; diff --git a/llama/src/main/cpp/json_helpers.hpp b/llama/src/main/cpp/json_helpers.hpp index 02fca9cb0..0c7d31a7c 100644 --- a/llama/src/main/cpp/json_helpers.hpp +++ b/llama/src/main/cpp/json_helpers.hpp @@ -34,6 +34,7 @@ // 8. parse_positive_int_config — used by nothing above it // 9. wrap_stream_chunk — used by nothing above it // 10. server_metrics_to_json — used by nothing above it +// 11. route_error_message — used by nothing above it #include #include @@ -299,3 +300,30 @@ out["slots"] = slots_result.slots_data; return out; } + +// --------------------------------------------------------------------------- +// route_error_message +// +// The human-readable message of an error response produced by an upstream +// `server_routes` handler. `server_res_generator::error()` writes the body as +// `{"error": {"code": ..., "message": ..., "type": ...}}`; this returns +// `error.message`, or the body unchanged when it does not have that shape. +// Never throws: it runs on a path that is already reporting an error. +// +// Used by handleSystemOne in jllama.cpp, which calls the upstream +// `post_systemone` handler directly instead of re-implementing it. +// --------------------------------------------------------------------------- +[[nodiscard]] inline std::string route_error_message(const std::string &body) { + try { + const json parsed = json::parse(body); + if (parsed.is_object() && parsed.contains("error")) { + const json &error = parsed.at("error"); + if (error.is_object() && error.contains("message") && error.at("message").is_string()) { + return error.at("message").get(); + } + } + } catch (const std::exception &) { + // not JSON: fall through and report the raw body + } + return body; +} diff --git a/llama/src/main/java/net/ladenthin/llama/LlamaModel.java b/llama/src/main/java/net/ladenthin/llama/LlamaModel.java index 5c20a9bca..82756724a 100644 --- a/llama/src/main/java/net/ladenthin/llama/LlamaModel.java +++ b/llama/src/main/java/net/ladenthin/llama/LlamaModel.java @@ -593,6 +593,31 @@ public LlamaOutput rerank(String query, String... documents) { */ public native String handleRerank(String query, String... documents); + /** + * Answer typed questions about a state with a decision model, in one forward pass per question + * and without generating a token. This is llama.cpp's TypeSafe-compatible + * {@code POST /v1/systemone} API (upstream b11361), served by the same upstream handler the HTTP + * server uses, so the request and response are exactly that endpoint's. + * + *

The request carries a {@code "state"} (a string, or any JSON value given to the model as + * JSON text), optional {@code "images"} (data URLs; needs a model that takes images and its + * {@code --mmproj}) and {@code "questions"}, an object mapping an id to a question with a + * {@code "type"} of {@code "choice"}, {@code "score"} or {@code "noul"}, its + * {@code "instructions"} and, depending on the type, its {@code "criteria"}. The response maps + * each id under {@code "answers"} to the answer of that type (the chosen option and the + * probabilities, the expected level, or the probability of {@code true}) and reports + * {@code "usage"}. The full description is upstream's {@code tools/server/README.md}.

+ * + *

Needs a decision model (laya, julia-1, lev, openjev, kev and the ones upstream adds later); + * any other model fails with a {@link LlamaException} saying it is not a decision model.

+ * + * @param requestJson the request body as JSON + * @return the response body as JSON + * @throws LlamaException when the model is not a decision model, the request is malformed, or it + * carries images the loaded model cannot take + */ + public native String handleSystemOne(String requestJson); + /** * Applies the chat template to the given inference parameters and returns the formatted string. * @@ -981,7 +1006,8 @@ public String eraseSlot(int slotId) { * llama.cpp stamps every state file with {@code LLAMA_STATE_SEQ_VERSION} and rejects one written * under a different value, so a file saved by a jar built against a different * {@link net.ladenthin.llama.value.LlamaCppVersion#LLAMA_CPP_VERSION} may not load — b10642 bumped - * that constant 2 → 3, invalidating every file written by an earlier release. Treat + * that constant 2 → 3 and b11411 3 → 4, each time invalidating every file + * written by an earlier release. Treat * these files as a cache to regenerate on upgrade, never as durable storage. A rejected file * surfaces as a {@link net.ladenthin.llama.exception.LlamaException} whose message is upstream's * wrapped form, {@code "Unable to restore slot: No available space in KV cache or invalid slot diff --git a/llama/src/main/java/net/ladenthin/llama/args/DraftSampling.java b/llama/src/main/java/net/ladenthin/llama/args/DraftSampling.java new file mode 100644 index 000000000..9c1a37fd3 --- /dev/null +++ b/llama/src/main/java/net/ladenthin/llama/args/DraftSampling.java @@ -0,0 +1,55 @@ +// SPDX-FileCopyrightText: 2026 Bernard Ladenthin +// +// SPDX-License-Identifier: MIT + +package net.ladenthin.llama.args; + +/** + * How speculative decoding samples the draft, for a draft model ({@code --spec-draft-model}) or a + * model's own MTP heads. + * + *

The string constants are the exact values accepted by llama.cpp's {@code --spec-draft-sampling} + * CLI argument (added in b11368, #27694), which sets {@code common_params_speculative::draft.probabilistic}. + * The output distribution is the target model's in both modes; what changes is how many drafted tokens + * it accepts at a non-zero temperature.

+ * + * @see net.ladenthin.llama.parameters.ModelParameters#setDraftSampling(DraftSampling) + */ +public enum DraftSampling implements CliArg { + + /** + * Draft the argmax token at every position and accept it while the target agrees. + * + *

CLI string: {@code "greedy"}. This is upstream's default, so passing it is equivalent to + * omitting the flag. + */ + GREEDY("greedy"), + + /** + * Sample the draft and let the target verify it by rejection sampling, which accepts more drafted + * tokens when the request samples at a temperature above zero. A grammar-constrained request is + * supported as well. + * + *

CLI string: {@code "probabilistic"}. + */ + PROBABILISTIC("probabilistic"); + + /** + * The CLI string passed to {@code --spec-draft-sampling} in llama.cpp's {@code common/arg.cpp}. + */ + private final String argValue; + + DraftSampling(String value) { + this.argValue = value; + } + + /** + * Returns the CLI string accepted by llama.cpp's {@code --spec-draft-sampling} argument. + * + * @return the mode string ({@code "greedy"} or {@code "probabilistic"}) + */ + @Override + public String getArgValue() { + return argValue; + } +} diff --git a/llama/src/main/java/net/ladenthin/llama/args/GpuSplitMode.java b/llama/src/main/java/net/ladenthin/llama/args/GpuSplitMode.java index 1b4749e2c..93b37e401 100644 --- a/llama/src/main/java/net/ladenthin/llama/args/GpuSplitMode.java +++ b/llama/src/main/java/net/ladenthin/llama/args/GpuSplitMode.java @@ -14,7 +14,13 @@ public enum GpuSplitMode implements CliArg { /** Split by transformer layer across GPUs. */ LAYER("layer"), /** Split by tensor row across GPUs. */ - ROW("row"); + ROW("row"), + /** + * Split weights and KV cache across GPUs and run them in parallel (tensor parallelism). Upstream + * marks it EXPERIMENTAL; since llama.cpp b11450 (#26610) it works across RPC servers as well, which + * then reduce their partial results with each other directly. + */ + TENSOR("tensor"); private final String argValue; diff --git a/llama/src/main/java/net/ladenthin/llama/args/ModelOption.java b/llama/src/main/java/net/ladenthin/llama/args/ModelOption.java index 43208b6a7..145b2b3d0 100644 --- a/llama/src/main/java/net/ladenthin/llama/args/ModelOption.java +++ b/llama/src/main/java/net/ladenthin/llama/args/ModelOption.java @@ -287,6 +287,9 @@ public enum ModelOption { /** CLI option {@code --spec-draft-p-min}. */ SPEC_DRAFT_P_MIN("--spec-draft-p-min"), + /** CLI option {@code --spec-draft-sampling}. */ + SPEC_DRAFT_SAMPLING("--spec-draft-sampling"), + /** CLI option {@code --split-mode}. */ SPLIT_MODE("--split-mode"), diff --git a/llama/src/main/java/net/ladenthin/llama/json/RouterModelsResponseParser.java b/llama/src/main/java/net/ladenthin/llama/json/RouterModelsResponseParser.java index a148cf0e3..97b1b9222 100644 --- a/llama/src/main/java/net/ladenthin/llama/json/RouterModelsResponseParser.java +++ b/llama/src/main/java/net/ladenthin/llama/json/RouterModelsResponseParser.java @@ -27,6 +27,7 @@ * "data": [ * {"id": "Qwen3-0.6B-Q4_K_M", * "status": {"value": "loaded", "args": [...]}, + * "architecture": {"input_modalities": ["text"], "output_modalities": ["text"]}, * "source": "models_dir", ...}, * {"id": "broken-model", * "status": {"value": "unloaded", "failed": true, "exit_code": 1}, ...} @@ -63,7 +64,9 @@ public List parse(String json) { * identifier is read from {@code "id"} (falling back to {@code "name"}). The lifecycle * status comes from {@code status.value}; a missing status maps to * {@link RouterModel.Status#UNKNOWN} with an empty raw value. The failure marker is read - * from {@code status.failed} / {@code status.exit_code}. + * from {@code status.failed} / {@code status.exit_code}, the modalities from + * {@code architecture.input_modalities} / {@code architecture.output_modalities} (empty when + * absent, as from a server before llama.cpp b11429). * * @param root pre-parsed router {@code GET /models} response * @return list of models; empty list when no entry array is present @@ -81,13 +84,25 @@ public List parse(JsonNode root) { String id = entry.path("id").asText(entry.path("name").asText("")); JsonNode status = entry.path("status"); String statusValue = status.path("value").asText(""); + JsonNode architecture = entry.path("architecture"); models.add(new RouterModel( id, RouterModel.Status.fromValue(statusValue), statusValue, status.path("failed").asBoolean(false), - status.path("exit_code").asInt(0))); + status.path("exit_code").asInt(0), + strings(architecture.path("input_modalities")), + strings(architecture.path("output_modalities")))); } return models; } + + /** The string elements of a JSON array; empty for a missing node (servers before b11429). */ + private static List strings(JsonNode array) { + List values = new ArrayList(); + for (JsonNode value : array) { + values.add(value.asText()); + } + return values; + } } diff --git a/llama/src/main/java/net/ladenthin/llama/parameters/ModelParameters.java b/llama/src/main/java/net/ladenthin/llama/parameters/ModelParameters.java index 05e52a510..dd8541cd2 100644 --- a/llama/src/main/java/net/ladenthin/llama/parameters/ModelParameters.java +++ b/llama/src/main/java/net/ladenthin/llama/parameters/ModelParameters.java @@ -1336,6 +1336,22 @@ public ModelParameters setDraftPMin(float draftPMin) { return putScalar(ModelOption.SPEC_DRAFT_P_MIN, draftPMin); } + /** + * Set how speculative decoding samples the draft ({@code --spec-draft-sampling}, llama.cpp b11368). + * + *

{@link DraftSampling#GREEDY}, upstream's default, drafts the argmax token at every position; + * {@link DraftSampling#PROBABILISTIC} samples the draft and has the target verify it by rejection + * sampling, which accepts more drafted tokens for a request with a temperature above zero. It + * applies to a draft model ({@link #setModelDraft(String)}) and to a model's own MTP heads; at + * temperature zero both modes behave the same.

+ * + * @param mode the draft sampling mode + * @return this builder + */ + public ModelParameters setDraftSampling(DraftSampling mode) { + return putEnum(ModelOption.SPEC_DRAFT_SAMPLING, mode); + } + /** * Set the comma-separated list of devices to use for offloading the draft model. * diff --git a/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java b/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java index 4928ee2ea..d42550a1d 100644 --- a/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java +++ b/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java @@ -9,13 +9,13 @@ * library was compiled against, exposed as a compile-time constant so callers can render a badge or * emit a startup log line without loading the native library. * - *

{@link #LLAMA_CPP_VERSION} is a pure-Java string ({@code "b11320"}) that mirrors the + *

{@link #LLAMA_CPP_VERSION} is a pure-Java string ({@code "b11457"}) that mirrors the * {@code GIT_TAG} in {@code llama/CMakeLists.txt}. It is available even when {@code libjllama} is * absent (pure-Java checkout, before {@code System.load}), which is what makes it suitable for a * lightweight version badge in Android or other UIs.

* *

For the authoritative value that is baked into the native binary — the build number - * plus the resolved upstream commit, e.g. {@code "b11320-"} — call + * plus the resolved upstream commit, e.g. {@code "b11457-"} — call * {@link net.ladenthin.llama.LlamaModel#getLlamaCppBuildInfo()} instead; that reads llama.cpp's own * {@code build-info} through JNI and therefore cannot drift from the compiled library (but requires * the native library to be loaded).

@@ -23,14 +23,14 @@ public final class LlamaCppVersion { /** - * The pinned llama.cpp release tag this library was built against, e.g. {@code "b11320"}. + * The pinned llama.cpp release tag this library was built against, e.g. {@code "b11457"}. * *

Kept in lockstep with {@code GIT_TAG} in {@code llama/CMakeLists.txt} — see the * "Upgrading/Downgrading llama.cpp Version" checklist in {@code CLAUDE.md}. This is the * compile-time pin; use {@link net.ladenthin.llama.LlamaModel#getLlamaCppBuildInfo()} for the * value actually linked into the native binary.

*/ - public static final String LLAMA_CPP_VERSION = "b11320"; + public static final String LLAMA_CPP_VERSION = "b11457"; // Constants holder — not instantiable. private LlamaCppVersion() {} diff --git a/llama/src/main/java/net/ladenthin/llama/value/ModelMeta.java b/llama/src/main/java/net/ladenthin/llama/value/ModelMeta.java index 5c1af0e3f..e9aa33046 100644 --- a/llama/src/main/java/net/ladenthin/llama/value/ModelMeta.java +++ b/llama/src/main/java/net/ladenthin/llama/value/ModelMeta.java @@ -5,6 +5,9 @@ package net.ladenthin.llama.value; import com.fasterxml.jackson.databind.JsonNode; +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; import lombok.EqualsAndHashCode; /** @@ -24,6 +27,12 @@ @EqualsAndHashCode public final class ModelMeta { + /** + * The output modality of a native decision model (llama.cpp b11429+), served by + * {@link net.ladenthin.llama.LlamaModel#handleSystemOne(String)}. See {@link #isDecisionModel()}. + */ + public static final String OUTPUT_MODALITY_DECISIONS = "decisions"; + private final JsonNode node; /** @@ -120,6 +129,43 @@ public boolean supportsVideo() { return node.at("/modalities/video").asBoolean(false); } + /** + * What the model can read, as upstream's {@code GET /models} reports it in + * {@code architecture.input_modalities}: always {@code "text"}, plus {@code "image"}, + * {@code "audio"} and {@code "video"} for each media type the loaded projector supports. + * + *

Empty for metadata from a vocab-only load or from a build before llama.cpp b11429.

+ * + * @return the input modalities, unmodifiable + */ + public List getInputModalities() { + return strings(node.path("input_modalities")); + } + + /** + * What the model can produce, as upstream's {@code GET /models} reports it in + * {@code architecture.output_modalities}: {@code ["decisions"]} for a native decision model, + * {@code ["text"]} otherwise. Upstream documents {@code "text"} as a compatibility default rather + * than a promise that the model generates text, and new values may appear, so test for membership. + * + *

Empty for metadata from a vocab-only load or from a build before llama.cpp b11429.

+ * + * @return the output modalities, unmodifiable + */ + public List getOutputModalities() { + return strings(node.path("output_modalities")); + } + + /** + * Whether the model is a native decision model, i.e. answers + * {@link net.ladenthin.llama.LlamaModel#handleSystemOne(String)} instead of generating text. + * + * @return {@code true} if {@link #getOutputModalities()} contains {@value #OUTPUT_MODALITY_DECISIONS} + */ + public boolean isDecisionModel() { + return getOutputModalities().contains(OUTPUT_MODALITY_DECISIONS); + } + /** * The model architecture string from GGUF {@code general.architecture} metadata * (e.g. {@code "llama"}, {@code "gemma3"}, {@code "mistral"}). @@ -212,6 +258,14 @@ public JsonNode asJson() { return node; } + private static List strings(JsonNode array) { + List values = new ArrayList<>(); + for (JsonNode value : array) { + values.add(value.asText()); + } + return Collections.unmodifiableList(values); + } + /** Re-serializes to compact JSON. Suitable for {@code assertEquals} in tests. */ @Override public String toString() { diff --git a/llama/src/main/java/net/ladenthin/llama/value/RouterModel.java b/llama/src/main/java/net/ladenthin/llama/value/RouterModel.java index 6ee8c00ae..1cfed7673 100644 --- a/llama/src/main/java/net/ladenthin/llama/value/RouterModel.java +++ b/llama/src/main/java/net/ladenthin/llama/value/RouterModel.java @@ -4,14 +4,20 @@ package net.ladenthin.llama.value; +import java.util.ArrayList; +import java.util.Collection; +import java.util.Collections; +import java.util.List; import lombok.EqualsAndHashCode; /** * One model entry from the router-mode model registry (the upstream {@code GET /models} * response served by {@link net.ladenthin.llama.server.NativeServer} when started with * {@code --models-dir}). Carries the model identifier, its lifecycle {@link Status}, the raw - * status string as emitted by the server, and the failure marker the router attaches when a - * worker exited abnormally. + * status string as emitted by the server, the failure marker the router attaches when a + * worker exited abnormally, and the model's input and output modalities (llama.cpp b11429+), which + * the router computes without loading the model -- so {@link #isDecisionModel()} finds a decision + * model for {@code /v1/systemone} before its first load. * *

{@code equals}/{@code hashCode} are generated by Lombok over all fields. * {@code toString} is intentionally handwritten (not Lombok-generated) so that router traces @@ -78,6 +84,8 @@ public static Status fromValue(String value) { private final String statusValue; private final boolean failed; private final int exitCode; + private final List inputModalities; + private final List outputModalities; /** * Construct a router model entry. @@ -93,11 +101,42 @@ public static Status fromValue(String value) { * when not failed */ public RouterModel(String id, Status status, String statusValue, boolean failed, int exitCode) { + this( + id, + status, + statusValue, + failed, + exitCode, + Collections.emptyList(), + Collections.emptyList()); + } + + /** + * Construct a router model entry with the model's modalities. + * + * @param id the model identifier + * @param status the parsed lifecycle status + * @param statusValue the raw {@code status.value} string as emitted by the server + * @param failed whether the router flagged the model's worker as failed + * @param exitCode the failed worker's exit code; {@code 0} when not failed + * @param inputModalities {@code architecture.input_modalities}; empty when the server omits it + * @param outputModalities {@code architecture.output_modalities}; empty when the server omits it + */ + public RouterModel( + String id, + Status status, + String statusValue, + boolean failed, + int exitCode, + Collection inputModalities, + Collection outputModalities) { this.id = id; this.status = status; this.statusValue = statusValue; this.failed = failed; this.exitCode = exitCode; + this.inputModalities = Collections.unmodifiableList(new ArrayList<>(inputModalities)); + this.outputModalities = Collections.unmodifiableList(new ArrayList<>(outputModalities)); } /** @@ -140,6 +179,39 @@ public int getExitCode() { return exitCode; } + /** + * What the model can read ({@code architecture.input_modalities}): always {@code "text"}, plus + * {@code "image"} / {@code "audio"} for what its projector supports. The router computes this + * offline, so it cannot see {@code "video"} until the model has been loaded once. + * + * @return the input modalities, unmodifiable; empty when the server did not report them + */ + public List getInputModalities() { + return inputModalities; + } + + /** + * What the model can produce ({@code architecture.output_modalities}): {@code ["decisions"]} for a + * native decision model, {@code ["text"]} otherwise. + * + * @return the output modalities, unmodifiable; empty when the server did not report them + */ + public List getOutputModalities() { + return outputModalities; + } + + /** + * Whether the model is a native decision model, to be served with {@code /v1/systemone} + * ({@link net.ladenthin.llama.LlamaModel#handleSystemOne(String)} in-process). Known before the + * model's first load, after an unload and while it sleeps. + * + * @return {@code true} if {@link #getOutputModalities()} contains + * {@value ModelMeta#OUTPUT_MODALITY_DECISIONS} + */ + public boolean isDecisionModel() { + return outputModalities.contains(ModelMeta.OUTPUT_MODALITY_DECISIONS); + } + @Override public String toString() { if (failed) { diff --git a/llama/src/test/cpp/test_json_helpers.cpp b/llama/src/test/cpp/test_json_helpers.cpp index 0319fcebb..cfbf11a71 100644 --- a/llama/src/test/cpp/test_json_helpers.cpp +++ b/llama/src/test/cpp/test_json_helpers.cpp @@ -659,3 +659,25 @@ TEST(ServerMetricsToJson, CountersAreNumbersNotBooleans) { EXPECT_TRUE(j.at(key).is_number()) << "key is not a number: " << key; } } + +// ============================================================ +// route_error_message +// ============================================================ + +TEST(RouteErrorMessage, ReturnsTheMessageOfAnUpstreamErrorBody) { + // The shape server_res_generator::error() writes, e.g. post_systemone on a non-decision model. + const std::string body = safe_json_to_str( + {{"error", format_error_response("This model is not a decision model", ERROR_TYPE_NOT_SUPPORTED)}}); + EXPECT_EQ(route_error_message(body), "This model is not a decision model"); +} + +TEST(RouteErrorMessage, ReturnsABodyThatIsNotJsonUnchanged) { + EXPECT_EQ(route_error_message("upstream crashed"), "upstream crashed"); + EXPECT_EQ(route_error_message(""), ""); +} + +TEST(RouteErrorMessage, ReturnsJsonWithoutAnErrorMessageUnchanged) { + EXPECT_EQ(route_error_message(R"({"answers":{}})"), R"({"answers":{}})"); + EXPECT_EQ(route_error_message(R"({"error":"plain"})"), R"({"error":"plain"})"); + EXPECT_EQ(route_error_message(R"({"error":{"message":42}})"), R"({"error":{"message":42}})"); +} diff --git a/llama/src/test/cpp/test_kolibri1.cpp b/llama/src/test/cpp/test_kolibri1.cpp new file mode 100644 index 000000000..205e6d736 --- /dev/null +++ b/llama/src/test/cpp/test_kolibri1.cpp @@ -0,0 +1,624 @@ +// SPDX-FileCopyrightText: 2026 Bernard Ladenthin +// +// SPDX-License-Identifier: MIT +// +// Runnable guard for patches/0016 (Aleph Alpha Kolibri-1, architecture "kolibri1"). +// +// Writes tiny random Kolibri-1 GGUFs, runs them through the real library on the CPU and compares +// every logit with an independent reference written from Aleph Alpha's own vLLM implementation +// (aleph_alpha_inference/kolibri1.py at 049a6a7) -- not from the patch. The model is too small to +// mean anything, but it exercises each architectural decision the patch had to get right: +// +// - RoPE on sliding-window layers only (the layer pattern puts a full-attention layer in the +// middle, so a "rope every layer" or "no rope at all" graph is off immediately); +// - the sliding-window mask, with sequences longer than the window, both as one batch and token +// by token through the iSWA KV cache; +// - the router: top-k on logits + expert_bias, weights = unbiased sigmoid(logits). The biases are +// large enough that DeepSeek-V3's router (top-k on sigmoid(logits) + bias) picks other experts; +// one test asserts exactly that, so the comparison cannot pass vacuously; +// - sandwich norms, the ungated shared expert, optional renormalization (norm_topk_prob); +// - both published GGUF dialects (gating_func 2 + pre "qwen2" + output.weight, and gating_func 5 +// + pre "kolibri1" + tied output), and the rejection of a gating function that is neither. +// +// A llama.cpp bump that drops the patch fails these at load time ("unknown model architecture"). +// When upstream adds kolibri1 itself and 0016 is dropped, these tests are what must keep passing. + +#include "llama.h" +#include "gguf.h" +#include "ggml.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { + +constexpr int N_VOCAB = 256; // one GPT-2 byte token per byte +constexpr int N_EMBD = 64; +constexpr int N_HEAD = 4; +constexpr int N_HEAD_KV = 2; +constexpr int HEAD_DIM = 16; +constexpr int N_EXPERT = 8; +constexpr int N_EXPERT_USED = 2; +constexpr int N_FF_EXP = 32; +constexpr int N_FF_SHEXP = 48; +constexpr int N_SWA = 4; +constexpr int N_CTX_TRAIN = 128; +constexpr float RMS_EPS = 1e-6f; +constexpr float ROPE_BASE = 10000.0f; +// sliding, sliding, FULL, sliding, FULL -- a full layer in the middle, unlike the default 4:1 pattern +const std::vector IS_SWA = {true, true, false, true, false}; +const int N_LAYER = (int)IS_SWA.size(); + +using Mat = std::vector; // row-major [rows][cols], i.e. the GGML layout of ne = {cols, rows} + +struct Layer { + Mat attn_norm, attn_post_norm, wq, wk, wv, wo, q_norm, k_norm; + Mat ffn_norm, ffn_post_norm, gate_inp, exp_bias; + std::vector exp_gate, exp_up, exp_down; // per expert: [n_ff][n_embd], [n_ff][n_embd], [n_embd][n_ff] + Mat sh_gate, sh_up, sh_down; +}; + +struct Weights { + Mat tok_embd, output_norm, output; // output empty = tied to tok_embd + std::vector layers; +}; + +Mat random(std::mt19937 &rng, size_t n, float scale, float offset = 0.0f) { + std::normal_distribution dist(0.0f, 1.0f); + Mat m(n); + for (float &v : m) { + v = offset + scale * dist(rng); + } + return m; +} + +Weights make_weights(uint32_t seed, bool tied_output) { + std::mt19937 rng(seed); + Weights w; + w.tok_embd = random(rng, (size_t)N_VOCAB * N_EMBD, 1.0f); + w.output_norm = random(rng, N_EMBD, 0.1f, 1.0f); + if (!tied_output) { + w.output = random(rng, (size_t)N_VOCAB * N_EMBD, 0.2f); + } + for (int il = 0; il < N_LAYER; ++il) { + Layer l; + l.attn_norm = random(rng, N_EMBD, 0.1f, 1.0f); + l.attn_post_norm = random(rng, N_EMBD, 0.1f, 1.0f); + l.wq = random(rng, (size_t)N_HEAD * HEAD_DIM * N_EMBD, 0.2f); + l.wk = random(rng, (size_t)N_HEAD_KV * HEAD_DIM * N_EMBD, 0.2f); + l.wv = random(rng, (size_t)N_HEAD_KV * HEAD_DIM * N_EMBD, 0.2f); + l.wo = random(rng, (size_t)N_EMBD * N_HEAD * HEAD_DIM, 0.2f); + l.q_norm = random(rng, HEAD_DIM, 0.1f, 1.0f); + l.k_norm = random(rng, HEAD_DIM, 0.1f, 1.0f); + l.ffn_norm = random(rng, N_EMBD, 0.1f, 1.0f); + l.ffn_post_norm = random(rng, N_EMBD, 0.1f, 1.0f); + l.gate_inp = random(rng, (size_t)N_EXPERT * N_EMBD, 0.5f); + // trained Kolibri biases reach ~20; large biases are what separate the two routers + l.exp_bias = random(rng, N_EXPERT, 5.0f); + for (int e = 0; e < N_EXPERT; ++e) { + l.exp_gate.push_back(random(rng, (size_t)N_FF_EXP * N_EMBD, 0.2f)); + l.exp_up.push_back(random(rng, (size_t)N_FF_EXP * N_EMBD, 0.2f)); + l.exp_down.push_back(random(rng, (size_t)N_EMBD * N_FF_EXP, 0.2f)); + } + l.sh_gate = random(rng, (size_t)N_FF_SHEXP * N_EMBD, 0.2f); + l.sh_up = random(rng, (size_t)N_FF_SHEXP * N_EMBD, 0.2f); + l.sh_down = random(rng, (size_t)N_EMBD * N_FF_SHEXP, 0.2f); + w.layers.push_back(std::move(l)); + } + return w; +} + +// --------------------------------------------------------------------------------------------- +// The reference: Kolibri1DecoderLayer / Kolibri1Attention / sigmoid_logit_add_routing, in double. +// --------------------------------------------------------------------------------------------- + +enum class Router { KOLIBRI, DEEPSEEK }; + +using Vec = std::vector; + +Vec matvec(const Mat &m, const Vec &x, int rows, int cols) { + Vec y(rows, 0.0); + for (int r = 0; r < rows; ++r) { + for (int c = 0; c < cols; ++c) { + y[r] += (double)m[(size_t)r * cols + c] * x[c]; + } + } + return y; +} + +Vec rms_norm(const Vec &x, const Mat &w, size_t off = 0, size_t n = 0) { + if (n == 0) { + n = x.size(); + } + double ss = 0.0; + for (size_t i = 0; i < n; ++i) { + ss += x[off + i] * x[off + i]; + } + const double scale = 1.0 / std::sqrt(ss / (double)n + (double)RMS_EPS); + Vec y(n); + for (size_t i = 0; i < n; ++i) { + y[i] = x[off + i] * scale * (double)w[i]; + } + return y; +} + +double silu(double v) { return v / (1.0 + std::exp(-v)); } +double sigmoid(double v) { return 1.0 / (1.0 + std::exp(-v)); } + +// NEOX rotary embedding over one head: dimension i pairs with i + HEAD_DIM/2 +void rope_neox(Vec &h, int pos) { + const int half = HEAD_DIM / 2; + for (int i = 0; i < half; ++i) { + const double theta = (double)pos * std::pow((double)ROPE_BASE, -2.0 * i / HEAD_DIM); + const double c = std::cos(theta), s = std::sin(theta); + const double x0 = h[i], x1 = h[i + half]; + h[i] = x0 * c - x1 * s; + h[i + half] = x0 * s + x1 * c; + } +} + +Vec swiglu(const Mat &gate, const Mat &up, const Mat &down, const Vec &x, int n_ff) { + const Vec g = matvec(gate, x, n_ff, N_EMBD); + const Vec u = matvec(up, x, n_ff, N_EMBD); + Vec a(n_ff); + for (int i = 0; i < n_ff; ++i) { + a[i] = silu(g[i]) * u[i]; + } + return matvec(down, a, N_EMBD, n_ff); +} + +std::vector route(const Layer &l, const Vec &logits, Router router) { + std::vector score(N_EXPERT); + for (int e = 0; e < N_EXPERT; ++e) { + score[e] = router == Router::KOLIBRI ? logits[e] + l.exp_bias[e] : sigmoid(logits[e]) + l.exp_bias[e]; + } + std::vector ids(N_EXPERT); + std::iota(ids.begin(), ids.end(), 0); + std::partial_sort(ids.begin(), ids.begin() + N_EXPERT_USED, ids.end(), + [&](int a, int b) { return score[a] > score[b]; }); + ids.resize(N_EXPERT_USED); + return ids; +} + +// Logits of every position of `tokens` (one causal pass), [n_tokens][N_VOCAB]. +std::vector reference(const Weights &w, const std::vector &tokens, bool renorm, + Router router = Router::KOLIBRI) { + const int n = (int)tokens.size(); + std::vector x(n, Vec(N_EMBD)); + for (int t = 0; t < n; ++t) { + for (int i = 0; i < N_EMBD; ++i) { + x[t][i] = w.tok_embd[(size_t)tokens[t] * N_EMBD + i]; + } + } + const int group = N_HEAD / N_HEAD_KV; + for (int il = 0; il < N_LAYER; ++il) { + const Layer &l = w.layers[il]; + // q/k/v of every position, q/k normed per head, RoPE on sliding-window layers only + std::vector q(n), k(n), v(n); + for (int t = 0; t < n; ++t) { + const Vec h = rms_norm(x[t], l.attn_norm); + const Vec qr = matvec(l.wq, h, N_HEAD * HEAD_DIM, N_EMBD); + const Vec kr = matvec(l.wk, h, N_HEAD_KV * HEAD_DIM, N_EMBD); + v[t] = matvec(l.wv, h, N_HEAD_KV * HEAD_DIM, N_EMBD); + q[t].resize(qr.size()); + k[t].resize(kr.size()); + for (int hd = 0; hd < N_HEAD; ++hd) { + Vec qh = rms_norm(qr, l.q_norm, (size_t)hd * HEAD_DIM, HEAD_DIM); + if (IS_SWA[il]) { + rope_neox(qh, t); + } + std::copy(qh.begin(), qh.end(), q[t].begin() + (size_t)hd * HEAD_DIM); + } + for (int hd = 0; hd < N_HEAD_KV; ++hd) { + Vec kh = rms_norm(kr, l.k_norm, (size_t)hd * HEAD_DIM, HEAD_DIM); + if (IS_SWA[il]) { + rope_neox(kh, t); + } + std::copy(kh.begin(), kh.end(), k[t].begin() + (size_t)hd * HEAD_DIM); + } + } + std::vector next(n); + for (int t = 0; t < n; ++t) { + // causal attention; a sliding-window layer sees the last N_SWA positions (itself included) + Vec attn(N_HEAD * HEAD_DIM, 0.0); + for (int hd = 0; hd < N_HEAD; ++hd) { + const int kvh = hd / group; + const int first = IS_SWA[il] ? std::max(0, t - N_SWA + 1) : 0; + std::vector s; + double mx = -1e300; + for (int p = first; p <= t; ++p) { + double d = 0.0; + for (int i = 0; i < HEAD_DIM; ++i) { + d += q[t][(size_t)hd * HEAD_DIM + i] * k[p][(size_t)kvh * HEAD_DIM + i]; + } + s.push_back(d / std::sqrt((double)HEAD_DIM)); + mx = std::max(mx, s.back()); + } + double sum = 0.0; + for (double &e : s) { + e = std::exp(e - mx); + sum += e; + } + for (int p = first; p <= t; ++p) { + const double pr = s[p - first] / sum; + for (int i = 0; i < HEAD_DIM; ++i) { + attn[(size_t)hd * HEAD_DIM + i] += pr * v[p][(size_t)kvh * HEAD_DIM + i]; + } + } + } + const Vec o = rms_norm(matvec(l.wo, attn, N_EMBD, N_HEAD * HEAD_DIM), l.attn_post_norm); + Vec x1(N_EMBD); + for (int i = 0; i < N_EMBD; ++i) { + x1[i] = x[t][i] + o[i]; + } + // MoE: route, routed experts weighted by the unbiased sigmoid, plus the shared expert + const Vec h2 = rms_norm(x1, l.ffn_norm); + const Vec logits = matvec(l.gate_inp, h2, N_EXPERT, N_EMBD); + const std::vector sel = route(l, logits, router); + std::vector wt; + double wsum = 0.0; + for (int e : sel) { + wt.push_back(sigmoid(logits[e])); + wsum += wt.back(); + } + Vec ffn = swiglu(l.sh_gate, l.sh_up, l.sh_down, h2, N_FF_SHEXP); + for (size_t j = 0; j < sel.size(); ++j) { + const double weight = renorm ? wt[j] / wsum : wt[j]; + const Vec y = swiglu(l.exp_gate[sel[j]], l.exp_up[sel[j]], l.exp_down[sel[j]], h2, N_FF_EXP); + for (int i = 0; i < N_EMBD; ++i) { + ffn[i] += weight * y[i]; + } + } + const Vec f = rms_norm(ffn, l.ffn_post_norm); + next[t].resize(N_EMBD); + for (int i = 0; i < N_EMBD; ++i) { + next[t][i] = x1[i] + f[i]; + } + } + x = std::move(next); + } + const Mat &out = w.output.empty() ? w.tok_embd : w.output; + std::vector res(n); + for (int t = 0; t < n; ++t) { + res[t] = matvec(out, rms_norm(x[t], w.output_norm), N_VOCAB, N_EMBD); + } + return res; +} + +// --------------------------------------------------------------------------------------------- +// GGUF writer (public gguf API; F32 tensors, GGML ne = {cols, rows[, n_expert]}) +// --------------------------------------------------------------------------------------------- + +struct Dialect { + uint32_t gating_func; // 0 = key absent + const char *pre; + bool write_weights_norm; + bool renorm; + bool tied_output; +}; + +// The 256 printable code points GPT-2's byte-level BPE maps the bytes to, as UTF-8. +std::vector byte_tokens() { + std::vector bs; + for (int b = '!'; b <= '~'; ++b) + bs.push_back(b); + for (int b = 0xA1; b <= 0xAC; ++b) + bs.push_back(b); + for (int b = 0xAE; b <= 0xFF; ++b) + bs.push_back(b); + std::vector cs = bs; + int extra = 0; + for (int b = 0; b < 256; ++b) { + if (std::find(bs.begin(), bs.end(), b) == bs.end()) { + bs.push_back(b); + cs.push_back(256 + extra++); + } + } + std::vector tokens(256); + for (size_t i = 0; i < bs.size(); ++i) { + const int cp = cs[i]; + std::string s; + if (cp < 0x80) { + s += (char)cp; + } else { + s += (char)(0xC0 | (cp >> 6)); + s += (char)(0x80 | (cp & 0x3F)); + } + tokens[bs[i]] = s; + } + return tokens; +} + +void add_tensor(gguf_context *g, ggml_context *ctx, const std::string &name, const Mat &data, + std::initializer_list ne) { + ggml_tensor *t = nullptr; + const std::vector d(ne); + if (d.size() == 1) { + t = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, d[0]); + } else if (d.size() == 2) { + t = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, d[0], d[1]); + } else { + t = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, d[0], d[1], d[2]); + } + ASSERT_EQ((size_t)ggml_nelements(t), data.size()) << name; + std::copy(data.begin(), data.end(), (float *)t->data); + ggml_set_name(t, name.c_str()); + gguf_add_tensor(g, t); +} + +Mat concat(const std::vector &parts) { + Mat m; + for (const Mat &p : parts) { + m.insert(m.end(), p.begin(), p.end()); + } + return m; +} + +std::string write_gguf(const Weights &w, const Dialect &d) { + static std::atomic counter{0}; + static const unsigned run_id = std::random_device{}(); + const std::string path = (std::filesystem::temp_directory_path() / + ("jllama-kolibri1-" + std::to_string(run_id) + "-" + std::to_string(counter++) + ".gguf")) + .string(); + + gguf_context *g = gguf_init_empty(); + const std::string a = "kolibri1"; + gguf_set_val_str(g, "general.architecture", a.c_str()); + gguf_set_val_u32(g, (a + ".context_length").c_str(), N_CTX_TRAIN); + gguf_set_val_u32(g, (a + ".embedding_length").c_str(), N_EMBD); + gguf_set_val_u32(g, (a + ".block_count").c_str(), N_LAYER); + gguf_set_val_u32(g, (a + ".feed_forward_length").c_str(), N_FF_EXP); + gguf_set_val_u32(g, (a + ".attention.head_count").c_str(), N_HEAD); + gguf_set_val_u32(g, (a + ".attention.head_count_kv").c_str(), N_HEAD_KV); + gguf_set_val_u32(g, (a + ".attention.key_length").c_str(), HEAD_DIM); + gguf_set_val_u32(g, (a + ".attention.value_length").c_str(), HEAD_DIM); + gguf_set_val_f32(g, (a + ".attention.layer_norm_rms_epsilon").c_str(), RMS_EPS); + gguf_set_val_u32(g, (a + ".attention.sliding_window").c_str(), N_SWA); + std::vector pattern(IS_SWA.begin(), IS_SWA.end()); // gguf bool is one byte + gguf_set_arr_data(g, (a + ".attention.sliding_window_pattern").c_str(), GGUF_TYPE_BOOL, pattern.data(), + pattern.size()); + gguf_set_val_u32(g, (a + ".rope.dimension_count").c_str(), HEAD_DIM); + gguf_set_val_f32(g, (a + ".rope.freq_base").c_str(), ROPE_BASE); + gguf_set_val_u32(g, (a + ".expert_count").c_str(), N_EXPERT); + gguf_set_val_u32(g, (a + ".expert_used_count").c_str(), N_EXPERT_USED); + gguf_set_val_u32(g, (a + ".expert_feed_forward_length").c_str(), N_FF_EXP); + gguf_set_val_u32(g, (a + ".expert_shared_feed_forward_length").c_str(), N_FF_SHEXP); + gguf_set_val_u32(g, (a + ".expert_shared_count").c_str(), 1); + if (d.write_weights_norm) { + gguf_set_val_bool(g, (a + ".expert_weights_norm").c_str(), d.renorm); + } + if (d.gating_func != 0) { + gguf_set_val_u32(g, (a + ".expert_gating_func").c_str(), d.gating_func); + } + + const std::vector tokens = byte_tokens(); + std::vector token_ptrs; + for (const std::string &s : tokens) { + token_ptrs.push_back(s.c_str()); + } + std::vector token_types(N_VOCAB, 1); // LLAMA_TOKEN_TYPE_NORMAL + const char *no_merges[] = {"Ġ Ġ"}; + gguf_set_val_str(g, "tokenizer.ggml.model", "gpt2"); + gguf_set_val_str(g, "tokenizer.ggml.pre", d.pre); + gguf_set_arr_str(g, "tokenizer.ggml.tokens", token_ptrs.data(), token_ptrs.size()); + gguf_set_arr_data(g, "tokenizer.ggml.token_type", GGUF_TYPE_INT32, token_types.data(), token_types.size()); + gguf_set_arr_str(g, "tokenizer.ggml.merges", no_merges, 1); + + ggml_init_params ip = {64u * 1024 * 1024, nullptr, false}; + ggml_context *ctx = ggml_init(ip); + add_tensor(g, ctx, "token_embd.weight", w.tok_embd, {N_EMBD, N_VOCAB}); + add_tensor(g, ctx, "output_norm.weight", w.output_norm, {N_EMBD}); + if (!w.output.empty()) { + add_tensor(g, ctx, "output.weight", w.output, {N_EMBD, N_VOCAB}); + } + for (int il = 0; il < N_LAYER; ++il) { + const Layer &l = w.layers[il]; + const std::string p = "blk." + std::to_string(il) + "."; + add_tensor(g, ctx, p + "attn_norm.weight", l.attn_norm, {N_EMBD}); + add_tensor(g, ctx, p + "post_attention_norm.weight", l.attn_post_norm, {N_EMBD}); + add_tensor(g, ctx, p + "attn_q.weight", l.wq, {N_EMBD, N_HEAD * HEAD_DIM}); + add_tensor(g, ctx, p + "attn_k.weight", l.wk, {N_EMBD, N_HEAD_KV * HEAD_DIM}); + add_tensor(g, ctx, p + "attn_v.weight", l.wv, {N_EMBD, N_HEAD_KV * HEAD_DIM}); + add_tensor(g, ctx, p + "attn_output.weight", l.wo, {N_HEAD * HEAD_DIM, N_EMBD}); + add_tensor(g, ctx, p + "attn_q_norm.weight", l.q_norm, {HEAD_DIM}); + add_tensor(g, ctx, p + "attn_k_norm.weight", l.k_norm, {HEAD_DIM}); + add_tensor(g, ctx, p + "ffn_norm.weight", l.ffn_norm, {N_EMBD}); + add_tensor(g, ctx, p + "post_ffw_norm.weight", l.ffn_post_norm, {N_EMBD}); + add_tensor(g, ctx, p + "ffn_gate_inp.weight", l.gate_inp, {N_EMBD, N_EXPERT}); + add_tensor(g, ctx, p + "exp_probs_b.bias", l.exp_bias, {N_EXPERT}); + add_tensor(g, ctx, p + "ffn_gate_exps.weight", concat(l.exp_gate), {N_EMBD, N_FF_EXP, N_EXPERT}); + add_tensor(g, ctx, p + "ffn_up_exps.weight", concat(l.exp_up), {N_EMBD, N_FF_EXP, N_EXPERT}); + add_tensor(g, ctx, p + "ffn_down_exps.weight", concat(l.exp_down), {N_FF_EXP, N_EMBD, N_EXPERT}); + add_tensor(g, ctx, p + "ffn_gate_shexp.weight", l.sh_gate, {N_EMBD, N_FF_SHEXP}); + add_tensor(g, ctx, p + "ffn_up_shexp.weight", l.sh_up, {N_EMBD, N_FF_SHEXP}); + add_tensor(g, ctx, p + "ffn_down_shexp.weight", l.sh_down, {N_FF_SHEXP, N_EMBD}); + } + const bool ok = gguf_write_to_file(g, path.c_str(), false); + gguf_free(g); + ggml_free(ctx); + return ok ? path : std::string(); +} + +// --------------------------------------------------------------------------------------------- +// Running the library +// --------------------------------------------------------------------------------------------- + +struct Loaded { + llama_model *model = nullptr; + llama_context *ctx = nullptr; + ~Loaded() { + if (ctx) + llama_free(ctx); + if (model) + llama_model_free(model); + } +}; + +void quiet_log(ggml_log_level, const char *, void *) {} + +bool load(const std::string &path, Loaded &out) { + llama_backend_init(); + llama_log_set(quiet_log, nullptr); + llama_model_params mp = llama_model_default_params(); + mp.n_gpu_layers = 0; + out.model = llama_model_load_from_file(path.c_str(), mp); + llama_log_set(nullptr, nullptr); + if (out.model == nullptr) { + return false; + } + llama_context_params cp = llama_context_default_params(); + cp.n_ctx = 64; + cp.n_batch = 64; + cp.n_ubatch = 64; + cp.n_seq_max = 1; + cp.n_threads = 2; + cp.n_threads_batch = 2; + out.ctx = llama_init_from_model(out.model, cp); + return out.ctx != nullptr; +} + +// logits of every position, decoding the whole sequence in one batch +std::vector> run_batch(Loaded &m, const std::vector &tokens) { + llama_batch batch = llama_batch_init((int32_t)tokens.size(), 0, 1); + for (size_t i = 0; i < tokens.size(); ++i) { + batch.token[i] = tokens[i]; + batch.pos[i] = (llama_pos)i; + batch.n_seq_id[i] = 1; + batch.seq_id[i][0] = 0; + batch.logits[i] = 1; + } + batch.n_tokens = (int32_t)tokens.size(); + std::vector> res; + if (llama_decode(m.ctx, batch) == 0) { + for (size_t i = 0; i < tokens.size(); ++i) { + const float *l = llama_get_logits_ith(m.ctx, (int32_t)i); + res.emplace_back(l, l + N_VOCAB); + } + } + llama_batch_free(batch); + return res; +} + +// logits of every position, decoding token by token through the KV cache +std::vector> run_incremental(Loaded &m, const std::vector &tokens) { + llama_memory_clear(llama_get_memory(m.ctx), true); + std::vector> res; + llama_batch batch = llama_batch_init(1, 0, 1); + for (size_t i = 0; i < tokens.size(); ++i) { + batch.token[0] = tokens[i]; + batch.pos[0] = (llama_pos)i; + batch.n_seq_id[0] = 1; + batch.seq_id[0][0] = 0; + batch.logits[0] = 1; + batch.n_tokens = 1; + if (llama_decode(m.ctx, batch) != 0) { + res.clear(); + break; + } + const float *l = llama_get_logits_ith(m.ctx, 0); + res.emplace_back(l, l + N_VOCAB); + } + llama_batch_free(batch); + return res; +} + +double max_abs_diff(const std::vector> &got, const std::vector &want) { + double d = 0.0; + for (size_t t = 0; t < want.size(); ++t) { + for (int i = 0; i < N_VOCAB; ++i) { + d = std::max(d, std::fabs((double)got[t][i] - want[t][i])); + } + } + return d; +} + +double max_abs(const std::vector &v) { + double m = 0.0; + for (const Vec &r : v) { + for (double x : r) { + m = std::max(m, std::fabs(x)); + } + } + return m; +} + +const std::vector TOKENS = {3, 141, 59, 26, 53, 58, 97, 93, 238, 46, 26, 43}; // 12 > N_SWA + +void expect_matches_reference(const Dialect &d, uint32_t seed) { + const Weights w = make_weights(seed, d.tied_output); + const std::string path = write_gguf(w, d); + ASSERT_FALSE(path.empty()); + { + Loaded m; + ASSERT_TRUE(load(path, m)) << "the library did not load the kolibri1 GGUF (pre=" << d.pre + << ", gating_func=" << d.gating_func << ")"; + + const std::vector want = reference(w, TOKENS, d.renorm); + const double tol = 1e-3 * std::max(1.0, max_abs(want)); + + const auto batch = run_batch(m, TOKENS); + ASSERT_EQ(batch.size(), TOKENS.size()); + EXPECT_LT(max_abs_diff(batch, want), tol) << "batch decode"; + + const auto inc = run_incremental(m, TOKENS); + ASSERT_EQ(inc.size(), TOKENS.size()); + EXPECT_LT(max_abs_diff(inc, want), tol) << "token-by-token decode through the iSWA KV cache"; + } + std::remove(path.c_str()); +} + +} // namespace + +// When upstream supports Kolibri-1 and patch 0016 is dropped: the reference comparison must stay +// green (red = upstream computes something else than Aleph Alpha's reference). The Dialect rows +// below are GGUF format, which upstream's converter decides; a red row means GGUFs of that dialect +// no longer load without the patch -- decide that deliberately, then move the row to upstream's +// format and leave the reference alone. + +// The AFMoE-based converter's dialect: gating_func 2, pre "qwen2", an explicit output.weight. +TEST(Kolibri1, AfmoeDialectMatchesTheReference) { expect_matches_reference({2, "qwen2", true, false, false}, 1); } + +// The Qwen3-MoE-based converter's dialect: gating_func 5, pre "kolibri1", tied output, no norm key. +TEST(Kolibri1, Qwen3MoeDialectWithTiedOutputMatchesTheReference) { + expect_matches_reference({5, "kolibri1", false, false, true}, 2); +} + +// norm_topk_prob = true renormalizes the selected sigmoid weights (Kolibri-1 itself ships it off). +TEST(Kolibri1, RenormalizedRoutingMatchesTheReference) { expect_matches_reference({0, "qwen2", true, true, false}, 3); } + +// Keeps the comparisons above honest: with these biases DeepSeek-V3's router (top-k on +// sigmoid(logits) + bias) gives a different model, so a graph that used it would fail them. +TEST(Kolibri1, TheDeepSeekRouterWouldGiveDifferentLogits) { + const Weights w = make_weights(1, false); + const std::vector kolibri = reference(w, TOKENS, false, Router::KOLIBRI); + const std::vector deepseek = reference(w, TOKENS, false, Router::DEEPSEEK); + double d = 0.0; + for (size_t t = 0; t < kolibri.size(); ++t) { + for (int i = 0; i < N_VOCAB; ++i) { + d = std::max(d, std::fabs(kolibri[t][i] - deepseek[t][i])); + } + } + EXPECT_GT(d, 1e-1 * std::max(1.0, max_abs(kolibri))); +} + +// Only the router Kolibri-1 has is accepted; a GGUF claiming another gating function is refused. +TEST(Kolibri1, AnotherGatingFunctionIsRejected) { + const Weights w = make_weights(4, false); + const std::string path = write_gguf(w, {1, "qwen2", true, false, false}); // 1 = softmax + ASSERT_FALSE(path.empty()); + { + Loaded m; + EXPECT_FALSE(load(path, m)); + } + std::remove(path.c_str()); +} diff --git a/llama/src/test/java/net/ladenthin/llama/LlamaModelTest.java b/llama/src/test/java/net/ladenthin/llama/LlamaModelTest.java index 8d45d8192..41b9dea1f 100644 --- a/llama/src/test/java/net/ladenthin/llama/LlamaModelTest.java +++ b/llama/src/test/java/net/ladenthin/llama/LlamaModelTest.java @@ -1403,6 +1403,11 @@ public void testGetModelMeta() throws LlamaException { assertFalse(meta.supportsVision(), "text-only model must not report vision support"); assertFalse(meta.supportsAudio(), "text-only model must not report audio support"); + // GET /models' architecture modalities (llama.cpp b11429): a text-only, non-decision model + assertEquals(Collections.singletonList("text"), meta.getInputModalities(), "input_modalities"); + assertEquals(Collections.singletonList("text"), meta.getOutputModalities(), "output_modalities"); + assertFalse(meta.isDecisionModel(), "CodeLlama is not a decision model"); + // Dynamic access via the underlying JsonNode assertTrue(meta.asJson().has("modalities"), "modalities field must be present"); assertTrue(meta.asJson().has("vocab_type"), "vocab_type field must be present"); diff --git a/llama/src/test/java/net/ladenthin/llama/SystemOneIntegrationTest.java b/llama/src/test/java/net/ladenthin/llama/SystemOneIntegrationTest.java new file mode 100644 index 000000000..f00bfbf44 --- /dev/null +++ b/llama/src/test/java/net/ladenthin/llama/SystemOneIntegrationTest.java @@ -0,0 +1,133 @@ +// SPDX-FileCopyrightText: 2026 Bernard Ladenthin +// +// SPDX-License-Identifier: MIT + +package net.ladenthin.llama; + +import static org.hamcrest.MatcherAssert.assertThat; +import static org.hamcrest.Matchers.closeTo; +import static org.hamcrest.Matchers.contains; +import static org.hamcrest.Matchers.containsString; +import static org.hamcrest.Matchers.greaterThan; +import static org.hamcrest.Matchers.is; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assumptions.assumeTrue; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import java.io.File; +import java.util.ArrayList; +import java.util.Iterator; +import java.util.List; +import net.ladenthin.llama.exception.LlamaException; +import net.ladenthin.llama.loader.NativeLibraryPresence; +import net.ladenthin.llama.parameters.ModelParameters; +import org.junit.jupiter.api.Test; + +/** + * {@link LlamaModel#handleSystemOne(String)}, llama.cpp's {@code /v1/systemone} decision API (upstream + * b11361). The JNI method forwards to upstream's own route handler, so what this pins is the bridge: + * that the handler is reachable from a loaded model, that its answer comes back verbatim, and that an + * error body turns into a {@link LlamaException} carrying upstream's message. + * + *

The rejection case runs with the cached draft model on every CI job. The answering cases need a + * decision model, which is not in the CI set; they self-skip unless + * {@link TestConstants#PROP_DECISION_MODEL_PATH} names one.

+ */ +@ClaudeGenerated( + purpose = "Pin the JNI bridge to llama.cpp's /v1/systemone handler: a non-decision model is " + + "rejected with upstream's message, and a decision model answers every question type.") +public class SystemOneIntegrationTest { + + private static final ObjectMapper MAPPER = new ObjectMapper(); + + /** The request of upstream's own {@code test_systemone.py}: one question of each type. */ + private static final String REQUEST = "{" + + "\"state\":\"I was charged twice for my order last week and nobody has replied.\"," + + "\"questions\":{" + + "\"route\":{\"type\":\"choice\",\"instructions\":\"Which team should handle this?\"," + + "\"criteria\":{\"billing\":\"payments and refunds\",\"shipping\":null,\"technical\":null}}," + + "\"urgency\":{\"type\":\"score\",\"instructions\":\"How urgent is this?\"," + + "\"criteria\":[\"can wait\",\"this week\",\"today\",\"right now\"]}," + + "\"angry\":{\"type\":\"noul\",\"instructions\":\"Is the customer angry?\"}" + + "}}"; + + @Test + public void aModelThatIsNotADecisionModelIsRejected() { + assumeTrue(NativeLibraryPresence.onClasspath(), "libjllama not on classpath"); + assumeTrue(new File(TestConstants.DRAFT_MODEL_PATH).exists(), "draft model not found"); + + try (LlamaModel model = new LlamaModel(new ModelParameters() + .setModel(TestConstants.DRAFT_MODEL_PATH) + .setCtxSize(256) + .setGpuLayers(0) + .setFit(false))) { + LlamaException e = assertThrows(LlamaException.class, () -> model.handleSystemOne(REQUEST)); + assertThat(e.getMessage(), containsString("not a decision model")); + } + } + + @Test + public void aDecisionModelAnswersEveryQuestionType() throws Exception { + try (LlamaModel model = loadDecisionModel()) { + JsonNode response = MAPPER.readTree(model.handleSystemOne(REQUEST)); + + JsonNode answers = response.get("answers"); + assertThat(fieldNames(answers), contains("route", "urgency", "angry")); + + JsonNode route = answers.get("route"); + assertThat(route.get("type").asText(), is("choice")); + assertThat(sum(route.get("probabilities")), closeTo(1.0, 1e-3)); + + JsonNode urgency = answers.get("urgency"); + assertThat(urgency.get("type").asText(), is("score")); + assertThat(sum(urgency.get("probabilities")), closeTo(1.0, 1e-3)); + assertThat(urgency.get("score").asDouble(), closeTo(1.5, 1.5)); + + JsonNode angry = answers.get("angry"); + assertThat(angry.get("type").asText(), is("noul")); + assertThat(angry.get("noul").asDouble(), closeTo(0.5, 0.5)); + + assertThat(response.get("usage").get("input_tokens").asInt(), greaterThan(0)); + assertThat(response.get("usage").get("output_tokens").asInt(), is(0)); + } + } + + @Test + public void aDecisionModelRejectsAnUnknownQuestionType() throws Exception { + try (LlamaModel model = loadDecisionModel()) { + String request = "{\"state\":\"x\",\"questions\":{\"q\":{\"type\":\"bogus\",\"instructions\":\"?\"}}}"; + assertThrows(LlamaException.class, () -> model.handleSystemOne(request)); + } + } + + private static LlamaModel loadDecisionModel() { + assumeTrue(NativeLibraryPresence.onClasspath(), "libjllama not on classpath"); + String path = TestConstants.resolveModelProperty(TestConstants.PROP_DECISION_MODEL_PATH); + assumeTrue( + path != null && !path.isEmpty(), + "decision model not set (-D" + TestConstants.PROP_DECISION_MODEL_PATH + "=...)"); + assumeTrue(new File(path).exists(), "decision model file missing: " + path); + return new LlamaModel(new ModelParameters() + .setModel(path) + .setCtxSize(1024) + .setGpuLayers(Integer.getInteger(TestConstants.PROP_TEST_NGL, TestConstants.DEFAULT_TEST_NGL)) + .setFit(false)); + } + + private static List fieldNames(JsonNode node) { + List names = new ArrayList<>(); + for (Iterator it = node.fieldNames(); it.hasNext(); ) { + names.add(it.next()); + } + return names; + } + + private static double sum(JsonNode probabilities) { + double total = 0; + for (JsonNode value : probabilities) { + total += value.asDouble(); + } + return total; + } +} diff --git a/llama/src/test/java/net/ladenthin/llama/TestConstants.java b/llama/src/test/java/net/ladenthin/llama/TestConstants.java index 4a0f4e0ee..98f162e02 100644 --- a/llama/src/test/java/net/ladenthin/llama/TestConstants.java +++ b/llama/src/test/java/net/ladenthin/llama/TestConstants.java @@ -144,6 +144,14 @@ public static String resolveModelProperty(String key) { */ public static final String DEFAULT_VISION_IMAGE_PATH = resolveModelPath("src/test/resources/images/test-image.jpg"); + /** + * System property holding a path to a decision model GGUF (laya, julia-1, lev, openjev, kev, ...) + * for {@code SystemOneIntegrationTest}, which drives llama.cpp's {@code /v1/systemone} API. Not in + * the CI model set, so the decision-model half of that test self-skips when this is unset or the + * file is missing; upstream tests with {@code ggml-org/tinylaya-for-testing-gguf}. + */ + public static final String PROP_DECISION_MODEL_PATH = LlamaSystemProperties.PREFIX + ".decision.model"; + /** * System property holding a path to an audio-input model GGUF (e.g. Ultravox / Qwen2.5-Omni). * Consumed by {@code AudioInputIntegrationTest} (llama.cpp discussion #13759). The test self-skips diff --git a/llama/src/test/java/net/ladenthin/llama/args/DraftSamplingTest.java b/llama/src/test/java/net/ladenthin/llama/args/DraftSamplingTest.java new file mode 100644 index 000000000..793d098c4 --- /dev/null +++ b/llama/src/test/java/net/ladenthin/llama/args/DraftSamplingTest.java @@ -0,0 +1,18 @@ +// SPDX-FileCopyrightText: 2026 Bernard Ladenthin +// +// SPDX-License-Identifier: MIT + +package net.ladenthin.llama.args; + +import java.util.Arrays; +import java.util.Collection; + +public class DraftSamplingTest extends AbstractCliArgEnumTest { + + public static Collection data() { + return Arrays.asList(new Object[][] { + {DraftSampling.GREEDY, "greedy", 2}, + {DraftSampling.PROBABILISTIC, "probabilistic", 2}, + }); + } +} diff --git a/llama/src/test/java/net/ladenthin/llama/args/GpuSplitModeTest.java b/llama/src/test/java/net/ladenthin/llama/args/GpuSplitModeTest.java index fd4f64335..182c68043 100644 --- a/llama/src/test/java/net/ladenthin/llama/args/GpuSplitModeTest.java +++ b/llama/src/test/java/net/ladenthin/llama/args/GpuSplitModeTest.java @@ -11,9 +11,10 @@ public class GpuSplitModeTest extends AbstractCliArgEnumTest { public static Collection data() { return Arrays.asList(new Object[][] { - {GpuSplitMode.NONE, "none", 3}, - {GpuSplitMode.LAYER, "layer", 3}, - {GpuSplitMode.ROW, "row", 3}, + {GpuSplitMode.NONE, "none", 4}, + {GpuSplitMode.LAYER, "layer", 4}, + {GpuSplitMode.ROW, "row", 4}, + {GpuSplitMode.TENSOR, "tensor", 4}, }); } } diff --git a/llama/src/test/java/net/ladenthin/llama/json/RouterModelsResponseParserTest.java b/llama/src/test/java/net/ladenthin/llama/json/RouterModelsResponseParserTest.java index a9395ea1f..ff37dd990 100644 --- a/llama/src/test/java/net/ladenthin/llama/json/RouterModelsResponseParserTest.java +++ b/llama/src/test/java/net/ladenthin/llama/json/RouterModelsResponseParserTest.java @@ -5,6 +5,8 @@ package net.ladenthin.llama.json; import static org.hamcrest.MatcherAssert.assertThat; +import static org.hamcrest.Matchers.contains; +import static org.hamcrest.Matchers.empty; import static org.hamcrest.Matchers.is; import java.util.List; @@ -99,4 +101,26 @@ public void missingArraysYieldEmptyList() { public void unparseableInputYieldsEmptyList() { assertThat(parser.parse("not json").isEmpty(), is(true)); } + + @Test + public void parsesTheArchitectureModalities() { + // GET /models since llama.cpp b11429 (#29987): computed offline, present before the first load. + String json = "{\"data\":[{\"id\":\"decider\",\"status\":{\"value\":\"unloaded\"}," + + "\"architecture\":{\"input_modalities\":[\"text\",\"image\"]," + + "\"output_modalities\":[\"decisions\"]}}]}"; + + RouterModel model = parser.parse(json).get(0); + assertThat(model.getInputModalities(), contains("text", "image")); + assertThat(model.getOutputModalities(), contains("decisions")); + assertThat(model.isDecisionModel(), is(true)); + } + + @Test + public void missingArchitectureYieldsNoModalities() { + // A server before b11429 omits the object; upstream asks clients to fall back, not to guess. + RouterModel model = parser.parse("{\"data\":[{\"id\":\"old\"}]}").get(0); + assertThat(model.getInputModalities(), is(empty())); + assertThat(model.getOutputModalities(), is(empty())); + assertThat(model.isDecisionModel(), is(false)); + } } diff --git a/llama/src/test/java/net/ladenthin/llama/parameters/ModelParametersTest.java b/llama/src/test/java/net/ladenthin/llama/parameters/ModelParametersTest.java index 49a936e41..8500f4fcc 100644 --- a/llama/src/test/java/net/ladenthin/llama/parameters/ModelParametersTest.java +++ b/llama/src/test/java/net/ladenthin/llama/parameters/ModelParametersTest.java @@ -18,6 +18,7 @@ import java.util.List; import net.ladenthin.llama.ClaudeGenerated; import net.ladenthin.llama.args.CacheType; +import net.ladenthin.llama.args.DraftSampling; import net.ladenthin.llama.args.GpuSplitMode; import net.ladenthin.llama.args.LazyMode; import net.ladenthin.llama.args.MiroStat; @@ -764,4 +765,20 @@ public void testSetLazyModeOn() { ModelParameters p = new ModelParameters().setLazyMode(LazyMode.ON); assertThat(p.parameters.get("--lazy-mode"), is("on")); } + + // ------------------------------------------------------------------------- + // setDraftSampling (llama.cpp b11368) + // ------------------------------------------------------------------------- + + @Test + public void testSetDraftSamplingGreedy() { + ModelParameters p = new ModelParameters().setDraftSampling(DraftSampling.GREEDY); + assertThat(p.parameters.get("--spec-draft-sampling"), is("greedy")); + } + + @Test + public void testSetDraftSamplingProbabilistic() { + ModelParameters p = new ModelParameters().setDraftSampling(DraftSampling.PROBABILISTIC); + assertThat(p.parameters.get("--spec-draft-sampling"), is("probabilistic")); + } } diff --git a/llama/src/test/java/net/ladenthin/llama/value/ModelMetaTest.java b/llama/src/test/java/net/ladenthin/llama/value/ModelMetaTest.java index 564465126..dd2c654c3 100644 --- a/llama/src/test/java/net/ladenthin/llama/value/ModelMetaTest.java +++ b/llama/src/test/java/net/ladenthin/llama/value/ModelMetaTest.java @@ -5,8 +5,11 @@ package net.ladenthin.llama.value; import static org.hamcrest.MatcherAssert.assertThat; +import static org.hamcrest.Matchers.contains; import static org.hamcrest.Matchers.containsString; +import static org.hamcrest.Matchers.empty; import static org.hamcrest.Matchers.is; +import static org.junit.jupiter.api.Assertions.assertThrows; import com.fasterxml.jackson.databind.ObjectMapper; import net.ladenthin.llama.ClaudeGenerated; @@ -216,4 +219,34 @@ public void testNewGettersDefaultWhenAbsent() throws Exception { assertThat(meta.getEotTokenId(), is(-1)); assertThat(meta.getMetadata("general.architecture"), is("")); } + + @Test + public void testModalitiesOfATextModel() throws Exception { + ModelMeta meta = + parse("{\"input_modalities\":[\"text\",\"image\",\"audio\"]," + "\"output_modalities\":[\"text\"]}"); + + assertThat(meta.getInputModalities(), contains("text", "image", "audio")); + assertThat(meta.getOutputModalities(), contains("text")); + assertThat(meta.isDecisionModel(), is(false)); + } + + @Test + public void testModalitiesOfADecisionModel() throws Exception { + ModelMeta meta = parse("{\"input_modalities\":[\"text\"],\"output_modalities\":[\"decisions\"]}"); + + assertThat(meta.getOutputModalities(), contains(ModelMeta.OUTPUT_MODALITY_DECISIONS)); + assertThat(meta.isDecisionModel(), is(true)); + } + + @Test + public void testModalitiesAbsentAreEmptyAndUnmodifiable() throws Exception { + ModelMeta meta = parse("{\"n_vocab\":100}"); + + assertThat(meta.getInputModalities(), is(empty())); + assertThat(meta.getOutputModalities(), is(empty())); + assertThat(meta.isDecisionModel(), is(false)); + assertThrows( + UnsupportedOperationException.class, + () -> meta.getOutputModalities().add("x")); + } } diff --git a/llama/src/test/java/net/ladenthin/llama/value/RouterModelTest.java b/llama/src/test/java/net/ladenthin/llama/value/RouterModelTest.java index 73ab8a524..2a1abd767 100644 --- a/llama/src/test/java/net/ladenthin/llama/value/RouterModelTest.java +++ b/llama/src/test/java/net/ladenthin/llama/value/RouterModelTest.java @@ -5,10 +5,17 @@ package net.ladenthin.llama.value; import static org.hamcrest.MatcherAssert.assertThat; +import static org.hamcrest.Matchers.contains; +import static org.hamcrest.Matchers.empty; import static org.hamcrest.Matchers.is; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertNotEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collections; +import java.util.List; import net.ladenthin.llama.ClaudeGenerated; import org.junit.jupiter.api.Test; @@ -84,4 +91,57 @@ public void toString_failedShapeIncludesExitCode() { RouterModel failed = new RouterModel("broken", RouterModel.Status.UNLOADED, "unloaded", true, 137); assertThat(failed.toString(), is("broken [unloaded, failed exit=137]")); } + + // ------------------------------------------------------------------------- + // architecture modalities (llama.cpp b11429, #29987) + // ------------------------------------------------------------------------- + + private static RouterModel withModalities(List input, List output) { + return new RouterModel("m", RouterModel.Status.UNLOADED, "unloaded", false, 0, input, output); + } + + @Test + public void legacyConstructorReportsNoModalities() { + RouterModel model = sample(); + assertThat(model.getInputModalities(), is(empty())); + assertThat(model.getOutputModalities(), is(empty())); + assertThat(model.isDecisionModel(), is(false)); + } + + @Test + public void modalitiesRoundTrip() { + RouterModel model = withModalities(Arrays.asList("text", "image"), Collections.singletonList("text")); + assertThat(model.getInputModalities(), contains("text", "image")); + assertThat(model.getOutputModalities(), contains("text")); + assertThat(model.isDecisionModel(), is(false)); + } + + @Test + public void decisionsOutputMarksADecisionModel() { + RouterModel model = withModalities(Collections.singletonList("text"), Arrays.asList("text", "decisions")); + assertThat(model.isDecisionModel(), is(true)); + } + + @Test + public void modalitiesAreACopyAndUnmodifiable() { + List input = new ArrayList<>(Collections.singletonList("text")); + RouterModel model = withModalities(input, Collections.singletonList("decisions")); + input.add("image"); + assertThat(model.getInputModalities(), contains("text")); + assertThrows( + UnsupportedOperationException.class, + () -> model.getInputModalities().add("x")); + assertThrows( + UnsupportedOperationException.class, + () -> model.getOutputModalities().add("x")); + } + + @Test + public void equals_differsPerModality() { + RouterModel base = withModalities(Collections.singletonList("text"), Collections.singletonList("text")); + assertEquals(base, withModalities(Collections.singletonList("text"), Collections.singletonList("text"))); + assertNotEquals(base, withModalities(Arrays.asList("text", "image"), Collections.singletonList("text"))); + assertNotEquals( + base, withModalities(Collections.singletonList("text"), Collections.singletonList("decisions"))); + } }