diff --git a/.github/verify-hip-offload-compressed.py b/.github/verify-hip-offload-compressed.py new file mode 100644 index 00000000..343d04d1 --- /dev/null +++ b/.github/verify-hip-offload-compressed.py @@ -0,0 +1,86 @@ +# SPDX-FileCopyrightText: 2026 Bernard Ladenthin +# +# SPDX-License-Identifier: MIT + +"""Fail when a ROCm/HIP build ships its GPU code uncompressed. + +llama/CMakeLists.txt compiles ggml-hip with clang's --offload-compress, so every embedded +device-code bundle is a compressed "CCOB" bundle. An uncompressed bundle starts with the magic +"__CLANG_OFFLOAD_BUNDLE__"; one of those in the shipped library means the flag was lost and the +library is back to carrying every GPU target's code uncompressed (~1 GB on Windows). + +Usage: verify-hip-offload-compressed.py ... + +Scans every jllama.dll / libjllama.so found, prints its size and the bundle counts, and exits +non-zero if a library has an uncompressed bundle, has no compressed bundle at all, or if no +library was found. Only the standard library, so it runs on any runner. +""" + +import os +import sys + +UNCOMPRESSED = b"__CLANG_OFFLOAD_BUNDLE__" +COMPRESSED = b"CCOB" +NAMES = ("jllama.dll", "libjllama.so") +CHUNK = 64 * 1024 * 1024 + + +def count(path): + """Counts both magics in one streaming pass (the file can be ~1 GB).""" + overlap = len(UNCOMPRESSED) - 1 + uncompressed = compressed = 0 + tail = b"" + with open(path, "rb") as f: + while True: + block = f.read(CHUNK) + if not block: + break + data = tail + block + # a match lying entirely inside `tail` was counted in the previous round + uncompressed += data.count(UNCOMPRESSED) - tail.count(UNCOMPRESSED) + compressed += data.count(COMPRESSED) - tail.count(COMPRESSED) + tail = data[-overlap:] + return uncompressed, compressed + + +def libraries(paths): + for p in paths: + if os.path.isfile(p): + yield p + for root, _, files in os.walk(p): + for name in files: + if name in NAMES: + yield os.path.join(root, name) + + +def main(): + if len(sys.argv) < 2: + print("usage: verify-hip-offload-compressed.py ...", file=sys.stderr) + return 2 + found = failed = 0 + summary = [] + for lib in libraries(sys.argv[1:]): + found += 1 + size = os.path.getsize(lib) + uncompressed, compressed = count(lib) + line = f"{lib}: {size / 1024 / 1024:.0f} MiB, {compressed} compressed / {uncompressed} uncompressed offload bundle(s)" + print(line) + summary.append(line) + if uncompressed: + print(f"::error::{lib} embeds {uncompressed} uncompressed GPU code bundle(s); is --offload-compress still set on ggml-hip?") + failed += 1 + elif not compressed: + print(f"::error::{lib} embeds no compressed GPU code bundle; was it built with GGML_HIP=ON?") + failed += 1 + if not found: + print(f"::error::no {' / '.join(NAMES)} found under {sys.argv[1:]}") + return 2 + step_summary = os.environ.get("GITHUB_STEP_SUMMARY") + if step_summary: + with open(step_summary, "a", encoding="utf-8") as f: + f.write("### ROCm/HIP device code\n\n" + "\n".join(f"- `{s}`" for s in summary) + "\n") + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 9342bbe1..f5ddb12a 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -2164,6 +2164,10 @@ jobs: run: | mvn --no-transfer-progress -f llama/pom.xml compile .github/build.sh "-DGGML_HIP=ON -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGPU_TARGETS=gfx900;gfx906;gfx908;gfx90a;gfx90c;gfx942;gfx950;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" + - name: Verify the GPU code is compressed + # ggml-hip is compiled with --offload-compress (llama/CMakeLists.txt); without it the + # library carries every target's kernels uncompressed. Fails on an uncompressed bundle. + run: python3 .github/verify-hip-offload-compressed.py llama/src/main/resources_linux_rocm - name: Upload artifacts uses: actions/upload-artifact@v7 with: @@ -2242,9 +2246,14 @@ jobs: # Upstream's compiler wiring for TheRock (clang under lib\llvm\bin, not bin\); # -Wno-error=incompatible-pointer-types is upstream's too. Targets: every Windows target # TheRock builds — the Linux list minus the Instinct parts, which ROCm has no Windows - # support for. Same rule for the extras upstream omits as on Linux. + # support for. Same rule as on Linux for the extras upstream omits -- here only + # gfx900/gfx906/gfx90c, since upstream's windows-rocm list already has gfx1153. run: | .github\build.bat -G "Ninja Multi-Config" -DGGML_HIP=ON -DGPU_TARGETS=gfx900;gfx906;gfx90c;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DCMAKE_PREFIX_PATH="%HIP_PATH%" -DHIP_PATH="%HIP_PATH%" -DCMAKE_C_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_CXX_COMPILER="%HIP_PATH%\lib\llvm\bin\clang++.exe" -DCMAKE_HIP_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" -DOS_NAME=Windows -DOS_ARCH=x86_64 + - name: Verify the GPU code is compressed + # Same check as the Linux ROCm job: an uncompressed bundle means ~1 GB jllama.dll again. + shell: pwsh + run: python .github/verify-hip-offload-compressed.py llama/src/main/resources_windows_rocm - name: Upload artifacts uses: actions/upload-artifact@v7 with: diff --git a/CLAUDE.md b/CLAUDE.md index 045c0754..b7cd7885 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,7 +6,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co Java bindings for [llama.cpp](https://github.com/ggerganov/llama.cpp) via JNI, providing a high-level API for LLM inference in Java. The Java layer communicates with a native C++ library through JNI. -Current llama.cpp pinned version: **b11247** +Current llama.cpp pinned version: **b11256** ## Upgrading CUDA Version @@ -348,13 +348,31 @@ its Python wheels (`rocm[libraries,devel]` from `stable.repo.amd.com/rocm/whl-ne llama.cpp's own `ubuntu-rocm` / `windows-rocm` release jobs do, and read the paths back with `rocm-sdk path`. The ROCm version follows upstream's `release.yml` at the pinned `GIT_TAG` — **re-check it on every llama.cpp bump**. The `GPU_TARGETS` lists deliberately go **further than -upstream's**: they are every target TheRock builds for that OS (its `SUPPORTED_GPUS.md`), which adds -gfx900/gfx906/gfx90c/gfx1153 — "build passing" there, not release-ready, and omitted by llama.cpp. +upstream's**: they are every target TheRock builds for that OS (its `SUPPORTED_GPUS.md`). On Linux +that adds gfx900/gfx906/gfx90c/gfx1153 to upstream's list, on Windows only gfx900/gfx906/gfx90c +(upstream's `windows-rocm` list already carries gfx1153, its `ubuntu-rocm` list does not — checked at +b11256). All four are "build passing" in TheRock, not release-ready. Supporting more rather than fewer is the policy, with one limit: an extra stays only while it builds without problems and without local patches; the moment one needs a patch or holds back a newer ROCm/llama.cpp, drop it. The two lists differ **only** by the Instinct parts (gfx908/gfx90a/gfx942/gfx950), which ROCm supports on Linux alone. +**The ROCm GPU code is compressed (`--offload-compress`), and CI enforces it.** Each HIP +translation unit embeds one code object per GPU target, i.e. the whole kernel set (flash attention, +mmq per quant type, …) once per architecture — stored **uncompressed** by default, which made the +Windows `jllama.dll` ~1 GB for its 23 targets (234 MB zipped, so the jar never showed it). That size +also lands on disk: `LlamaLoader` extracts the library to the temp dir on every start, and in the +all-backends fat jar ROCm is tried right after CUDA, i.e. on nearly every machine without an NVIDIA +card. `llama/CMakeLists.txt` therefore adds `--offload-compress` to the `ggml-hip` target only +(`$`: its sources are HIP on Linux and CXX on Windows, where upstream +compiles HIP as C++), so clang stores every bundle zstd-compressed (a `CCOB` bundle) and the HIP +runtime inflates it when the module loads. Upstream llama.cpp does **not** do this; its +`ggml-hip.dll` carries the same uncompressed code (for 20 targets). `.github/verify-hip-offload-compressed.py` runs after the build in +both ROCm jobs, prints the library size and bundle counts (also into the job summary), and fails on +any uncompressed bundle (`__CLANG_OFFLOAD_BUNDLE__`) or on none compressed — a toolchain or upstream +change that drops the flag reds the job instead of quietly shipping the 1 GB library again. The +jar barely shrinks (zip already compressed the code); what shrinks is the extracted library. + Two routing notes mirror existing precedent: **Linux SYCL** ships two precision variants at the *same* arch, so `CMakeLists.txt` routes them to two *distinct* trees by `GGML_SYCL_F16` (fp16 vs fp32). **Windows OpenCL** now holds both `x86_64` (desktop ICD) and `aarch64` (Snapdragon/Adreno) in the one @@ -538,7 +556,7 @@ needs no extra step here, `build-webui` re-reads the tag and rebuilds the matchi ships no UI): ```bash # needs node/npm + network for the asset build; the embed step is plain cmake -P -git clone --depth 1 --branch b11247 https://github.com/ggml-org/llama.cpp /tmp/lc +git clone --depth 1 --branch b11256 https://github.com/ggml-org/llama.cpp /tmp/lc ( cd /tmp/lc/tools/ui && npm ci && npm run build ) mkdir -p webui-generated /tmp/ui-gen cmake -DUI_SOURCE_DIR=/tmp/lc/tools/ui -DUI_BINARY_DIR=/tmp/ui-gen \ @@ -578,7 +596,7 @@ cache lives in **Depot Cache** over sccache's **WebDAV** backend: - `SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}` — a Depot **organization** token, stored as the repo secret **`DEPOT_TOKEN`**. -Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11247`), the +Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11256`), the ~280 upstream object files are byte-identical every run, so a warm cache recompiles only the *changed* files. Depot's cache is **shared across all branches** (unlike GitHub's per-branch `actions/cache`), so every branch builds incrementally; a `b` version bump @@ -1797,7 +1815,7 @@ ctest --test-dir build --output-on-failure -R "ResultsToJson" #### Upstream source location (in CMake build tree) -llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11247`. +llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11256`. **GoogleTest** is a separate `BUILD_TESTING`-only FetchContent (`GIT_TAG v1.18.0`), used solely by the `jllama_test` C++ unit-test binary — not by the shipped library, and not coupled to the diff --git a/README.md b/README.md index 1f3cc9c9..b3afd06b 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ **Build:** ![Java 8+](https://img.shields.io/badge/Java-8%2B-informational) ![Platform](https://img.shields.io/badge/Platform-Linux%20%7C%20macOS%20%7C%20Windows%20%7C%20Android-lightgrey) -[![llama.cpp b11247](https://img.shields.io/badge/llama.cpp-%23b11247-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11247) +[![llama.cpp b11256](https://img.shields.io/badge/llama.cpp-%23b11256-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11256) [![JPMS](https://img.shields.io/badge/JPMS-modular%20JAR-25A162)](https://openjdk.org/projects/jigsaw/) ![JUnit](https://img.shields.io/badge/tested%20with-JUnit6-25A162) [![JSpecify](https://img.shields.io/badge/JSpecify-1.0.0%20%40NullMarked-25A162)](https://jspecify.dev) diff --git a/docs/history/llama-cpp-breaking-changes.md b/docs/history/llama-cpp-breaking-changes.md index 05b00006..eca0d356 100644 --- a/docs/history/llama-cpp-breaking-changes.md +++ b/docs/history/llama-cpp-breaking-changes.md @@ -766,3 +766,5 @@ Used during `llama.cpp` version bumps: when upgrading, scan this file from the r | b11236–b11237 | patches + upstream verification | **All nine patches apply unchanged.** No patch-target file is in the range. | | b11237–b11247 | Ten commits, 29 files, ~465/253 lines. **#29595** (`common/common.{h,cpp}`): `fs_get_cache_directory()` / `fs_get_cache_file()` now return `std::filesystem::path` instead of `std::string` (and the directory no longer carries a trailing separator), and `fs_create_directory_with_parents()` is removed — none of the three is referenced by project code, and upstream's only in-range caller (`common/arg.cpp` `get_default_local_path`) was adapted in the same commit. **#29556** (`tools/server/server-common.{h,cpp}`, `server-context.cpp`): `/v1/embeddings` accepts OAI typed content (`{"content":[{"type":"text"\|"image_url"\|"input_audio"\|"input_video",…}]}`), which exports the previously `static` `tokenize_input_subprompt` and adds `tokenize_oai_content_array`; the media-loading loop of `oaicompat_chat_params_parse` moved into a shared helper verbatim (pure refactor). The same PR stops embedding/rerank tasks from reusing a cached prompt prefix (`is_stateless_task`) — a slot-internal change behind an unchanged request contract. jllama's own `handleEmbeddings` path calls `tokenize_input_prompts`, whose signature and shapes are unchanged, so the new typed-content input is reachable only through `NativeServer` for now. Server contract check (request-field set + bounds, response keys in `server-schema.cpp` / `server-task.cpp`): **no change**. Rest: #29615 Muse Glimmer `response_format` with `--jinja` (chat parser), #29607 `mtmd` GCC 15 warning fix, #29567 `ggml_pad_ext` in `dflash`/`mtmd` models, two Vulkan-internal changes (#29597, #29280), `rpc-server.cpp` tool refactor (not compiled here), test/CI-only commits. Upstream #29273 moved its SYCL release builds to oneAPI **2026.1** (Windows: `sycl9.dll`, MKL `.3` DLLs); this project's Windows SYCL job still pins the 2025.3.3 installer, which keeps building — the jar bundles no oneAPI runtime, so the choice only fixes which runtime the consumer must install. Version-only from this project's side. | | b11237–b11247 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11247 with `git apply --check`. Two patch-target files are in the range — `common/arg.cpp` (`0001`, `0015`; #29595 touched only `get_default_local_path`) and `tools/server/server-context.cpp` (`0002`, `0003`; #29556 touched the prompt-cache gate and `handle_embeddings_impl`) — both away from the patched hunks. Drop-checks against pristine b11247: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override (`0001` needed); no `split_sum == 0` guard (`0012` needed); no `common_log_set_callback` (`0014` needed); no `stop_server` in `ggml-rpc.cpp` (`0015` needed). No new standalone `main()` calls `common_params_parse` directly. | +| b11247–b11256 | Nine commits, 20 files, ~116/117 lines. **#29632** (`include/llama.h` comment only; examples/tools): `simple`, `simple-chat`, `test-fusion` switch from `ggml_backend_load_all()` to `llama_backend_init()`, and `llama-cli` now calls `llama_backend_init()` + `llama_numa_init()` itself — `jllama.cpp`, `tts_engine.cpp` and `train_engine.cpp` already call both, so nothing to follow. **#29642** (`common/common.{h,cpp}`): new `fs_write_atomic()`, used by `download.cpp` / `hf-cache.cpp`; not referenced by project code. **#29634** (`ggml-backend.cpp`): the scheduler collects every graph input into `graph_inputs` in a new pass 6 (also inputs no node consumes), so switching batch types no longer reallocates the graph — internal. **#29598** (`gguf.cpp`): duplicate-key / duplicate-tensor-name checks use hash sets instead of O(n²) loops (faster model loading, same errors, slightly reworded log). `server-context.cpp` only fixes an error-message typo (`does not support logits computation`); `server-schema.cpp` / `server-task.cpp` are untouched, so the request/response contract checks have nothing to compare. Rest: Metal FWHT perf (#29602), zDNN 0-row fix (#29636), MUSA CI/docker (#29624), ROCm CI log cleanup (#28940), a server test regex (#29648). No `release.yml` change, so the CUDA/ROCm/OpenVINO pins stay. Version-only from this project's side. | +| b11247–b11256 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11256 with `git apply` and by a fresh configure. Two patch-target files are in the range — `tools/cli/cli.cpp` (`0001`; #29632 added two lines after the argument parse, away from the flipped call) and `tools/server/server-context.cpp` (`0002`, `0003`; only the error-message typo). Drop-checks against pristine b11256: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override and `common_params_parse_main` is absent (`0001` needed); `load_progress_callback` still assigned unconditionally (`0002` needed); no `split_sum == 0` guard (`0012` needed); no `common_log_set_callback` (`0014` needed); no `stop_server` in `ggml-rpc.cpp` (`0015` needed). | diff --git a/llama/CMakeLists.txt b/llama/CMakeLists.txt index 17b87bc5..17fb2b0a 100644 --- a/llama/CMakeLists.txt +++ b/llama/CMakeLists.txt @@ -188,7 +188,7 @@ set(LLAMA_BUILD_APP OFF CACHE BOOL "" FORCE) FetchContent_Declare( llama.cpp GIT_REPOSITORY https://github.com/ggerganov/llama.cpp.git - GIT_TAG b11247 + GIT_TAG b11256 PATCH_COMMAND ${CMAKE_COMMAND} -DPATCH_DIR=${CMAKE_CURRENT_SOURCE_DIR}/patches -DLLAMA_SRC= @@ -196,6 +196,19 @@ FetchContent_Declare( ) FetchContent_MakeAvailable(llama.cpp) +# ROCm/HIP: compress the embedded GPU code. Without it every HIP translation unit carries one +# UNcompressed code object per GPU target, so the shipped library holds the complete kernel set +# (flash attention, mmq per quant type, ...) once per architecture: ~1 GB for the 23 Windows +# targets, which LlamaLoader then extracts to the temp dir on every start. --offload-compress +# makes clang store each bundle zstd-compressed (a "CCOB" bundle); the HIP runtime inflates it +# when the module is loaded. Scoped to the ggml-hip target (the only one with device code): +# its sources are LANGUAGE HIP on Linux and CXX on Windows (upstream compiles HIP as C++ there). +# CI (.github/verify-hip-offload-compressed.py) fails the ROCm jobs if an uncompressed bundle +# ever reappears. +if(TARGET ggml-hip) + target_compile_options(ggml-hip PRIVATE $<$:--offload-compress>) +endif() + # b8831 added ggml_graph_next_uid() which calls _InterlockedIncrement64 via # on x86. The intrinsic only exists on x64; provide the # implementation in a compat TU so the linker resolves __InterlockedIncrement64. diff --git a/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java b/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java index 9412e85c..55148132 100644 --- a/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java +++ b/llama/src/main/java/net/ladenthin/llama/value/LlamaCppVersion.java @@ -10,13 +10,13 @@ * library was compiled against, exposed as a compile-time constant so callers can render a badge or * emit a startup log line without loading the native library. * - *

{@link #LLAMA_CPP_VERSION} is a pure-Java string ({@code "b11247"}) that mirrors the + *

{@link #LLAMA_CPP_VERSION} is a pure-Java string ({@code "b11256"}) that mirrors the * {@code GIT_TAG} in {@code llama/CMakeLists.txt}. It is available even when {@code libjllama} is * absent (pure-Java checkout, before {@code System.load}), which is what makes it suitable for a * lightweight version badge in Android or other UIs.

* *

For the authoritative value that is baked into the native binary — the build number - * plus the resolved upstream commit, e.g. {@code "b11247-"} — call + * plus the resolved upstream commit, e.g. {@code "b11256-"} — call * {@link net.ladenthin.llama.LlamaModel#getLlamaCppBuildInfo()} instead; that reads llama.cpp's own * {@code build-info} through JNI and therefore cannot drift from the compiled library (but requires * the native library to be loaded).

@@ -24,14 +24,14 @@ public final class LlamaCppVersion { /** - * The pinned llama.cpp release tag this library was built against, e.g. {@code "b11247"}. + * The pinned llama.cpp release tag this library was built against, e.g. {@code "b11256"}. * *

Kept in lockstep with {@code GIT_TAG} in {@code llama/CMakeLists.txt} — see the * "Upgrading/Downgrading llama.cpp Version" checklist in {@code CLAUDE.md}. This is the * compile-time pin; use {@link net.ladenthin.llama.LlamaModel#getLlamaCppBuildInfo()} for the * value actually linked into the native binary.

*/ - public static final String LLAMA_CPP_VERSION = "b11247"; + public static final String LLAMA_CPP_VERSION = "b11256"; // Constants holder — not instantiable. private LlamaCppVersion() {}