Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
86 changes: 86 additions & 0 deletions .github/verify-hip-offload-compressed.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
#
# SPDX-License-Identifier: MIT

"""Fail when a ROCm/HIP build ships its GPU code uncompressed.

llama/CMakeLists.txt compiles ggml-hip with clang's --offload-compress, so every embedded
device-code bundle is a compressed "CCOB" bundle. An uncompressed bundle starts with the magic
"__CLANG_OFFLOAD_BUNDLE__"; one of those in the shipped library means the flag was lost and the
library is back to carrying every GPU target's code uncompressed (~1 GB on Windows).

Usage: verify-hip-offload-compressed.py <dir-or-file>...

Scans every jllama.dll / libjllama.so found, prints its size and the bundle counts, and exits
non-zero if a library has an uncompressed bundle, has no compressed bundle at all, or if no
library was found. Only the standard library, so it runs on any runner.
"""

import os
import sys

UNCOMPRESSED = b"__CLANG_OFFLOAD_BUNDLE__"
COMPRESSED = b"CCOB"
NAMES = ("jllama.dll", "libjllama.so")
CHUNK = 64 * 1024 * 1024


def count(path):
"""Counts both magics in one streaming pass (the file can be ~1 GB)."""
overlap = len(UNCOMPRESSED) - 1
uncompressed = compressed = 0
tail = b""
with open(path, "rb") as f:
while True:
block = f.read(CHUNK)
if not block:
break
data = tail + block
# a match lying entirely inside `tail` was counted in the previous round
uncompressed += data.count(UNCOMPRESSED) - tail.count(UNCOMPRESSED)
compressed += data.count(COMPRESSED) - tail.count(COMPRESSED)
tail = data[-overlap:]
return uncompressed, compressed


def libraries(paths):
for p in paths:
if os.path.isfile(p):
yield p
for root, _, files in os.walk(p):
for name in files:
if name in NAMES:
yield os.path.join(root, name)


def main():
if len(sys.argv) < 2:
print("usage: verify-hip-offload-compressed.py <dir-or-file>...", file=sys.stderr)
return 2
found = failed = 0
summary = []
for lib in libraries(sys.argv[1:]):
found += 1
size = os.path.getsize(lib)
uncompressed, compressed = count(lib)
line = f"{lib}: {size / 1024 / 1024:.0f} MiB, {compressed} compressed / {uncompressed} uncompressed offload bundle(s)"
print(line)
summary.append(line)
if uncompressed:
print(f"::error::{lib} embeds {uncompressed} uncompressed GPU code bundle(s); is --offload-compress still set on ggml-hip?")
failed += 1
elif not compressed:
print(f"::error::{lib} embeds no compressed GPU code bundle; was it built with GGML_HIP=ON?")
failed += 1
if not found:
print(f"::error::no {' / '.join(NAMES)} found under {sys.argv[1:]}")
return 2
step_summary = os.environ.get("GITHUB_STEP_SUMMARY")
if step_summary:
with open(step_summary, "a", encoding="utf-8") as f:
f.write("### ROCm/HIP device code\n\n" + "\n".join(f"- `{s}`" for s in summary) + "\n")
return 1 if failed else 0


if __name__ == "__main__":
sys.exit(main())
11 changes: 10 additions & 1 deletion .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2164,6 +2164,10 @@ jobs:
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_HIP=ON -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGPU_TARGETS=gfx900;gfx906;gfx908;gfx90a;gfx90c;gfx942;gfx950;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Verify the GPU code is compressed
# ggml-hip is compiled with --offload-compress (llama/CMakeLists.txt); without it the
# library carries every target's kernels uncompressed. Fails on an uncompressed bundle.
run: python3 .github/verify-hip-offload-compressed.py llama/src/main/resources_linux_rocm
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
Expand Down Expand Up @@ -2242,9 +2246,14 @@ jobs:
# Upstream's compiler wiring for TheRock (clang under lib\llvm\bin, not bin\);
# -Wno-error=incompatible-pointer-types is upstream's too. Targets: every Windows target
# TheRock builds — the Linux list minus the Instinct parts, which ROCm has no Windows
# support for. Same rule for the extras upstream omits as on Linux.
# support for. Same rule as on Linux for the extras upstream omits -- here only
# gfx900/gfx906/gfx90c, since upstream's windows-rocm list already has gfx1153.
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_HIP=ON -DGPU_TARGETS=gfx900;gfx906;gfx90c;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DCMAKE_PREFIX_PATH="%HIP_PATH%" -DHIP_PATH="%HIP_PATH%" -DCMAKE_C_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_CXX_COMPILER="%HIP_PATH%\lib\llvm\bin\clang++.exe" -DCMAKE_HIP_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Verify the GPU code is compressed
# Same check as the Linux ROCm job: an uncompressed bundle means ~1 GB jllama.dll again.
shell: pwsh
run: python .github/verify-hip-offload-compressed.py llama/src/main/resources_windows_rocm
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
Expand Down
30 changes: 24 additions & 6 deletions CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co

Java bindings for [llama.cpp](https://github.com/ggerganov/llama.cpp) via JNI, providing a high-level API for LLM inference in Java. The Java layer communicates with a native C++ library through JNI.

Current llama.cpp pinned version: **b11247**
Current llama.cpp pinned version: **b11256**

## Upgrading CUDA Version

Expand Down Expand Up @@ -348,13 +348,31 @@ its Python wheels (`rocm[libraries,devel]` from `stable.repo.amd.com/rocm/whl-ne
llama.cpp's own `ubuntu-rocm` / `windows-rocm` release jobs do, and read the paths back with
`rocm-sdk path`. The ROCm version follows upstream's `release.yml` at the pinned `GIT_TAG` —
**re-check it on every llama.cpp bump**. The `GPU_TARGETS` lists deliberately go **further than
upstream's**: they are every target TheRock builds for that OS (its `SUPPORTED_GPUS.md`), which adds
gfx900/gfx906/gfx90c/gfx1153 — "build passing" there, not release-ready, and omitted by llama.cpp.
upstream's**: they are every target TheRock builds for that OS (its `SUPPORTED_GPUS.md`). On Linux
that adds gfx900/gfx906/gfx90c/gfx1153 to upstream's list, on Windows only gfx900/gfx906/gfx90c
(upstream's `windows-rocm` list already carries gfx1153, its `ubuntu-rocm` list does not — checked at
b11256). All four are "build passing" in TheRock, not release-ready.
Supporting more rather than fewer is the policy, with one limit: an extra stays only while it builds
without problems and without local patches; the moment one needs a patch or holds back a newer
ROCm/llama.cpp, drop it. The two lists differ **only** by the Instinct parts
(gfx908/gfx90a/gfx942/gfx950), which ROCm supports on Linux alone.

**The ROCm GPU code is compressed (`--offload-compress`), and CI enforces it.** Each HIP
translation unit embeds one code object per GPU target, i.e. the whole kernel set (flash attention,
mmq per quant type, …) once per architecture — stored **uncompressed** by default, which made the
Windows `jllama.dll` ~1 GB for its 23 targets (234 MB zipped, so the jar never showed it). That size
also lands on disk: `LlamaLoader` extracts the library to the temp dir on every start, and in the
all-backends fat jar ROCm is tried right after CUDA, i.e. on nearly every machine without an NVIDIA
card. `llama/CMakeLists.txt` therefore adds `--offload-compress` to the `ggml-hip` target only
(`$<COMPILE_LANGUAGE:HIP,CXX>`: its sources are HIP on Linux and CXX on Windows, where upstream
compiles HIP as C++), so clang stores every bundle zstd-compressed (a `CCOB` bundle) and the HIP
runtime inflates it when the module loads. Upstream llama.cpp does **not** do this; its
`ggml-hip.dll` carries the same uncompressed code (for 20 targets). `.github/verify-hip-offload-compressed.py` runs after the build in
both ROCm jobs, prints the library size and bundle counts (also into the job summary), and fails on
any uncompressed bundle (`__CLANG_OFFLOAD_BUNDLE__`) or on none compressed — a toolchain or upstream
change that drops the flag reds the job instead of quietly shipping the 1 GB library again. The
jar barely shrinks (zip already compressed the code); what shrinks is the extracted library.

Two routing notes mirror existing precedent: **Linux SYCL** ships two precision variants at the *same*
arch, so `CMakeLists.txt` routes them to two *distinct* trees by `GGML_SYCL_F16` (fp16 vs fp32).
**Windows OpenCL** now holds both `x86_64` (desktop ICD) and `aarch64` (Snapdragon/Adreno) in the one
Expand Down Expand Up @@ -538,7 +556,7 @@ needs no extra step here, `build-webui` re-reads the tag and rebuilds the matchi
ships no UI):
```bash
# needs node/npm + network for the asset build; the embed step is plain cmake -P
git clone --depth 1 --branch b11247 https://github.com/ggml-org/llama.cpp /tmp/lc
git clone --depth 1 --branch b11256 https://github.com/ggml-org/llama.cpp /tmp/lc
( cd /tmp/lc/tools/ui && npm ci && npm run build )
mkdir -p webui-generated /tmp/ui-gen
cmake -DUI_SOURCE_DIR=/tmp/lc/tools/ui -DUI_BINARY_DIR=/tmp/ui-gen \
Expand Down Expand Up @@ -578,7 +596,7 @@ cache lives in **Depot Cache** over sccache's **WebDAV** backend:
- `SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}` — a Depot **organization** token, stored
as the repo secret **`DEPOT_TOKEN`**.

Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11247`), the
Because `sccache` is **content-addressed** and llama.cpp is pinned (`GIT_TAG b11256`), the
~280 upstream object files are byte-identical every run, so a warm cache recompiles only the
*changed* files. Depot's cache is **shared across all branches** (unlike GitHub's
per-branch `actions/cache`), so every branch builds incrementally; a `b<nnnn>` version bump
Expand Down Expand Up @@ -1797,7 +1815,7 @@ ctest --test-dir build --output-on-failure -R "ResultsToJson"

#### Upstream source location (in CMake build tree)

llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11247`.
llama.cpp is fetched via CMake FetchContent, pinned to `GIT_TAG b11256`.

**GoogleTest** is a separate `BUILD_TESTING`-only FetchContent (`GIT_TAG v1.18.0`), used solely
by the `jllama_test` C++ unit-test binary — not by the shipped library, and not coupled to the
Expand Down
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@
**Build:**
![Java 8+](https://img.shields.io/badge/Java-8%2B-informational)
![Platform](https://img.shields.io/badge/Platform-Linux%20%7C%20macOS%20%7C%20Windows%20%7C%20Android-lightgrey)
[![llama.cpp b11247](https://img.shields.io/badge/llama.cpp-%23b11247-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11247)
[![llama.cpp b11256](https://img.shields.io/badge/llama.cpp-%23b11256-informational)](https://github.com/ggml-org/llama.cpp/releases/tag/b11256)
[![JPMS](https://img.shields.io/badge/JPMS-modular%20JAR-25A162)](https://openjdk.org/projects/jigsaw/)
![JUnit](https://img.shields.io/badge/tested%20with-JUnit6-25A162)
[![JSpecify](https://img.shields.io/badge/JSpecify-1.0.0%20%40NullMarked-25A162)](https://jspecify.dev)
Expand Down
2 changes: 2 additions & 0 deletions docs/history/llama-cpp-breaking-changes.md
Original file line number Diff line number Diff line change
Expand Up @@ -766,3 +766,5 @@ Used during `llama.cpp` version bumps: when upgrading, scan this file from the r
| b11236–b11237 | patches + upstream verification | **All nine patches apply unchanged.** No patch-target file is in the range. |
| b11237–b11247 | Ten commits, 29 files, ~465/253 lines. **#29595** (`common/common.{h,cpp}`): `fs_get_cache_directory()` / `fs_get_cache_file()` now return `std::filesystem::path` instead of `std::string` (and the directory no longer carries a trailing separator), and `fs_create_directory_with_parents()` is removed — none of the three is referenced by project code, and upstream's only in-range caller (`common/arg.cpp` `get_default_local_path`) was adapted in the same commit. **#29556** (`tools/server/server-common.{h,cpp}`, `server-context.cpp`): `/v1/embeddings` accepts OAI typed content (`{"content":[{"type":"text"\|"image_url"\|"input_audio"\|"input_video",…}]}`), which exports the previously `static` `tokenize_input_subprompt` and adds `tokenize_oai_content_array`; the media-loading loop of `oaicompat_chat_params_parse` moved into a shared helper verbatim (pure refactor). The same PR stops embedding/rerank tasks from reusing a cached prompt prefix (`is_stateless_task`) — a slot-internal change behind an unchanged request contract. jllama's own `handleEmbeddings` path calls `tokenize_input_prompts`, whose signature and shapes are unchanged, so the new typed-content input is reachable only through `NativeServer` for now. Server contract check (request-field set + bounds, response keys in `server-schema.cpp` / `server-task.cpp`): **no change**. Rest: #29615 Muse Glimmer `response_format` with `--jinja` (chat parser), #29607 `mtmd` GCC 15 warning fix, #29567 `ggml_pad_ext` in `dflash`/`mtmd` models, two Vulkan-internal changes (#29597, #29280), `rpc-server.cpp` tool refactor (not compiled here), test/CI-only commits. Upstream #29273 moved its SYCL release builds to oneAPI **2026.1** (Windows: `sycl9.dll`, MKL `.3` DLLs); this project's Windows SYCL job still pins the 2025.3.3 installer, which keeps building — the jar bundles no oneAPI runtime, so the choice only fixes which runtime the consumer must install. Version-only from this project's side. |
| b11237–b11247 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11247 with `git apply --check`. Two patch-target files are in the range — `common/arg.cpp` (`0001`, `0015`; #29595 touched only `get_default_local_path`) and `tools/server/server-context.cpp` (`0002`, `0003`; #29556 touched the prompt-cache gate and `handle_embeddings_impl`) — both away from the patched hunks. Drop-checks against pristine b11247: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override (`0001` needed); no `split_sum == 0` guard (`0012` needed); no `common_log_set_callback` (`0014` needed); no `stop_server` in `ggml-rpc.cpp` (`0015` needed). No new standalone `main()` calls `common_params_parse` directly. |
| b11247–b11256 | Nine commits, 20 files, ~116/117 lines. **#29632** (`include/llama.h` comment only; examples/tools): `simple`, `simple-chat`, `test-fusion` switch from `ggml_backend_load_all()` to `llama_backend_init()`, and `llama-cli` now calls `llama_backend_init()` + `llama_numa_init()` itself — `jllama.cpp`, `tts_engine.cpp` and `train_engine.cpp` already call both, so nothing to follow. **#29642** (`common/common.{h,cpp}`): new `fs_write_atomic()`, used by `download.cpp` / `hf-cache.cpp`; not referenced by project code. **#29634** (`ggml-backend.cpp`): the scheduler collects every graph input into `graph_inputs` in a new pass 6 (also inputs no node consumes), so switching batch types no longer reallocates the graph — internal. **#29598** (`gguf.cpp`): duplicate-key / duplicate-tensor-name checks use hash sets instead of O(n²) loops (faster model loading, same errors, slightly reworded log). `server-context.cpp` only fixes an error-message typo (`does not support logits computation`); `server-schema.cpp` / `server-task.cpp` are untouched, so the request/response contract checks have nothing to compare. Rest: Metal FWHT perf (#29602), zDNN 0-row fix (#29636), MUSA CI/docker (#29624), ROCm CI log cleanup (#28940), a server test regex (#29648). No `release.yml` change, so the CUDA/ROCm/OpenVINO pins stay. Version-only from this project's side. |
| b11247–b11256 | patches + upstream verification | **All nine patches apply unchanged**, verified in order against pristine b11256 with `git apply` and by a fresh configure. Two patch-target files are in the range — `tools/cli/cli.cpp` (`0001`; #29632 added two lines after the argument parse, away from the flipped call) and `tools/server/server-context.cpp` (`0002`, `0003`; only the error-message typo). Drop-checks against pristine b11256: `common_params_parse` still carries the `#ifdef _WIN32` `argv = utf8.ptrs.data()` override and `common_params_parse_main` is absent (`0001` needed); `load_progress_callback` still assigned unconditionally (`0002` needed); no `split_sum == 0` guard (`0012` needed); no `common_log_set_callback` (`0014` needed); no `stop_server` in `ggml-rpc.cpp` (`0015` needed). |
15 changes: 14 additions & 1 deletion llama/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -188,14 +188,27 @@ set(LLAMA_BUILD_APP OFF CACHE BOOL "" FORCE)
FetchContent_Declare(
llama.cpp
GIT_REPOSITORY https://github.com/ggerganov/llama.cpp.git
GIT_TAG b11247
GIT_TAG b11256
PATCH_COMMAND ${CMAKE_COMMAND}
-DPATCH_DIR=${CMAKE_CURRENT_SOURCE_DIR}/patches
-DLLAMA_SRC=<SOURCE_DIR>
-P ${CMAKE_CURRENT_SOURCE_DIR}/cmake/apply-llama-patches.cmake
)
FetchContent_MakeAvailable(llama.cpp)

# ROCm/HIP: compress the embedded GPU code. Without it every HIP translation unit carries one
# UNcompressed code object per GPU target, so the shipped library holds the complete kernel set
# (flash attention, mmq per quant type, ...) once per architecture: ~1 GB for the 23 Windows
# targets, which LlamaLoader then extracts to the temp dir on every start. --offload-compress
# makes clang store each bundle zstd-compressed (a "CCOB" bundle); the HIP runtime inflates it
# when the module is loaded. Scoped to the ggml-hip target (the only one with device code):
# its sources are LANGUAGE HIP on Linux and CXX on Windows (upstream compiles HIP as C++ there).
# CI (.github/verify-hip-offload-compressed.py) fails the ROCm jobs if an uncompressed bundle
# ever reappears.
if(TARGET ggml-hip)
target_compile_options(ggml-hip PRIVATE $<$<COMPILE_LANGUAGE:HIP,CXX>:--offload-compress>)
endif()

# b8831 added ggml_graph_next_uid() which calls _InterlockedIncrement64 via
# <intrin.h> on x86. The intrinsic only exists on x64; provide the
# implementation in a compat TU so the linker resolves __InterlockedIncrement64.
Expand Down
Loading
Loading