Skip to content

llama.cpp b11320, shared plugin versions in the parent, agent on Maven Central #1084

llama.cpp b11320, shared plugin versions in the parent, agent on Maven Central

llama.cpp b11320, shared plugin versions in the parent, agent on Maven Central #1084

Workflow file for this run

# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
#
# SPDX-License-Identifier: MIT OR Apache-2.0
name: Publish
on:
push:
branches: [ main ]
tags: ['v*']
pull_request:
workflow_dispatch:
inputs:
publish_to_central:
description: "Deploy to Maven Central (snapshot if -SNAPSHOT, release if a vX.Y.Z tag)"
type: boolean
default: false
use_cache:
description: "Use the shared sccache/Depot compiler cache (faster incremental builds)"
type: boolean
default: true
env:
# The JDK every job builds with is .java-version (setup-java's java-version-file), so the
# reusable workflows read the same one without an input.
# The Gradle builds (Android AAR, the consumer fixture, the llmservice app, the signing self-test).
# AGP 9.4.x needs Gradle >= 9.6.0 (android-llmservice/CLAUDE.md). Not seen by Dependabot: bump by hand,
# together with the literal in verify-signing-key-gradle (a job kept identical in all four sibling
# repositories, so it names its version itself).
GRADLE_VERSION: '9.8.0'
# The sccache/Depot compiler cache (CLAUDE.md, "CI build cache"). Inert without the token, which
# stays per build job (SCCACHE_WEBDAV_TOKEN): at workflow level it would reach every job,
# including those running third-party actions.
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
# The CI model set is .github/models.csv (URLs, and the model cache key is its hash). The Java
# tests default to exactly that set (TestConstants; TestConstantsTest asserts both directions),
# so the test jobs pass no model properties. The names below are only for the consumers outside
# the llama module's tests -- the smoke scripts, the Android emulator jobs and the langchain4j /
# agent integration jobs; check-natives.py fails when one is not a filename of models.csv.
RERANKING_MODEL_NAME: "jina-reranker-v1-tiny-en-Q4_0.gguf"
DRAFT_MODEL_NAME: "AMD-Llama-135m-code.Q2_K.gguf"
REASONING_MODEL_NAME: "Qwen3-0.6B-Q4_K_M.gguf"
TOOL_MODEL_NAME: "Qwen2.5-1.5B-Instruct-Q4_K_M.gguf"
NOMIC_EMBED_MODEL_NAME: "nomic-embed-text-v1.5.f16.gguf"
# Supersede an in-flight run when a PR branch is pushed again.
#
# Without this every push starts a full parallel pipeline and the older ones keep
# draining -- four were live at once during one session, which makes "what is CI
# saying right now" genuinely ambiguous and wastes a lot of runner time on results
# nobody will read.
#
# cancel-in-progress is deliberately scoped to pull_request ONLY. A push to main or
# to a v* tag is a release path: cancelling one midway could leave a partially
# published set of artifacts.
#
# cancel-in-progress: false is NOT sufficient on its own to protect a release run.
# GitHub cancels a *pending* run whenever a newer run joins the same group behind an
# in-progress one -- that rule is independent of cancel-in-progress. So with a plain
# `workflow-ref` group, a queued `publish_to_central` dispatch on main could be
# silently dropped by a later push to main, both sharing `Publish-refs/heads/main`.
# Giving every non-PR run its own group (via the unique run_id) means such a run is
# never queued behind a sibling and therefore can never be cancelled, while PR runs
# still share a group per ref and supersede each other as intended.
#
# One-time effect when this expression changes: GitHub reads `concurrency` from the
# workflow file at each run's own ref, so a run started before the change sits in the
# old group and a run started after it sits in the new one. They are different groups,
# so the new push does NOT supersede the in-flight old run -- exactly once, on the
# commit that lands this. It self-heals from the next push on. Expect the same overlap
# when porting this to a sibling repo; it is not a sign the expression is wrong.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name == 'pull_request' && 'pr' || github.run_id }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
jobs:
# ---------------------------------------------------------------------------
# Start gate — single cancellable abort window before the pipeline starts.
# The wait duration lives in the `startgate` GitHub Environment (Settings →
# Environments → startgate → Wait timer).
# ---------------------------------------------------------------------------
startgate:
name: Start gate (abort window)
runs-on: ubuntu-latest
environment: startgate
steps:
- run: echo "Start gate elapsed — proceeding with pipeline."
# ---------------------------------------------------------------------------
# Files kept byte-identical with the sibling repositories (java-llama.cpp,
# BitcoinAddressFinder, srcmorph, streambuffer), listed in .github/shared-files.sha256.
# A copy changed here alone fails; one the siblings list with another hash warns.
# Also runs the tests of the shared build checks and the release-gate check. This
# job is itself identical in all four repositories.
# ---------------------------------------------------------------------------
shared-files:
name: Shared files + build checks
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
with:
persist-credentials: false
- name: Shared files match .github/shared-files.sha256 (siblings compared, warnings only)
run: python3 .github/check-shared-files.py
- name: Build-check tests
run: python3 -m unittest discover -s .github/buildcheck/tests -t .github
- name: Release gate (every job gates both publish jobs, or release-gate-exemptions.txt says why)
run: python3 .github/check-release-gate.py
- name: Run scripts parse (bash -n over every bash run script of the workflows and actions)
run: python3 .github/check-run-scripts.py
- name: Maven versions (warns where a dependency or plugin differs from a sibling repository)
run: python3 .github/check-versions.py
# ---------------------------------------------------------------------------
# GPG signing-key preflight (standalone, no `needs:` — runs in parallel at the
# very start on every trigger). Reproduces what maven-gpg-plugin does at deploy
# time so a bad/expired key or wrong passphrase is caught in ~20s instead of
# failing the publish stage. Declares `environment: maven-central` so it reads
# the SAME GPG_PRIVATE_KEY / GPG_PASSPHRASE secret the publish jobs use.
#
# It is EXPECTED to go RED on refs where the secret is not delivered — fork PRs
# and other contributors' branches (secrets are withheld there). That red is
# the intended signal: "this ref cannot sign a release", not a regression.
#
# SECURITY: this job NEVER prints secret material. It imports the key into an
# ephemeral keyring, prints only PUBLIC key metadata (key id, fingerprint,
# owner UID, algorithm, created/expiry — all of which live on public
# keyservers), and validates the passphrase by producing + verifying a
# throwaway signature. The passphrase is passed on fd 3 (never argv, never a
# log line), `set -x` is deliberately never enabled, and the passphrase is
# additionally `::add-mask::`ed.
# ---------------------------------------------------------------------------
verify-signing-key:
name: Verify GPG signing key (no secrets printed)
runs-on: ubuntu-latest
environment: maven-central
steps:
- uses: actions/checkout@v7
with:
persist-credentials: false
- name: Import key + run sign/verify self-test (prints only PUBLIC metadata)
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# Shared, byte-identical in all four repos; the security notes are in the script.
run: bash .github/verify-signing-key.sh
# ---------------------------------------------------------------------------
# GPG signing-key preflight — GRADLE / BouncyCastle path.
# Companion to the `verify-signing-key` (gpg) job above: that one mirrors
# maven-gpg-plugin (how the Maven artifacts are signed); this one drives
# Gradle's `signing` plugin + `useInMemoryPgpKeys` (BouncyCastle) — the path any
# Gradle-based publish (e.g. an Android AAR) uses to sign. BouncyCastle is a
# STRICTER parser of the armored key than gpg, so it catches key/format problems
# gpg tolerates (e.g. the primary-vs-signing-subkey null-PGPPrivateKey issue).
# It signs a throwaway project (.github/signing-selftest/) — no repo build is
# involved — so this job is IDENTICAL across the sibling repos and validates the
# release key via the Gradle path even in repos that do not publish via Gradle
# yet ("prepared for Gradle"). Standalone (no `needs:`), parallel at pipeline
# start, `environment: maven-central` so it reads the same secret the publish
# uses. Red-by-design where the secret is not delivered (see the gpg job's note).
#
# SECURITY: prints no secret material. Key/passphrase reach Gradle only via env
# (read by System.getenv at runtime); `set -x` is never enabled; the passphrase
# is `::add-mask::`ed; Gradle runs with `--stacktrace` only. Only the produced
# `.asc` (exit code) is asserted. Uses Gradle 9.6.1.
# ---------------------------------------------------------------------------
verify-signing-key-gradle:
name: Verify GPG signing key — Gradle/BouncyCastle path (no secrets printed)
runs-on: ubuntu-latest
environment: maven-central
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: "9.8.0"
- name: Sign a throwaway artifact via useInMemoryPgpKeys (BouncyCastle)
shell: bash
env:
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
run: |
set -euo pipefail # NOTE: deliberately NO `set -x` — it would echo the passphrase.
if [ -z "${MAVEN_GPG_PRIVATE_KEY:-}" ]; then
echo "::error::MAVEN_GPG_PRIVATE_KEY is empty for this run. The maven-central environment did not deliver the secret to this ref (fork PR / other branch). Nothing to verify."
exit 1
fi
if [ -n "${MAVEN_GPG_PASSPHRASE:-}" ]; then echo "::add-mask::${MAVEN_GPG_PASSPHRASE}"; fi
PROJ=".github/signing-selftest"
echo "== Sign a throwaway artifact through Gradle's useInMemoryPgpKeys (BouncyCastle) =="
gradle --no-daemon -p "$PROJ" signMakeArtifact --stacktrace
ASC="$PROJ/build/signing-selftest.zip.asc"
if [ -f "$ASC" ]; then
echo " Detached signature produced: $(wc -c < "$ASC") armored bytes"
echo "RESULT: OK — Gradle's useInMemoryPgpKeys accepted the armored key + passphrase and produced a signature."
else
echo "::error::Gradle signing produced no .asc — useInMemoryPgpKeys could not build a usable signatory from MAVEN_GPG_PRIVATE_KEY / MAVEN_GPG_PASSPHRASE."
exit 1
fi
# ---------------------------------------------------------------------------
# Download + cache the GGUF test models ONCE, upfront, for the whole pipeline.
# Every Java test job (`test-java-*`) and the langchain4j integration job `needs:` this
# job and then only RESTORES the shared cache (key gguf-models-<manifest hash>) — so the ~5 GB model
# set is fetched from HuggingFace at most once per cache lifetime instead of racing to
# download in each job. GGUF is platform-independent, so this single ubuntu job's cache
# is reused by the macOS and Windows jobs too. On a warm cache this job is a no-op
# restore; on a cold cache it downloads all models, validates them, and saves the cache
# at job end (immutable key, so downstream jobs restore-hit). This is the single source
# of the download logic — do not re-add per-job downloads.
# ---------------------------------------------------------------------------
download-models:
name: Download + cache GGUF models (once)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Cache GGUF models (GitHub Actions cache; avoids re-downloading from HuggingFace)
# This job is the ONLY writer of the model cache (every consumer job uses the
# restore-only action). enableCrossOsArchive makes the one ubuntu-built entry
# restorable on macOS AND Windows — without it, cache entries are versioned
# per-OS, and the unreachable Windows-side entry was once found re-saved EMPTY
# (343 B) after an eviction, silently starving the Windows jobs of models.
uses: actions/cache@v6
with:
path: models/
# GGUF is platform-independent, so ubuntu + macOS + Windows share one entry.
# The key is derived from the model manifest, so EDITING models.csv
# automatically creates a fresh complete entry — no manual cache deletion.
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Download missing models (manifest-driven, .github/models.csv)
# The manifest is the single source of truth for the model set; this loop is
# the ONLY place in CI that downloads models. Files already restored from the
# cache are skipped, so a warm cache downloads nothing.
run: |
while IFS=, read -r name url; do
case "$name" in ''|\#*) continue ;; esac
test -f "models/$name" || curl -L --proto =https --proto-redir =https --fail --retry 5 --retry-all-errors "$url" --create-dirs -o "models/$name"
done < .github/models.csv
- name: List files in models directory
run: ls -l models/
- name: Validate model files
run: bash .github/validate-models.sh
# Prove the freshly written cache entry is restorable AND complete on every OS the
# pipeline uses BEFORE any model-consuming job starts: the same restore-only action
# the consumers use (fail-on-cache-miss makes an unrestorable entry fail here, not
# deep inside a test job), followed by the full validate gate. Every model-backed
# job `needs:` this instead of download-models directly.
verify-model-cache:
name: "Verify model cache (${{ matrix.os }})"
needs: download-models
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, macos-15, windows-2025-vs2026]
runs-on: ${{ matrix.os }}
steps:
- uses: actions/checkout@v7
- uses: ./.github/actions/restore-models
with:
fail-on-cache-miss: true
# ---------------------------------------------------------------------------
# Cross-compile jobs (Docker / dockcross) — produce release artifacts, no testing
# ---------------------------------------------------------------------------
code-style:
name: Code style (spotless) + package graph
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
# .github/buildcheck/natives.py; its unit tests run in the shared-files job.
- name: Natives list check (everything that names a natives jar agrees with natives.csv)
run: python3 .github/check-natives.py
- name: Spotless check (fail fast on format violations)
run: mvn -B --no-transfer-progress -f llama/pom.xml spotless:check
- name: SpotBugs check (fail fast on static-analysis findings)
run: mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile spotbugs:check
- name: Print internal package dependency graph (jdeps, informational)
continue-on-error: true
run: |
mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile
echo "=== internal package dependency graph (jdeps, bytecode) ==="
jdeps -verbose:package llama/target/classes | grep 'net.ladenthin.llama' || true
# ---------------------------------------------------------------------------
# Sibling module `llama-langchain4j` (LangChain4j adapters). Pure Java, no native
# code and no per-classifier matrix: it compiles against the core's stable Java API
# (identical across every classifier) and the backend is a runtime choice for the
# consumer. This job installs the parent + core into the local repo, then builds + tests
# the module (Java 17; langchain4j 1.x baseline). It runs its mapping unit tests; the
# model-backed integration test self-skips without a GGUF. `verify` also builds the
# javadoc/sources jars so a release-time javadoc break is caught here in PR CI. Version
# lockstep is now guaranteed by construction (both modules inherit the parent's version),
# so the old lockstep guard is gone.
# ---------------------------------------------------------------------------
test-java-llama-langchain4j:
name: Build and Test llama-langchain4j
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- name: Install parent + core net.ladenthin:llama into the local repo (Java only)
uses: ./.github/actions/build-core
- name: Build and test llama-langchain4j
run: mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml verify
# ---------------------------------------------------------------------------
# Model-free unit tests for the Kotlin coroutines facade (llama-kotlin).
# Pure Kotlin/JVM reactor module; its 6 tests fake the Iterable+AutoCloseable
# seam, so no native library and no model are needed.
# ---------------------------------------------------------------------------
test-java-llama-kotlin:
name: Build and Test llama-kotlin
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- name: Install parent + core net.ladenthin:llama into the local repo (Java only)
uses: ./.github/actions/build-core
- name: Build and test llama-kotlin
run: mvn -B --no-transfer-progress -f llama-kotlin/pom.xml verify
# ---------------------------------------------------------------------------
# Model-backed integration for the langchain4j adapters. Reuses the SAME shared GGUF
# cache (populated once by download-models) and the SAME Linux-x86_64 native artifact the
# core Java jobs already use — no extra model download and no duplicated download logic
# (restore-only cache, no curl steps). It exercises the chat / embedding / scoring
# adapters against the already-cached chat (Qwen3-0.6B), nomic-embedding and jina-reranker
# models. The model-backed tests self-skip when a model is absent, so a cold cache degrades
# to a skip, never a failure.
# ---------------------------------------------------------------------------
test-java-llama-langchain4j-integration:
name: Integration Test llama-langchain4j (model-backed)
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Download Linux x86_64 native library (reused, not rebuilt)
uses: actions/download-artifact@v8
with:
name: natives-cpu-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
- uses: ./.github/actions/restore-models
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install parent + core net.ladenthin:llama (classes; the test JVM loads the downloaded native library via lib.path)
uses: ./.github/actions/build-core
- name: Run llama-langchain4j model-backed integration tests (reused cached models)
run: >
mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml test
-Dnet.ladenthin.llama.model.path=models/${REASONING_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.embedding.model=models/${NOMIC_EMBED_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.rerank.model=models/${RERANKING_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.tool.model=models/${TOOL_MODEL_NAME}
-Dnet.ladenthin.llama.lib.path=${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/Linux/x86_64/cpu
# Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1).
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: bash .github/print-crash-logs.sh llama-langchain4j
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-langchain4j-integration
path: |
${{ github.workspace }}/llama-langchain4j/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama-langchain4j/*.hprof
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dump
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.txt
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
# ---------------------------------------------------------------------------
# Build the llama.cpp WebUI ONCE, from the same pinned tag CMakeLists.txt fetches,
# and share it to every native build as the generated, platform-independent
# ui.cpp/ui.h ("webui-generated" artifact). The native builds embed it into
# libjllama (CMake's "WebUI assets" block); when this job's artifact is absent the
# build falls back to the empty-asset stub. npm runs only here, in one controlled
# job — never in the dockcross cross-compilers (which have no node) or per-platform.
# ---------------------------------------------------------------------------
# ---------------------------------------------------------------------------
# llama-atmosphere-agent: the standalone (non-reactor) local
# coding-agent project that wires Atmosphere's built-in OpenAI-compatible agent runtime to
# this project's OpenAiCompatServer. Three jobs, all publish gates:
# - model-free: unit tests + the wire-contract tests, which drive the REAL
# OpenAiCompatServer over a loopback socket with a scripted backend (no native lib,
# no GGUF) and pin the streamed tool_calls / role=tool / multi-round shape — seconds,
# on every PR. It also builds the GitHub Release asset (the agent jar WITHOUT the core).
# - model-backed: the same loop against the cached Qwen2.5-1.5B tool model through the
# downloaded Linux native library (chat, streaming, tool call + result, read/write/read
# loop). A gate since its assertions stopped pinning wording (content checks are limited
# to facts no instruct model gets wrong and to tool results) and it ran green throughout.
# - smoke-agent-linux (further down, after package-fatjars): the release asset itself,
# started next to the real core fat jar.
# The project is built with -Dllama.version=<reactor version> against the core that was just
# installed to the local repo, so it always tests the code of this checkout. Only the classes and
# the llama-platform pom are installed; -Dllama.natives=none keeps Maven from resolving the natives
# jars that pom names (they exist only after the package job). The agent itself is published to
# Maven Central by the two publish jobs, after the reactor deploy.
# ---------------------------------------------------------------------------
test-java-llama-atmosphere-agent:
name: Build and Test llama-atmosphere-agent (model-free)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- name: Install parent + core net.ladenthin:llama + the llama-platform pom into the local repo (Java only)
uses: ./.github/actions/build-core
with:
modules: llama,llama-platform
- name: Resolve the reactor version
run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV"
- name: Spotless check
run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none spotless:check
- name: Build and test (unit + model-free wire contract against the real OpenAiCompatServer)
run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none verify
# The GitHub Release asset llama-atmosphere-agent-<core version>-jar-with-dependencies.jar:
# the agent plus Atmosphere/JLine, WITHOUT the core (src/assembly/agent-jar.xml), so it is a
# few MB and the natives are not in the release twice. Never deployed to Maven Central (the
# thin jar is; see the publish jobs).
# smoke-agent-linux launches it next to the real core fat jar; the attach jobs sign it.
- name: Build the agent release jar (without the core)
run: >
mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none
-P assembly -DskipTests package
- name: Collect the agent release jar + sha256
run: |
mkdir -p agent-jar
cp "llama-atmosphere-agent/target/llama-atmosphere-agent-${VERSION}-jar-with-dependencies.jar" agent-jar/
(cd agent-jar && for f in *.jar; do sha256sum "$f" > "$f.sha256"; done)
ls -la agent-jar
- name: Upload the agent release jar
uses: actions/upload-artifact@v7
with:
name: llama-atmosphere-agent-jar
path: agent-jar/
compression-level: 0 # jars are already deflated
retention-days: 7
if-no-files-found: error
test-java-llama-atmosphere-agent-integration:
name: Integration Test llama-atmosphere-agent (model-backed)
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Download Linux x86_64 native library (reused, not rebuilt)
uses: actions/download-artifact@v8
with:
name: natives-cpu-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
- uses: ./.github/actions/restore-models
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install parent + core net.ladenthin:llama + the llama-platform pom (classes; the test JVM loads the downloaded native library via lib.path)
uses: ./.github/actions/build-core
with:
modules: llama,llama-platform
- name: Resolve the reactor version
run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV"
- name: Run the Atmosphere tool-loop integration test (cached Qwen2.5-1.5B tool model, CPU)
run: >
mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none test
-Dtest=AtmosphereToolLoopIntegrationTest -Dsurefire.failIfNoSpecifiedTests=false
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME}
-Dnet.ladenthin.llama.test.ngl=0
-Dnet.ladenthin.llama.lib.path=${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/Linux/x86_64/cpu
# Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1).
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: bash .github/print-crash-logs.sh llama-atmosphere-agent
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-atmosphere-agent-integration
path: |
${{ github.workspace }}/llama-atmosphere-agent/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama-atmosphere-agent/*.hprof
build-webui:
name: Build WebUI assets (shared)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Resolve pinned llama.cpp tag from CMakeLists.txt
id: tag
shell: bash
run: |
TAG=$(grep -oE 'GIT_TAG[[:space:]]+b[0-9]+' llama/CMakeLists.txt | grep -oE 'b[0-9]+' | head -1)
if [ -z "$TAG" ]; then
echo "could not resolve llama.cpp GIT_TAG (b<nnnn>) from CMakeLists.txt" >&2
exit 1
fi
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "Pinned llama.cpp WebUI tag: $TAG"
- name: Checkout llama.cpp tools/ui at the pinned tag
uses: actions/checkout@v7
with:
repository: ggml-org/llama.cpp
ref: ${{ steps.tag.outputs.tag }}
path: llamacpp-ui
# tools/ui holds the npm project AND the ui.cpp.in / ui.h.in templates;
# scripts/ holds ui-assets.cmake, which consumes them. Both are needed
# since upstream #28445 replaced the embed.cpp host tool.
sparse-checkout: |
tools/ui
scripts
sparse-checkout-cone-mode: true
- uses: actions/setup-node@v7
with:
node-version: '24'
cache: npm
cache-dependency-path: llamacpp-ui/tools/ui/package-lock.json
- name: Build WebUI (Svelte/Vite)
working-directory: llamacpp-ui/tools/ui
env:
HF_UI_VERSION: ${{ steps.tag.outputs.tag }}
LLAMA_BUILD_NUMBER: ${{ steps.tag.outputs.tag }}
run: |
npm ci --ignore-scripts
npm run build
test -f dist/index.html
- name: Embed assets into ui.cpp / ui.h (upstream scripts/ui-assets.cmake)
shell: bash
run: |
set -euo pipefail
# Upstream #28445 ("ui : embed assets directly with CMake") deleted the
# tools/ui/embed.cpp host tool this step used to compile, and replaced it
# with scripts/ui-assets.cmake -- a plain `cmake -P` script, no npm and no
# host executable. Priority 1 of its provisioning order is "pre-built
# assets in <UI_SOURCE_DIR>/dist", which is exactly what the npm step
# above produced, so BUILD_UI and HF_ENABLED stay OFF: no second npm run
# and no Hugging Face download happen here. LLAMA_UI_GZIP is upstream's
# own knob and replaces the hand-rolled gzip loop this step used to do.
GEN="${RUNNER_TEMP}/ui-assets"
OUT="${GITHUB_WORKSPACE}/llama/webui-generated"
mkdir -p "$GEN" "$OUT"
cmake \
"-DUI_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui/tools/ui" \
"-DUI_BINARY_DIR=${GEN}" \
"-DLLAMA_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui" \
-DBUILD_UI=OFF \
-DHF_ENABLED=OFF \
-DLLAMA_UI_GZIP=ON \
-P "${GITHUB_WORKSPACE}/llamacpp-ui/scripts/ui-assets.cmake"
# The script also drops a ui-gzip/ working tree next to the generated
# sources; copy only the two files the artifact is defined to carry.
cp "$GEN/ui.cpp" "$GEN/ui.h" "$OUT/"
echo "=== generated WebUI assets ==="
ls -la "$OUT"
# Guard against a silently empty WebUI. A bare `grep LLAMA_UI_HAS_ASSETS`
# does NOT work here and would pass the failure case: upstream's ui.h.in
# emits "/* #undef LLAMA_UI_HAS_ASSETS */" when the table is empty, so the
# token is present either way. (The old embed.cpp emitted no such line at
# all, which is why the naive grep used to be sufficient.) Assert the ACTIVE
# #define and a non-zero asset count instead -- verified against both paths.
N=$(sed -n 's/.*std::array<llama_ui_asset, \([0-9]\+\)>.*/\1/p' "$OUT/ui.h" | head -1)
if grep -qE '^[[:space:]]*#define[[:space:]]+LLAMA_UI_HAS_ASSETS' "$OUT/ui.h" \
&& [ -n "$N" ] && [ "$N" -gt 0 ]; then
echo "LLAMA_UI_HAS_ASSETS: present, $N assets embedded"
else
echo "ERROR: ui-assets.cmake produced an empty asset table (assets=${N:-unknown})" >&2
exit 1
fi
- name: Upload WebUI artifact
uses: actions/upload-artifact@v7
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
retention-days: 1
if-no-files-found: error
crosscompile-linux-x86_64-cuda:
name: Cross-Compile manylinux_2_28 x86_64 (CUDA)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# CUDA cache. build_cuda_linux.sh execs build.sh, so the same sccache probe guards this job.
# build.sh also wraps nvcc (CMAKE_CUDA_COMPILER_LAUNCHER=sccache) for CUDA builds, so the
# per-arch .cu device passes — the dominant cost of this job — cache over Depot alongside the
# gcc host TUs. Verified on a warm run: 100% hit on CUDA / CUBIN / device-code (139 CUDA hits,
# 99.86% overall), cutting the job from ~51 min cold to ~15 min warm. The job therefore always
# builds the FULL CMAKE_CUDA_ARCHITECTURES set (no single-arch shortcut) and leans on the warm
# cache for speed, so every artifact stays release-safe (runs on every GPU generation) on PR /
# push as well as publish. CUDA_FAST_BUILD still exists in build_cuda_linux.sh as a LOCAL-dev
# knob, but CI no longer sets it. The first-run sccache debug diagnostics (SCCACHE_LOG /
# SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that caching is confirmed; build.sh still
# prints the `sccache --show-stats` hit table at the end of every run. Inert without DEPOT_TOKEN
# (fork PRs) or use_cache=false.
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-manylinux_2_28-x64 .github/build_cuda_linux.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cuda13-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
crosscompile-linux-x86_64:
name: Cross-Compile manylinux2014 x86_64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 1, VERIFIED green in CI (PR #245): sccache v0.16.0
# probe passed in-container (devtoolset-10 gcc), cache ON over Depot WebDAV (cold run: 275
# objects stored). Steady-state env below — the first-run diagnostics (SCCACHE_LOG /
# SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that it is proven. Inert without
# DEPOT_TOKEN (fork PRs) or with use_cache=false; a crashing sccache still falls back to a
# green uncached build via the build.sh probe.
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-manylinux2014-x64 .github/build.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
crosscompile-linux-aarch64:
name: Build and Test Linux aarch64
needs: [startgate, build-webui]
# Native ARM64 build on GitHub's free arm64 runner, mirroring upstream llama.cpp's
# `ubuntu-cpu` aarch64 release job (ubuntu-24.04-arm + GCC 14). Replaces the former dockcross
# `linux-arm64-lts` cross-compile (GCC 8.5, glibc 2.17), which can no longer compile llama.cpp
# b9789 — its C++17 CTAD-in-`new` needs GCC >= 12. Building natively also lets us run the C++
# unit suite (ctest) on real ARM hardware for the first time (the cross build ran no tests).
# Trade-off: the glibc floor rises 2.17 -> ~2.39, the same envelope upstream's own ARM binaries
# require. GGML_NATIVE=OFF keeps the artifact portable across ARMv8 CPU generations (no
# build-host -march baked in). The job id is kept (a `needs:` target downstream); only the
# display name changed, so update any branch-protection required-check that pinned the old name.
runs-on: ubuntu-24.04-arm
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install toolchain (GCC 14, mirrors upstream llama.cpp ARM release)
run: |
sudo apt-get update
sudo apt-get install -y gcc-14 g++-14
echo "CC=gcc-14" >> "$GITHUB_ENV"
echo "CXX=g++-14" >> "$GITHUB_ENV"
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DOS_NAME=Linux -DOS_ARCH=aarch64 -DGGML_NATIVE=OFF -DBUILD_TESTING=ON"
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-linux-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-linux-s390x:
name: Build and Test Linux s390x (big-endian, qemu)
needs: [startgate, build-webui]
# Cross-compile for IBM Z (s390x, BIG-ENDIAN) with the GCC cross toolchain, then run the full
# C++ unit suite under qemu-user — a real big-endian correctness gate for our helpers and
# serializers (esp. the little-endian WAV writer, JSON/token/embedding transforms). The BUILD
# is native speed (x86 cross-gcc); only the tiny test binary is emulated. s390x is a CPU
# platform like aarch64 and ships as the cpu-linux-s390x natives jar. Model-backed Java tests
# are NOT run under emulation (a JVM + GGUF inference under qemu-user is slow/flaky); the
# C++ gate covers the actual byte-order risk since the Java<->JNI boundary uses host-native
# array copies. GGML_OPENMP=OFF avoids cross-libgomp issues (ggml uses
# its own std::thread pool). CMAKE_CROSSCOMPILING_EMULATOR makes ctest run the s390x exe via qemu;
# QEMU_LD_PREFIX lets the emulated binary find the s390x sysroot libs.
runs-on: ubuntu-latest
env:
QEMU_LD_PREFIX: /usr/s390x-linux-gnu
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install s390x cross toolchain + qemu-user
run: |
sudo apt-get update
sudo apt-get install -y gcc-s390x-linux-gnu g++-s390x-linux-gnu qemu-user-static
# NOTE: GGML_NATIVE=OFF is required here (an x86 build host must not bake -march=native into an
# s390x artifact), and it has a non-obvious second effect: ggml declares
# `option(GGML_VXE "ggml: enable vxe" ${GGML_NATIVE})`, so VXE is off too. No `-mvx -mzvector` is
# passed, `__VEC__` stays undefined, and ggml-cpu-impl.h's `#if defined(__s390x__) && defined(__VEC__)`
# never self-defines `__VXE__`/`__VXE2__`. This job therefore builds a SCALAR s390x binary -- which is
# exactly right for what it is (a big-endian correctness gate for our own layer, not a perf target).
# Do NOT "fix" a VXE-related compile error by adding -DGGML_VXE=ON: that define sets __VXE__ and
# __VXE2__ together while -march stays at the toolchain default arch11, so every z14+ builtin is
# rejected. See the `0013` note in docs/history/dropped-llama-patches.md for the measured
# comparison (the patch itself is gone -- upstream merged it as ggml-org/llama.cpp#28775).
- name: Build libraries (cross-compile s390x)
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_NATIVE=OFF -DGGML_OPENMP=OFF -DBUILD_TESTING=ON -DCMAKE_SYSTEM_NAME=Linux -DCMAKE_SYSTEM_PROCESSOR=s390x -DCMAKE_C_COMPILER=s390x-linux-gnu-gcc -DCMAKE_CXX_COMPILER=s390x-linux-gnu-g++ -DCMAKE_CROSSCOMPILING_EMULATOR=/usr/bin/qemu-s390x-static -DOS_NAME=Linux -DOS_ARCH=s390x"
- name: Run C++ unit tests under qemu-s390x (big-endian gate)
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-linux-s390x
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-linux-x86_64-vulkan:
name: Build Linux x86_64 Vulkan
needs: [startgate, build-webui]
# Native ubuntu build (NOT dockcross) — the Vulkan SDK is trivial to apt-install here, and
# upstream llama.cpp builds its ubuntu-vulkan artifact the same way. GPU runtime libvulkan.so.1
# is supplied by the consumer's driver (nothing bundled). GitHub runners have NO GPU, so this
# is a BUILD-ONLY job (no -DBUILD_TESTING/ctest: a Vulkan-linked jllama_test errors enumerating
# devices on a GPU-less runner — same rationale as the Windows GPU jobs). GGML_NATIVE=OFF keeps
# the artifact portable across x86_64 CPU generations. Trade-off vs the manylinux CPU jar: the
# glibc floor rises to the ubuntu-latest baseline (same as the native aarch64 job). build.sh
# self-fetches sccache; the probe guards it (a miss just builds uncached).
runs-on: ubuntu-latest
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install Vulkan SDK (headers + loader + glslc shader compiler)
run: |
sudo apt-get update
sudo apt-get install -y libvulkan-dev glslc glslang-tools spirv-headers
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-vulkan-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-linux-aarch64-vulkan:
name: Build Linux aarch64 Vulkan
needs: [startgate, build-webui]
# Native ARM64 Vulkan build on GitHub's free arm64 runner (same runner as the aarch64 CPU job).
# Build-only (GPU-less runner); GGML_NATIVE=OFF for portability across ARMv8 generations; GCC 14
# to match the aarch64 CPU job. Ships as the vulkan-linux-aarch64 natives jar.
runs-on: ubuntu-24.04-arm
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install toolchain (GCC 14) + Vulkan SDK
run: |
sudo apt-get update
sudo apt-get install -y gcc-14 g++-14 libvulkan-dev glslc glslang-tools spirv-headers
echo "CC=gcc-14" >> "$GITHUB_ENV"
echo "CXX=g++-14" >> "$GITHUB_ENV"
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=aarch64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-vulkan-linux-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
crosscompile-android-aarch64:
name: Cross-Compile Android aarch64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 4. Same steady-state env as manylinux2014 (job 1);
# the build.sh probe makes it safe to enable without a separate verification run. Inert
# without DEPOT_TOKEN (fork PRs) or use_cache=false.
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-arm64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-android-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
crosscompile-android-x86_64:
name: Cross-Compile Android x86_64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Android x86_64 CPU ABI: consumed by the llama-android AAR as jni/x86_64 (so the
# Android emulator — and x86_64 Android devices/Chromebooks — can run the binding)
# and shipped as the cpu-android-x86-64 natives jar. Same dockcross + sccache steady-state env as the arm64 job; the wrapper
# pins the same image tag. Fail-loud and in the package/publish needs graphs.
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-x86_64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-android-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
crosscompile-android-aarch64-opencl:
name: Cross-Compile Android aarch64 (OpenCL/Adreno)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 5. build_opencl_android.sh stages the OpenCL
# headers/loader, then delegates the jllama cmake build to build.sh (which owns the
# sccache probe + launcher). Same steady-state env as the other dockcross jobs. Inert
# without DEPOT_TOKEN (fork PRs) or use_cache=false.
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-arm64 .github/build_opencl_android.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64 -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-opencl-android-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Android AAR packaging (net.ladenthin:llama-android + llama-android-opencl).
# Plain-Gradle AAR assembly — no AGP and no Android SDK needed to BUILD (an AAR
# is a documented zip; see llama-android/README.md); AGP is only needed to
# CONSUME it, which the consumer smoke test below exercises on the runner's
# preinstalled Android SDK: a minimal app resolves the AAR from mavenLocal and
# runs a full R8 release build (validating AAR format, manifest minSdk merge,
# jni/ packaging, and the shipped consumer proguard rules). Structural checks
# additionally pin the AAR entries and the 16 KB LOAD-segment alignment
# (Google Play requirement for Android 15+ targets). Fail-loud and in the
# publish `needs:` graphs — a broken AAR blocks publishing, same policy as
# every native artifact job.
# ---------------------------------------------------------------------------
package-android-aar:
name: Package + Validate Android AARs
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, crosscompile-android-aarch64-opencl]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0.
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Build core jar (byte-identical classes payload for the AAR)
uses: ./.github/actions/build-core
with:
goal: package
- name: Download Android CPU natives (arm64)
uses: actions/download-artifact@v8
with:
name: natives-cpu-android-aarch64
path: stage/cpu/
- name: Download Android CPU natives (x86_64)
uses: actions/download-artifact@v8
with:
name: natives-cpu-android-x86-64
path: stage/cpu-x86_64/
- name: Download Android OpenCL natives
uses: actions/download-artifact@v8
with:
name: natives-opencl-android-aarch64
path: stage/opencl/
- name: Stage natives for the AAR build
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp stage/cpu/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp stage/cpu-x86_64/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/
cp stage/opencl/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/
- name: Assemble AARs + publish to mavenLocal
run: gradle -p llama-android aarCpu aarOpencl publishToMavenLocal
- name: Validate AAR structure
shell: bash
run: |
set -euo pipefail
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
for flavor in llama-android llama-android-opencl; do
AAR="llama-android/build/aar/${flavor}-${VERSION}.aar"
echo "== validating $AAR"
test -f "$AAR"
# The CPU AAR is multi-ABI (arm64 devices + x86_64 emulators/devices);
# the OpenCL flavor stays arm64-only (Adreno is Qualcomm ARM hardware).
ABIS="arm64-v8a"
if [ "$flavor" = "llama-android" ]; then ABIS="arm64-v8a x86_64"; fi
ENTRIES="AndroidManifest.xml classes.jar proguard.txt R.txt"
for abi in $ABIS; do ENTRIES="$ENTRIES jni/$abi/libjllama.so"; done
for entry in $ENTRIES; do
unzip -l "$AAR" | grep -q "${entry}$" || { echo "::error::$AAR is missing $entry"; exit 1; }
done
unzip -p "$AAR" AndroidManifest.xml | grep -q 'android:minSdkVersion="28"' \
|| { echo "::error::$AAR manifest lost minSdkVersion 28"; exit 1; }
unzip -p "$AAR" classes.jar > /tmp/aar-classes.jar
unzip -l /tmp/aar-classes.jar | grep -q "net/ladenthin/llama/LlamaModel.class" \
|| { echo "::error::$AAR classes.jar lost LlamaModel"; exit 1; }
if unzip -l /tmp/aar-classes.jar | grep -qE "module-info\.class|net/ladenthin/llama/(Linux|Mac|Windows)/"; then
echo "::error::$AAR classes.jar carries module-info or desktop native resources"; exit 1
fi
# The libraries themselves (bionic-only DT_NEEDED + libOpenCL.so for the OpenCL
# flavour, every LOAD segment 16 KB aligned for Google Play) are checked on the staged
# tree below, by the same allowlists the package job applies to every natives jar.
cmp -s <(unzip -p "$AAR" jni/arm64-v8a/libjllama.so) \
"llama-android/natives/$([ "$flavor" = llama-android ] && echo cpu || echo opencl)/arm64-v8a/libjllama.so" \
|| { echo "::error::$AAR jni/arm64-v8a/libjllama.so is not the staged library"; exit 1; }
done
- name: Android libraries -- dependency allowlist + 16 KB page alignment (buildcheck/nativedeps.py)
run: python3 .github/verify-native-deps.py stage
- name: AGP consumer smoke test (R8 release build from mavenLocal)
shell: bash
run: |
set -euo pipefail
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
gradle -p .github/android-consumer-test assembleRelease "-PjllamaVersion=${VERSION}"
APK=.github/android-consumer-test/app/build/outputs/apk/release/app-release.apk
test -f "$APK"
for abi in arm64-v8a x86_64; do
unzip -l "$APK" | grep -q "lib/$abi/libjllama.so" \
|| { echo "::error::consumer APK is missing lib/$abi/libjllama.so"; exit 1; }
done
# The AAR's consumer proguard.txt must have carried the binding through R8.
unzip -p "$APK" "classes*.dex" | grep -aq "Lnet/ladenthin/llama/LlamaModel;" \
|| { echo "::error::R8 stripped net.ladenthin.llama.LlamaModel — consumer proguard rules broken"; exit 1; }
- name: Upload AARs
uses: actions/upload-artifact@v7
with:
name: llama-android-aars
path: llama-android/build/aar/*.aar
if-no-files-found: error
# ---------------------------------------------------------------------------
# On-emulator runtime validation of the Android AAR: boots a KVM-accelerated
# x86_64 emulator (GitHub Linux runners have KVM; arm64 images cannot run here,
# which is exactly why the AAR carries the jni/x86_64 ABI), publishes the CPU AAR
# to mavenLocal, adb-pushes the already-cached tiny draft model, and runs the
# consumer fixture's connectedDebugAndroidTest — System.loadLibrary from the APK,
# pure-Java GgufInspector on-device, and real native inference on Android/bionic.
# RELEASE GATE (in both publish needs graphs) since PR #298: the job ran
# flake-free through the PR's validation cycle (boot ~30 s, on-device inference
# green), so a broken on-device runtime now blocks publishing — same fail-loud
# policy as every native artifact job. If emulator-boot flakiness ever appears,
# re-run the job first; demote it from the needs graphs only as a last resort.
# ---------------------------------------------------------------------------
test-android-emulator:
name: Android emulator on-device test (x86_64)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- name: Free up disk space
# The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages"
# (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap
# file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it.
uses: ggml-org/free-disk-space@v1.3.1
with:
android: false
large-packages: false
tool-cache: false
swap-storage: false
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0.
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Enable KVM group permissions (GitHub-hosted runner)
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build core jar (classes payload for the AAR)
uses: ./.github/actions/build-core
with:
goal: package
- uses: ./.github/actions/publish-cpu-aar-local
- uses: ./.github/actions/restore-models
- name: Free disk space for the emulator (only the draft model is needed on-device)
# Same guard as test-android-llmservice: the AVD userdata partition needs ~7.4 GB; drop the
# rest of the restored ~10 GB GGUF cache (this job adb-pushes only the tiny draft model) plus
# the few preinstalled toolchains the free-disk-space step leaves, so the emulator can create
# userdata and boot.
run: |
echo "Disk before cleanup:"; df -h / | tail -1
find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true
sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true
echo "Disk after cleanup:"; df -h / | tail -1
- name: Run on-emulator instrumentation (connectedDebugAndroidTest)
uses: reactivecircus/android-emulator-runner@v2
with:
api-level: 30
arch: x86_64
target: default
disable-animations: true
emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim
# One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh,
# so shell control flow must live in the committed helper script.
script: .github/run-android-emulator-test.sh
- name: Upload instrumentation reports (on failure)
if: failure()
uses: actions/upload-artifact@v7
with:
name: android-emulator-test-reports
path: .github/android-consumer-test/app/build/reports/androidTests/
if-no-files-found: ignore
# ---------------------------------------------------------------------------
# The shippable Android app "LLM Service" (android-llmservice) — a KISS, fully-offline
# on-device chat app consuming the llama-android AAR + llama-kotlin facade. Split into TWO jobs
# so the installable artifacts are ALWAYS produced even when the on-device UI test is flaky/slow:
# * build-android-llmservice — builds the signed release AAB (real upload key when the
# ANDROID_UPLOAD_KEYSTORE_BASE64 secret is set, else debug-signed) + the installable APK and
# uploads both. No emulator, so a flaky/slow emulator can NEVER block getting the APK.
# * test-android-llmservice — a SEPARATE, non-gating check that boots the KVM x86_64 emulator
# and runs the app's Compose UI test (type a prompt, tap Send) against real on-device
# inference. It can go red on its own without stopping the build/artifacts.
# Neither is a publish gate (unlike test-android-emulator): a Compose/AGP toolchain hiccup must
# not block a Maven Central release of the library. Keep test-android-llmservice non-required in
# branch protection to keep the emulator test optional (visible-but-non-blocking).
# ---------------------------------------------------------------------------
build-android-llmservice:
name: Build the LLM Service Android app (AAB + APK)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0; also satisfies Kotlin
# 2.4's Gradle-plugin floor. The AAR/lib-only jobs stay on an older Gradle since
# they build no AGP project.
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Build core jar + install llama-kotlin facade to mavenLocal
# install (not package) so the app's Gradle build resolves llama-kotlin +
# the core POM from mavenLocal. llama-kotlin's core dep is provided-scope, so the
# desktop JAR is never pulled into the APK — the AAR supplies net.ladenthin.llama.*.
uses: ./.github/actions/build-core
with:
modules: llama,llama-kotlin
- uses: ./.github/actions/publish-cpu-aar-local
- name: Decode upload keystore (if configured)
# Optional Play upload key. Set the ANDROID_UPLOAD_KEYSTORE_BASE64 secret
# (base64 of a PKCS12/JKS upload keystore) plus the three password/alias secrets
# below to sign the release AAB with a real upload key. Without it the release
# build falls back to debug signing (app/build.gradle.kts) so this step is inert
# on forks/PRs where secrets are withheld.
env:
KEYSTORE_B64: ${{ secrets.ANDROID_UPLOAD_KEYSTORE_BASE64 }}
run: |
if [ -n "${KEYSTORE_B64}" ]; then
echo "${KEYSTORE_B64}" | base64 -d > "${RUNNER_TEMP}/upload-keystore.jks"
echo "JLLAMA_UPLOAD_STORE_FILE=${RUNNER_TEMP}/upload-keystore.jks" >> "$GITHUB_ENV"
echo "Upload keystore decoded -> release AAB will be signed with the upload key."
else
echo "No ANDROID_UPLOAD_KEYSTORE_BASE64 secret -> release AAB will be debug-signed."
fi
- name: Build release bundle (AAB) + installable APK
env:
JLLAMA_UPLOAD_STORE_PASSWORD: ${{ secrets.ANDROID_UPLOAD_STORE_PASSWORD }}
JLLAMA_UPLOAD_KEY_ALIAS: ${{ secrets.ANDROID_UPLOAD_KEY_ALIAS }}
JLLAMA_UPLOAD_KEY_PASSWORD: ${{ secrets.ANDROID_UPLOAD_KEY_PASSWORD }}
run: |
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
# bundleRelease -> app-release.aab (Play upload, NOT directly installable);
# assembleRelease -> app-release.apk (sideloadable installable, debug-signed unless the
# upload-key secrets are set). The debug + androidTest APKs are built by the emulator
# test job (test-android-llmservice), not here — this job never touches the emulator.
gradle -p android-llmservice bundleRelease assembleRelease "-PjllamaVersion=${VERSION}"
test -f android-llmservice/app/build/outputs/bundle/release/app-release.aab
test -f android-llmservice/app/build/outputs/apk/release/app-release.apk
- name: Upload release bundle (AAB — for Play upload)
uses: actions/upload-artifact@v7
with:
name: android-llmservice-aab
path: android-llmservice/app/build/outputs/bundle/release/*.aab
if-no-files-found: error
- name: Upload installable APK (release — sideload / adb install)
# Directly installable APK, downloadable from this run's Artifacts (independent of any
# release). Debug-signed unless the ANDROID_UPLOAD_* secrets are set. This is the file to
# grab to try the app on a phone; the .aab above is only for the Play Console.
uses: actions/upload-artifact@v7
with:
name: android-llmservice-apk
path: android-llmservice/app/build/outputs/apk/release/*.apk
if-no-files-found: error
test-android-llmservice:
name: LLM Service app UI test on emulator (non-gating)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- name: Free up disk space
# The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages"
# (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap
# file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it.
uses: ggml-org/free-disk-space@v1.3.1
with:
android: false
large-packages: false
tool-cache: false
swap-storage: false
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0.
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Enable KVM group permissions (GitHub-hosted runner)
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build core jar + install llama-kotlin facade to mavenLocal
uses: ./.github/actions/build-core
with:
modules: llama,llama-kotlin
- uses: ./.github/actions/publish-cpu-aar-local
- uses: ./.github/actions/restore-models
- name: Free disk space for the emulator (only the draft model is needed on-device)
# The AVD userdata partition needs ~7.4 GB; the full ~10 GB GGUF cache restore leaves too
# little free, so the emulator FATALs ("Not enough space to create userdata partition") and
# the boot poll loops until timeout. This job only adb-pushes the tiny draft model, so drop
# the rest of the restored cache plus what the free-disk-space step leaves (PowerShell, CodeQL).
run: |
echo "Disk before cleanup:"; df -h / | tail -1
find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true
sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true
echo "Disk after cleanup:"; df -h / | tail -1
- name: Run LLM Service UI test on emulator (connectedDebugAndroidTest)
uses: reactivecircus/android-emulator-runner@v2
with:
api-level: 30
arch: x86_64
target: default
disable-animations: true
emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim
# One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh.
script: .github/run-android-llmservice-test.sh
- name: Upload instrumentation reports (on failure)
if: failure()
uses: actions/upload-artifact@v7
with:
name: android-llmservice-test-reports
path: android-llmservice/app/build/reports/androidTests/
if-no-files-found: ignore
# ---------------------------------------------------------------------------
# Native build jobs — produce release artifacts + run C++ unit tests
# ---------------------------------------------------------------------------
build-macos-arm64-no-metal:
name: Build and Test macOS 15 arm64 (no Metal)
needs: [startgate, build-webui]
runs-on: macos-15
env:
BUILD_JOBS: 2
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DGGML_METAL=OFF -DGGML_NATIVE=OFF -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: macos-15-no-metal
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-macos-arm64-metal:
name: Build and Test macOS 14 arm64 (Metal)
needs: [startgate, build-webui]
runs-on: macos-14
env:
BUILD_JOBS: 2
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: macos-14-metal
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-windows-x86_64-msvc:
name: Build and Test Windows 2025 x86_64 (MSVC / VS 2026, classifier)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Visual Studio 18 2026" -A "x64" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-msvc-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-windows-x86-msvc:
name: Build and Test Windows 2025 x86 (MSVC / VS 2026, classifier)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Visual Studio 18 2026" -A "Win32" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-msvc-windows-x86
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Windows Ninja Multi-Config + sccache — the DEFAULT Windows CPU natives.
# The Visual Studio generator ignores CMAKE_{C,CXX}_COMPILER_LAUNCHER, so only the
# Ninja Multi-Config generator can front cl.exe with sccache over Depot WebDAV
# (build.bat probe-guards it). Both generators use the same MSVC toolchain (cl.exe,
# static /MT CRT) on the same runner, so the produced jllama.dll/llama.dll/ggml.dll
# are functionally equivalent with identical runtime dependencies — the only delta
# is build-system plumbing + caching. The Ninja build therefore ships as the
# cpu-windows-* natives jars; the MSVC build above as msvc-windows-*, for anyone who
# wants the Visual-Studio-generator natives. Upstream llama.cpp also
# builds its Windows artifacts with Ninja Multi-Config + MSVC.
# ---------------------------------------------------------------------------
build-windows-x86_64:
name: Build and Test Windows 2025 x86_64 (Ninja Multi-Config + sccache, default)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-windows-x86:
name: Build and Test Windows 2025 x86 (Ninja Multi-Config + sccache, default)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x86)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x86
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-windows-x86
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
build-windows-arm64:
name: Build and Test Windows 11 arm64 (Ninja Multi-Config, default)
needs: [startgate, build-webui]
# Native arm64 build on GitHub's free windows-11-arm runner; ships as the cpu-windows-aarch64
# natives jar: OSInfo maps a Windows-on-ARM JVM (os.arch=aarch64) to Windows/aarch64, the same
# path CMake emits here. sccache: the native aarch64-pc-windows-msvc release, wrapping clang-cl
# (guarded by build.bat's probe + uncached retry like every other Windows job).
#
# Compiler: clang-cl, NOT MSVC cl.exe. ggml's ggml-cpu/CMakeLists.txt aborts with "MSVC is not
# supported for ARM, use clang" via `if (MSVC AND NOT CMAKE_C_COMPILER_ID STREQUAL "Clang")`.
# clang-cl (LLVM's MSVC-compatible driver) satisfies that guard (its compiler id is "Clang")
# while still leaving CMake's MSVC=TRUE, so our static /MT CRT block (CMAKE_MSVC_RUNTIME_LIBRARY
# in CMakeLists.txt) keeps applying and the generator stays Ninja Multi-Config. msvc-dev-cmd
# (arm64) supplies the MSVC headers/libs/linker AND the bundled clang-cl / lld-link under
# VC\Tools\Llvm\ARM64, so no separate LLVM install is needed.
#
# GGML_OPENMP=OFF: with clang-cl, ggml links LLVM's OpenMP (libomp.lib -> needs libomp140.aarch64.dll
# at runtime), which is NOT on PATH like MSVC's ambient vcomp140.dll on x64 — so gtest_discover_tests
# (and any consumer) failed to launch the binary with 0xc0000135 STATUS_DLL_NOT_FOUND. Turning OpenMP
# off makes ggml use its own std::thread threadpool, so the arm64 jllama.dll (and the test exe) are
# self-contained with no libomp dependency to ship. The x86_64/x86 jobs keep OpenMP (MSVC vcomp).
runs-on: windows-11-arm
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (arm64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: arm64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# No mvn compile needed: the JNI header (jllama.h) is committed and the native build
# uses the bundled JNI headers in .github/include, and OS_NAME/OS_ARCH are passed
# explicitly (so the OSInfo-class OS-detection path is skipped) — same as the x86_64 job.
# clang-cl (see the job comment) is required: ggml refuses MSVC cl.exe on ARM.
run: |
.github\build.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DOS_NAME=Windows -DOS_ARCH=aarch64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cpu-windows-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Windows GPU classifiers (x86_64 only) — CUDA, Vulkan, OpenCL.
# All three use the same Ninja Multi-Config + MSVC + sccache toolchain as the
# default CPU build; they differ only by the GGML backend flag (and the build-time
# SDK each needs). CMakeLists.txt routes each backend's output to its own
# src/main/natives/.../Windows/x86_64/<backend>/ directory, which the natives Maven
# profile turns into a natives jar. GPU runtime libraries are NOT bundled — the
# consumer's GPU driver / toolkit provides them (CUDA: cudart64_13/cublas64_13 from the CUDA Toolkit; Vulkan:
# vulkan-1.dll from the driver; OpenCL: System32\OpenCL.dll from the driver).
# NOTE: GitHub-hosted Windows runners have NO GPU, so these jobs build + run the
# C++ unit suite (ctest, CPU-only) but cannot run model-backed GPU inference;
# end-to-end GPU validation is local / self-hosted.
# ---------------------------------------------------------------------------
build-windows-x86_64-cuda:
name: Build Windows 2025 x86_64 CUDA (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install CUDA Toolkit 13.4 (NVIDIA redist archives)
# KEEP IN SYNC WITH UPSTREAM: the component set and versions mirror llama.cpp's
# .github/actions/windows-setup-cuda ("Install Cuda Toolkit 13.4 for x64") at the pinned
# GIT_TAG. Jimver/cuda-toolkit (used here up to CUDA 13.3.1) has no 13.4 in any release or
# on master, so the toolkit is assembled from NVIDIA's per-component redist zips instead —
# which is also how upstream builds its own Windows CUDA release. cuda_crt is listed
# explicitly: since 13.x the nvcc crt headers (crt/host_config.h) ship in their own
# archive, and without them CMake's CUDA compiler detection fails at configure.
shell: pwsh
run: |
$ErrorActionPreference = "Stop"
$cuda = "C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v13.4"
$base = "https://developer.download.nvidia.com/compute/cuda/redist"
$parts = @(
"cuda_crt/cuda_crt-windows-x86_64-13.4.59",
"cuda_cudart/cuda_cudart-windows-x86_64-13.4.49",
"cuda_nvcc/cuda_nvcc-windows-x86_64-13.4.59",
"cuda_nvrtc/cuda_nvrtc-windows-x86_64-13.4.59",
"libcublas/libcublas-windows-x86_64-13.7.0.27",
"libnvvm/libnvvm-windows-x86_64-13.4.59",
"cuda_nvtx/cuda_nvtx-windows-x86_64-13.4.49",
"cuda_profiler_api/cuda_profiler_api-windows-x86_64-13.4.49",
"visual_studio_integration/visual_studio_integration-windows-x86_64-13.4.49",
"cccl/cccl-windows-x86_64-13.3.4.2.1"
)
New-Item -ItemType Directory -Force -Path $cuda | Out-Null
foreach ($p in $parts) {
$dir, $name = $p.Split("/")
$url = "$base/$dir/windows-x86_64/$name-archive.zip"
$zip = "$env:RUNNER_TEMP\$name.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile $zip
Expand-Archive -Path $zip -DestinationPath "$env:RUNNER_TEMP\cuda-redist" -Force
Copy-Item -Path "$env:RUNNER_TEMP\cuda-redist\$name-archive\*" -Destination $cuda -Recurse -Force
}
"$cuda\bin" | Out-File -FilePath $env:GITHUB_PATH -Append -Encoding utf8
"CUDA_PATH=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8
"CUDA_PATH_V13_4=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# GPU jobs build the artifact only — no -DBUILD_TESTING / ctest. The C++ unit
# suite is CPU-only and fully covered by the `C++ Tests` job + the CPU Windows
# jobs; a GPU-linked jllama_test.exe cannot be discovered/run on a GPU-less
# GitHub runner (it errors probing for a CUDA device -> ctest *_NOT_BUILT).
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_CUDA=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-cuda13-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-vulkan:
name: Build Windows 2025 x86_64 Vulkan (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install Vulkan SDK
uses: jakoch/install-vulkan-sdk-action@v1.6.0
with:
vulkan_version: 1.4.350.0
cache: true
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# Build the artifact only (see the CUDA job's note: GPU-less runner can't run a
# GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs).
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_VULKAN=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-vulkan-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-opencl:
name: Build Windows 2025 x86_64 OpenCL (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# Build the artifact only (see the CUDA job's note: GPU-less runner can't run a
# GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs).
run: |
.github\build_opencl_windows.bat -G "Ninja Multi-Config" -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-opencl-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
# ---------------------------------------------------------------------------
# Additional GPU-backend classifiers (fail-loud, same wiring as the CUDA/Vulkan/
# OpenCL jobs): AMD ROCm/HIP, Intel SYCL (oneAPI), Windows-on-ARM OpenCL (Adreno),
# Intel OpenVINO. All BUILD-ONLY (GitHub runners have no AMD/Intel/Adreno GPU, and
# no ctest — a GPU-linked jllama_test can't enumerate a device). GPU runtime libs
# are NOT bundled — the consumer's driver/toolkit supplies them. CMakeLists.txt
# routes each backend to its own src/main/natives/<OS>/<ARCH>/<backend>/ directory; the
# natives Maven profile turns it into a natives jar. Toolchain install steps are first-pass —
# if a vendor URL/version 404s in CI, adjust it (the failure is intentional signal).
# ---------------------------------------------------------------------------
build-linux-x86_64-rocm:
name: Build Linux x86_64 ROCm/HIP (AMD)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The TheRock wheels unpack to several GB; upstream's ubuntu-rocm job frees the runner
# the same way. Runs first, before setup-java puts the JDK into the tool cache.
uses: ggml-org/free-disk-space@v1.3.1
with:
tool-cache: true
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install ROCm/HIP (TheRock wheels)
# KEEP IN SYNC WITH UPSTREAM: ROCM_VERSION tracks the ubuntu-rocm job in llama.cpp's
# .github/workflows/release.yml at the pinned GIT_TAG (GPU targets: see the build step). Since ROCm 7.14
# AMD builds and releases ROCm through TheRock (https://github.com/ROCm/TheRock), replacing
# the monolithic releases behind the repo.radeon.com apt repo this job used before (it was
# pinned at 6.3.4). The wheels carry the HIP runtime + CMake configs ("libraries") and the
# compilers/headers ("devel"); rocm-sdk reports where they landed.
env:
ROCM_VERSION: "10.0.0"
run: |
python3 -m venv "$RUNNER_TEMP/rocm-venv"
source "$RUNNER_TEMP/rocm-venv/bin/activate"
python -m pip install --upgrade pip
python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==${ROCM_VERSION}"
ROCM_PATH=$(rocm-sdk path --root)
echo "ROCM_PATH=$ROCM_PATH" >> "$GITHUB_ENV"
echo "HIP_PATH=$ROCM_PATH" >> "$GITHUB_ENV"
echo "CMAKE_PREFIX_PATH=$(rocm-sdk path --cmake)" >> "$GITHUB_ENV"
echo "LD_LIBRARY_PATH=$ROCM_PATH/lib:${LD_LIBRARY_PATH:-}" >> "$GITHUB_ENV"
echo "$(rocm-sdk path --bin)" >> "$GITHUB_PATH"
echo "$RUNNER_TEMP/rocm-venv/bin" >> "$GITHUB_PATH"
- name: Build libraries
shell: bash
# Native CMake HIP language (upstream's form): the HIP compiler is ROCm's clang, the C/C++
# TUs stay on the runner's gcc. The target list is every Linux target TheRock builds
# (its SUPPORTED_GPUS.md) — a superset of upstream llama.cpp's list, which omits
# gfx900/gfx906/gfx90c/gfx1153 ("build passing" only, not release-ready). Better too many
# than too few, but only while they cost nothing: drop an extra the moment it needs a
# patch or blocks a newer ROCm. Linux alone has the Instinct parts (gfx908/90a/942/950):
# ROCm does not support them on Windows.
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_HIP=ON -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGPU_TARGETS=gfx900;gfx906;gfx908;gfx90a;gfx90c;gfx942;gfx950;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Verify the GPU code is compressed
# ggml-hip is compiled with --offload-compress (llama/CMakeLists.txt); without it the
# library carries every target's kernels uncompressed. Fails on an uncompressed bundle.
run: python3 .github/verify-hip-offload-compressed.py llama/src/main/natives
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-rocm-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-rocm:
name: Build Windows x86_64 ROCm/HIP (AMD)
needs: [startgate, build-webui]
# windows-2022 (MSVC 14.4x), NOT windows-2025-vs2026 (VS 2026 / MSVC 14.51): ROCm 7.1's
# HIP clang headers (__clang_hip_cmath.h) cannot overload the __host__ __device__
# isgreater/isless/... that the very new MSVC <cmath> declares via _CLANG_BUILTIN2, so the
# device-code compile fails. Upstream llama.cpp builds win-hip on windows-2022 for the same
# reason (it drives ROCm's own clang and relies on the older MSVC STL).
runs-on: windows-2022
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install ROCm/HIP (TheRock wheels)
shell: pwsh
# KEEP IN SYNC WITH UPSTREAM: mirrors llama.cpp's windows-rocm release job
# (.github/actions/windows-setup-rocm at the pinned GIT_TAG) — the same TheRock wheels as
# the Linux job, in place of the former AMD-Software-PRO-Edition HIP SDK installer. The
# venv lives under C:\TheRock\build, the path upstream uses.
env:
ROCM_VERSION: "10.0.0"
run: |
$ErrorActionPreference = "Stop"
New-Item -Path "C:\TheRock\build" -ItemType Directory -Force | Out-Null
python -m venv C:\TheRock\build\.venv
& C:\TheRock\build\.venv\Scripts\Activate.ps1
python -m pip install --upgrade pip
python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==$env:ROCM_VERSION"
if ($LASTEXITCODE -ne 0) { throw "ROCm wheel install failed with exit code $LASTEXITCODE" }
rocm-sdk init
if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" }
$rocm = (rocm-sdk path --root)
if (-not $rocm) { throw "rocm-sdk path --root returned empty - devel package may not be installed" }
$rocm = $rocm.Trim()
"HIP_PATH=$rocm" | Out-File -FilePath $env:GITHUB_ENV -Append
"HIP_DEVICE_LIB_PATH=$rocm\lib\llvm\amdgcn\bitcode" | Out-File -FilePath $env:GITHUB_ENV -Append
"HIP_PLATFORM=amd" | Out-File -FilePath $env:GITHUB_ENV -Append
"LLVM_PATH=$rocm\lib\llvm" | Out-File -FilePath $env:GITHUB_ENV -Append
(rocm-sdk path --bin).Trim() | Out-File -FilePath $env:GITHUB_PATH -Append
"C:\TheRock\build\.venv\Scripts" | Out-File -FilePath $env:GITHUB_PATH -Append
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# Upstream's compiler wiring for TheRock (clang under lib\llvm\bin, not bin\);
# -Wno-error=incompatible-pointer-types is upstream's too. Targets: every Windows target
# TheRock builds — the Linux list minus the Instinct parts, which ROCm has no Windows
# support for. Same rule as on Linux for the extras upstream omits -- here only
# gfx900/gfx906/gfx90c, since upstream's windows-rocm list already has gfx1153.
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_HIP=ON -DGPU_TARGETS=gfx900;gfx906;gfx90c;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DCMAKE_PREFIX_PATH="%HIP_PATH%" -DHIP_PATH="%HIP_PATH%" -DCMAKE_C_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_CXX_COMPILER="%HIP_PATH%\lib\llvm\bin\clang++.exe" -DCMAKE_HIP_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Verify the GPU code is compressed
# Same check as the Linux ROCm job: an uncompressed bundle means ~1 GB jllama.dll again.
shell: pwsh
run: python .github/verify-hip-offload-compressed.py llama/src/main/natives
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-rocm-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-sycl-fp16:
name: Build Linux x86_64 SYCL fp16 (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install Intel oneAPI (DPC++ + MKL)
run: |
wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null
echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list
sudo apt-get update
sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel
- name: Build libraries
shell: bash
run: |
source /opt/intel/oneapi/setvars.sh
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-sycl-fp16-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-sycl-fp32:
name: Build Linux x86_64 SYCL fp32 (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install Intel oneAPI (DPC++ + MKL)
run: |
wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null
echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list
sudo apt-get update
sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel
- name: Build libraries
shell: bash
run: |
source /opt/intel/oneapi/setvars.sh
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-sycl-fp32-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-sycl:
name: Build Windows 2025 x86_64 SYCL (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install Intel oneAPI (DPC++ + MKL + oneDNN + TBB)
shell: cmd
# Mirrors upstream llama.cpp's windows-sycl release job: extract the offline
# installer, then run its bootstrapper with the DPC++/MKL/oneDNN/TBB components.
run: |
curl -fSL -o "%RUNNER_TEMP%\oneapi.exe" "https://registrationcenter-download.intel.com/akdlm/IRC_NAS/b60765d1-2b85-4e85-86b6-cb0e9563a699/intel-deep-learning-essentials-2025.3.3.18_offline.exe"
"%RUNNER_TEMP%\oneapi.exe" -s -x -f "%RUNNER_TEMP%\oneapi_extracted" --log "%RUNNER_TEMP%\extract.log"
"%RUNNER_TEMP%\oneapi_extracted\bootstrapper.exe" -s --action install --components=intel.oneapi.win.cpp-dpcpp-common:intel.oneapi.win.mkl.devel:intel.oneapi.win.dnnl:intel.oneapi.win.tbb.devel --eula=accept -p=NEED_VS2022_INTEGRATION=0 --log-dir="%RUNNER_TEMP%"
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
run: |
call "C:\Program Files (x86)\Intel\oneAPI\setvars.bat" intel64 --force
.github\build.bat -G "Ninja Multi-Config" -DGGML_SYCL=ON -DCMAKE_C_COMPILER=cl -DCMAKE_CXX_COMPILER=icx -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-sycl-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-arm64-opencl:
name: Build Windows 11 arm64 OpenCL (Adreno)
needs: [startgate, build-webui]
# Windows-on-ARM OpenCL (Snapdragon X / Adreno). Same clang-cl + GGML_OPENMP=OFF
# toolchain as the arm64 CPU job (ggml refuses MSVC cl.exe on ARM). Ships as the
# opencl-windows-aarch64 natives jar. build_opencl_windows.bat stages the
# OpenCL headers + ICD loader before delegating to build.bat.
runs-on: windows-11-arm
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (arm64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: arm64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
run: |
.github\build_opencl_windows.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=aarch64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-opencl-windows-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-openvino:
name: Build Linux x86_64 OpenVINO (Intel)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Install OpenCL dev + Intel OpenVINO 2026.4 (archive)
run: |
# Intel's OpenVINO APT repo only publishes up to ~2025 (the /openvino/2026 path 404s), and
# 2025.x has the older ov::Allocator API that breaks ggml-openvino's template compile. So use
# the ARCHIVE — exactly what upstream llama.cpp's linux-setup-openvino action does, from the
# same URL template.
#
# KEEP IN SYNC WITH UPSTREAM. The version tracks llama.cpp's own OPENVINO_VERSION_MAJOR /
# OPENVINO_VERSION_FULL (.github/workflows/release.yml at the pinned GIT_TAG); ggml-openvino
# is developed against that pair, so lagging it is what eventually breaks the compile. Both
# OpenVINO jobs here (Linux + Windows) use the same two values — bump them together:
# major = 2026.4 full = 2026.4.0.22959.99c81491cc3
# OpenCL headers (incl. the C++ CL/cl2.hpp via opencl-clhpp-headers) come from Ubuntu's own repos.
sudo apt-get update
sudo apt-get install -y ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd
url="https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/linux/openvino_toolkit_ubuntu24_2026.4.0.22959.99c81491cc3_x86_64.tgz"
sudo mkdir -p /opt/intel/openvino
curl -fSL "$url" | sudo tar -xz --strip-components=1 -C /opt/intel/openvino
echo "OpenVINO_DIR=/opt/intel/openvino/runtime/cmake" >> "$GITHUB_ENV"
- name: Build libraries
shell: bash
run: |
source /opt/intel/openvino/setupvars.sh || true
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_OPENVINO=ON -DOpenVINO_DIR=$OpenVINO_DIR -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-openvino-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-openvino:
name: Build Windows 2025 x86_64 OpenVINO (Intel)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install OpenCL headers (vcpkg) + Intel OpenVINO 2026.4
shell: pwsh
# vcpkg's opencl port ships the full C++ headers incl. CL/cl2.hpp that OpenVINO's
# ocl_wrapper.hpp needs (the Khronos OpenCL-Headers dropped cl2.hpp) — same as upstream
# llama.cpp's windows-openvino job. OpenVINO 2026.4 matches ggml-openvino's target API.
# Keep the version in sync with the Linux OpenVINO job above (and with upstream's
# OPENVINO_VERSION_MAJOR / OPENVINO_VERSION_FULL) — see the note there.
run: |
C:\vcpkg\vcpkg install opencl:x64-windows
$url = "https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/windows/openvino_toolkit_windows_2026.4.0.22959.99c81491cc3_x86_64.zip"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\openvino.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\openvino.zip" -DestinationPath "C:\openvino" -Force
# The archive extracts into a nested versioned folder; point OpenVINO_DIR at its runtime/cmake.
$root = (Get-ChildItem "C:\openvino" -Directory | Select-Object -First 1).FullName
"OpenVINO_DIR=$root\runtime\cmake" | Out-File -FilePath $env:GITHUB_ENV -Append
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
uses: ./.github/actions/install-sccache-windows
- name: Build libraries
shell: cmd
# vcpkg toolchain file wires in the OpenCL (incl. cl2.hpp) that ggml-openvino needs.
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_OPENVINO=ON -DOpenVINO_DIR="%OpenVINO_DIR%" -DCMAKE_TOOLCHAIN_FILE=C:\vcpkg\scripts\buildsystems\vcpkg.cmake -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-openvino-windows-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
if-no-files-found: error
# ---------------------------------------------------------------------------
# CI-only jobs — no release artifact, purely for test coverage
# ---------------------------------------------------------------------------
test-cpp-linux-x86_64:
name: C++ Tests Ubuntu Latest x86_64
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Build libraries
run: |
mvn -q --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DBUILD_TESTING=ON
# Every patch has a runnable guard that reds this job if it goes missing (a link error, or
# test_utils.cpp / test_model_split.cpp / test_common_log_callback.cpp). This is the second
# line: it asserts the applier actually ran and nothing reverted the patched tree, which no
# per-patch guard covers directly. Model-free, milliseconds.
- name: Verify llama.cpp patches are applied
run: .github/verify-patches-applied.sh
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
build-macos-arm64-metal-15:
name: Build and Test macOS 15 arm64 (Metal)
needs: [startgate, build-webui]
runs-on: macos-15
env:
BUILD_JOBS: 2
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DGGML_NATIVE=OFF -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: natives-metal-macos-aarch64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Java test jobs — download release artifact, run mvn test
# ---------------------------------------------------------------------------
test-java-linux-x86_64:
name: Java Tests Ubuntu Latest x86_64
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: bash
run: bash .github/print-host-info.sh
- uses: actions/download-artifact@v8
with:
name: natives-cpu-linux-x86-64
path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/
- uses: ./.github/actions/restore-models
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Memory before tests
run: free -h
- name: Enable core dumps
run: |
ulimit -c unlimited
echo "${{ github.workspace }}/core.%e.%p" | sudo tee /proc/sys/kernel/core_pattern
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml -P jcstress test
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- uses: actions/upload-artifact@v7
if: success()
with:
name: jacoco-report
path: llama/target/site/jacoco/jacoco.xml
if-no-files-found: ignore
- name: Run PIT mutation tests
run: mvn --batch-mode --no-transfer-progress -f llama/pom.xml test-compile org.pitest:pitest-maven:mutationCoverage
- name: Extract PIT survivors
if: always()
run: |
echo "=== PIT Survived Mutations ==="
for html_file in $(find llama/target/pit-reports -name "*.html" -type f 2>/dev/null | sort); do
if grep -q "SURVIVED" "$html_file"; then
echo "Found survivors in $html_file:"
grep -B 2 -A 3 "SURVIVED" "$html_file"
echo ""
fi
done
- uses: actions/upload-artifact@v7
if: always()
with: { name: pit-reports, path: llama/target/pit-reports/ }
- name: Memory after tests
if: always()
run: free -h
# Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1).
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: bash .github/print-crash-logs.sh llama
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-linux-x86_64
path: |
${{ github.workspace }}/llama/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama/*.hprof
${{ github.workspace }}/llama/target/surefire-reports/*.dump
${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama/target/surefire-reports/*.txt
${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
# ---------------------------------------------------------------------------
# vmlens interleaving analysis — pure-Java, needs no native library or models.
# Staged to a single smoke test for now (see the `vmlens` profile in pom.xml).
# ---------------------------------------------------------------------------
vmlens:
name: Test (vmlens interleavings)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
cache: maven
- name: Test under vmlens (interleaving analysis)
# Add each new test in the `vmlens` package to this -Dtest list (surefire
# -Dtest matches simple class names, not package globs; the default suite is
# excluded from the vmlens package via pom.xml managed surefire <excludes>).
run: >-
mvn --batch-mode --no-transfer-progress -f llama/pom.xml -Pvmlens test
-Dtest=VmlensInterleavingSmokeTest,SessionStateInterleavingTest -DfailIfNoTests=false
- uses: actions/upload-artifact@v7
if: always()
with:
name: vmlens-report
path: llama/target/vmlens-report/
if-no-files-found: ignore
# The macOS and Windows Java test jobs: one reusable workflow (.github/workflows/java-tests.yml),
# called with the native build under test. The ids, names and needs stay here, so `package`,
# the release gate and every reader of this file see the same jobs as before.
test-java-macos-arm64-metal:
name: Java Tests macOS 14 arm64 (Metal)
needs: [build-macos-arm64-metal, verify-model-cache]
uses: ./.github/workflows/java-tests.yml
with:
runs-on: macos-14
natives-artifact: macos-14-metal
maven-args: -Dnet.ladenthin.llama.test.ngl=0
error-artifact: error-log-macos-14-metal
test-java-macos-arm64-no-metal:
name: Java Tests macOS 15 arm64 (no Metal)
needs: [build-macos-arm64-no-metal, verify-model-cache]
uses: ./.github/workflows/java-tests.yml
with:
runs-on: macos-15
natives-artifact: macos-15-no-metal
error-artifact: error-log-macos-15-no-metal
test-java-macos-arm64-metal-15:
name: Java Tests macOS 15 arm64 (Metal)
needs: [build-macos-arm64-metal-15, verify-model-cache]
uses: ./.github/workflows/java-tests.yml
with:
runs-on: macos-15
natives-artifact: natives-metal-macos-aarch64
error-artifact: error-log-macos-15-metal
test-java-windows-x86_64:
name: Java Tests Windows 2025 x86_64 (default / Ninja)
needs: [build-windows-x86_64, verify-model-cache]
uses: ./.github/workflows/java-tests.yml
with:
runs-on: windows-2025-vs2026
natives-artifact: natives-cpu-windows-x86-64
error-artifact: windows-output
# Java/inference validation of the MSVC-built x86_64 DLL (the analogue of
# test-java-windows-x86_64 for the default Ninja build). Loads the MSVC jllama.dll
# via JNI and runs the full model-backed suite, so both Windows generators are
# validated end-to-end before the `msvc-windows` natives jar ships.
test-java-windows-x86_64-msvc:
name: Java Tests Windows 2025 x86_64 (MSVC classifier)
needs: [build-windows-x86_64-msvc, verify-model-cache]
uses: ./.github/workflows/java-tests.yml
with:
runs-on: windows-2025-vs2026
natives-artifact: natives-msvc-windows-x86-64
error-artifact: windows-output-msvc
# ---------------------------------------------------------------------------
# Package and publish
# ---------------------------------------------------------------------------
package:
name: Package JARs
needs:
- crosscompile-linux-x86_64-cuda
- crosscompile-linux-aarch64
- build-linux-s390x
- build-linux-x86_64-vulkan
- build-linux-aarch64-vulkan
- crosscompile-android-aarch64
- crosscompile-android-x86_64
- crosscompile-android-aarch64-opencl
- build-windows-x86_64
- build-windows-x86
- build-windows-arm64
- build-windows-x86_64-msvc
- build-windows-x86-msvc
- build-windows-x86_64-cuda
- build-windows-x86_64-vulkan
- build-windows-x86_64-opencl
- build-linux-x86_64-rocm
- build-windows-x86_64-rocm
- build-linux-x86_64-sycl-fp16
- build-linux-x86_64-sycl-fp32
- build-windows-x86_64-sycl
- build-windows-arm64-opencl
- build-linux-x86_64-openvino
- build-windows-x86_64-openvino
- test-cpp-linux-x86_64
- build-macos-arm64-metal-15
- test-java-linux-x86_64
- test-java-macos-arm64-metal
- test-java-macos-arm64-no-metal
- test-java-macos-arm64-metal-15
- test-java-windows-x86_64
- test-java-windows-x86_64-msvc
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
# Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one
# subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact
# holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is
# claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can
# byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS
# dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named
# outside the glob for that reason.
- uses: actions/download-artifact@v8
with:
pattern: "natives-*"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the natives tree (checked per artifact)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/"
# Runtime dependencies of every shipped native library, read from the file itself (ELF
# DT_NEEDED, PE imports incl. Windows arm64, Mach-O load commands). The CPU builds must
# match an exact per-<OS>/<ARCH>/<backend> allowlist -- a new dependency is a load failure
# on every machine that lacks it. The GPU backends legitimately need their vendor runtime,
# so there only the denylist applies. The case this was written for: ggml-rpc's RDMA
# transport, which upstream enables whenever the build host has libibverbs/librdma
# (llama/CMakeLists.txt forces it off).
- name: Verify native runtime dependencies
run: |
python3 .github/verify-native-deps.py llama/src/main/natives
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Build JARs
# `assembly` additionally produces the fat jar-with-dependencies uber JAR
# (llama-<version>-jar-with-dependencies.jar: library classes + Java runtime deps +
# default-platform native libs in one drop-on-classpath JAR, runnable via its
# ServerLauncher Main-Class). It lands in target/ and is uploaded in the `llama-jars`
# artifact below - never a Maven Central asset; the `package-fatjars` job downstream
# combines it with the natives jars into the GitHub-Release fat-jar assets.
# `natives` builds one jar per <backend>-<os>-<arch> from the tree merged above; its
# enforcer rule fails the build when any of them is missing.
run: mvn --batch-mode --no-transfer-progress -P release,natives,assembly -Dmaven.test.skip=true -Dgpg.skip=true package
# Class-file floor, checked on every jar this job just built (the classes jar, the
# natives jars and the default fat jar). Production code targets Java 8, so anything a
# consumer's JVM can load must be major 52 or lower: a single Java 11 class kills
# the process with UnsupportedClassVersionError before any of our code runs, which
# is exactly what shipped when logback's LogbackServiceProvider was the binding.
# module-info.class and META-INF/versions/** are skipped unconditionally because a
# classpath JVM never loads them. Kept byte-identical across all four sibling repos.
- name: Verify Java 8 bytecode (no class newer than major 52)
run: .github/verify-bytecode-version.sh --max-major 52 llama/target
# The jars as a consumer uses them: the classes jar, its runtime dependencies and all
# natives jars at once, on the classpath and on the module path. Proves the directories
# never collide and each natives jar is its own automatic module; on this GPU-less runner the
# loader normally falls through the GPU backends to the CPU library.
- name: Load the natives jars together (classpath and module path)
run: |
mvn -B -q --no-transfer-progress -f llama/pom.xml dependency:build-classpath \
-DincludeScope=runtime -Dmdep.outputFile=runtime-classpath.txt
bash .github/smoke-natives-jars.sh llama/target "$(cat llama/runtime-classpath.txt)"
- name: Upload JARs
uses: actions/upload-artifact@v7
with:
name: llama-jars
path: llama/target/*.jar
package-fatjars:
name: Package all-backends fat jars
needs: [package]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-jars
path: jars/
# One self-contained multi-backend server fat jar per OS/arch
# (llama-<version>-all-<os>-<arch>-jar-with-dependencies.jar): the default
# jar-with-dependencies reduced to that OS/arch plus every natives jar of it; each
# backend keeps its own directory, which LlamaLoader tries in its fixed priority
# order (first loadable backend wins, CPU fallback).
# These are GitHub-Release download assets ONLY - never deployed to Maven
# Central (the deploy jobs run without the `assembly` profile and are untouched).
# The script takes the natives jars from .github/natives.csv and fails loud on any
# mismatch with the built jars or on a natives jar holding anything but its own
# directory, so a new natives jar cannot be silently skipped.
- name: Assemble all-backends fat jars
run: bash .github/package-fatjars.sh jars fatjars
- name: Upload fat jars
uses: actions/upload-artifact@v7
with:
name: llama-fatjars
path: fatjars/
compression-level: 0 # jars are already deflated
retention-days: 7 # multi-GB artifact; release jobs consume it within the same run
if-no-files-found: error
# Small single-jar artifacts, llama-fatjar-smoke-<target>, one per all-backends fat jar, so the
# smoke-fatjar matrix does not download the multi-GB set. check-natives.py holds these, the
# matrix rows and the targets derived from natives.csv to the same list.
- name: Upload linux-x86-64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-linux-x86-64
path: fatjars/llama-*-all-linux-x86-64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
- name: Upload linux-aarch64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-linux-aarch64
path: fatjars/llama-*-all-linux-aarch64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
- name: Upload windows-x86-64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-windows-x86-64
path: fatjars/llama-*-all-windows-x86-64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
- name: Upload windows-aarch64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-windows-aarch64
path: fatjars/llama-*-all-windows-aarch64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
# Every all-backends fat jar, launched on a GPU-less runner of its OS/arch: every GPU backend
# must fail its load cleanly (missing vendor runtimes) and the server must come up on the CPU
# backend -- backend probing, per-backend extraction and the fallback chain end-to-end through a
# real `java -jar`, then /health + /v1/chat/completions. One row per fat-jar target;
# check-natives.py fails when the rows differ from the targets natives.csv derives, so a new
# all-<os>-<arch> jar cannot ship unlaunched -- which is how the two aarch64 jars once did:
# built, signed and attached to every release while nothing ran them. That breaks the rule of
# workspace/policies/fat-jar-release-assets.md ("No release asset is attached that CI has not
# run"), which exists because a corrupt macOS dylib shipped in three releases under a green
# pipeline. fail-fast is off: one platform's failure must not hide another's.
smoke-fatjar:
name: Smoke test all-backends fat jar (${{ matrix.target }})
needs: [package-fatjars, verify-model-cache]
strategy:
fail-fast: false
matrix:
include:
- { target: linux-x86-64, runner: ubuntu-latest }
- { target: linux-aarch64, runner: ubuntu-24.04-arm }
- { target: windows-x86-64, runner: windows-2025-vs2026 }
- { target: windows-aarch64, runner: windows-11-arm }
runs-on: ${{ matrix.runner }}
env:
JAR_GLOB: llama-*-all-${{ matrix.target }}-jar-with-dependencies.jar
steps:
- uses: actions/checkout@v7
with:
persist-credentials: false
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-${{ matrix.target }}
path: fatjars/
- uses: ./.github/actions/restore-models
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
# The Java 8 floor, re-checked on the ASSEMBLED release asset rather than the module jars the
# `package` job verified: package-fatjars rewrites the zip per OS/arch (drops the other
# platforms, adds the backend trees), and this is the artifact users download. The script is
# bash-only, so the Linux rows check; the Windows jars are assembled from the same classes.
- name: Verify Java 8 bytecode (no class newer than major 52)
if: runner.os == 'Linux'
run: .github/verify-bytecode-version.sh --max-major 52 fatjars
- name: Run fat-jar server smoke test
if: runner.os != 'Windows'
run: .github/smoke-test-fatjar.sh fatjars "$JAR_GLOB" "models/${DRAFT_MODEL_NAME}"
- name: Run fat-jar server smoke test
if: runner.os == 'Windows'
shell: pwsh
run: .github/smoke-test-fatjar.ps1 -JarDir fatjars -JarGlob $env:JAR_GLOB -Model "models/$env:DRAFT_MODEL_NAME"
# RPC over two JVMs from the same release asset: RpcServer in one, the default NativeServer
# with --rpc in the other. Requires a chat completion, the model buffer on the RPC endpoint in
# the load log (the layers really went over RPC), an accepted client on the server, and a clean
# non-SIGABRT failure naming the endpoint when --rpc names a server nobody runs.
- name: Run fat-jar RPC smoke test (two JVMs)
if: matrix.target == 'linux-x86-64'
run: .github/smoke-rpc-fatjar.sh fatjars "$JAR_GLOB" "models/${DRAFT_MODEL_NAME}"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: fatjar-smoke-${{ matrix.target }}-logs
path: |
server-out.log
server-err.log
rpc-server.log
rpc-client-out.log
rpc-client-err.log
rpc-unreachable.log
if-no-files-found: warn
# The agent release asset, launched the way the README tells a user to: `java -jar` on the agent
# jar lying next to the real all-backends Linux fat jar, so the core is found only through the
# agent manifest's Class-Path. Proves the two assets fit together (matching version in the file
# names, nothing missing on either side), that the agent jar carries no core, and that a
# one-shot answer and a read_file tool round work through the shipped jars.
smoke-agent-linux:
name: Smoke test the agent release jar (Linux)
needs: [test-java-llama-atmosphere-agent, package-fatjars, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: agent-assets/
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-linux-x86-64
path: agent-assets/
- uses: ./.github/actions/restore-models
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
# The agent is Java 21 (Atmosphere's floor), unlike the Java 8 core: its ceiling is 65.
# One waiver: JLine ships its FFM terminal provider (org/jline/terminal/impl/ffm, 25 classes)
# as Java 22 bytecode. It is discovered through META-INF/jline/providers/ffm, not loaded
# eagerly: on Java 21 JLine picks its JNI provider, and a forced FFM provider falls back to a
# dumb terminal instead of failing (verified on 21.0.10). Only that package is waived, so
# any other class above 65 still fails the gate.
- name: Verify Java 21 bytecode (no class newer than major 65)
run: >
.github/verify-bytecode-version.sh --max-major 65
--allow 'llama-atmosphere-agent-*:org/jline/terminal/impl/ffm/*'
agent-assets/llama-atmosphere-agent-*-jar-with-dependencies.jar
- name: Run the agent release-jar smoke test
run: .github/smoke-agent-jar.sh agent-assets "models/${TOOL_MODEL_NAME}"
- name: Upload agent logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: agent-smoke-linux-logs
path: agent-*.log
if-no-files-found: warn
# ---------------------------------------------------------------------------
# macOS member of the cross-repo "no release asset is attached that CI has not run" convention
# (workspace/policies/fat-jar-release-assets.md; BitcoinAddressFinder and srcmorph run the shared
# .github/smoke-fatjar-cli.sh in the same job shape). It closes the gap that let a corrupt dylib
# ship: the three macOS Java test jobs each test the dylib THEIR OWN build job produced, so until
# now nothing ever loaded the one that goes into the published jar — see CLAUDE.md, "macOS arm64:
# three build jobs, one shipped dylib".
#
# macOS-specific in two ways. It targets the DEFAULT fat jar from `llama-jars`, because there is
# no `all-macos-*` fat jar to target: macOS has no GPU classifier (Metal ships in the default
# jar), so package-fatjars builds no macOS variant. And it asserts native loadability rather than
# a CLI exit code — this jar's Main-Class is a server that never returns, and `codesign --strict`
# is what actually detects a dylib assembled from two builds. Deliberately model-free (no GGUF,
# no cache restore, ~1 min): a full model-backed macOS server smoke would be strictly more, but
# this catches the failure class that shipped and is cheap enough to always run.
# ---------------------------------------------------------------------------
smoke-fatjar-macos:
name: Smoke test packaged natives (macOS)
needs: [package]
runs-on: macos-15
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-jars
path: jars/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version-file: .java-version
- name: Run packaged-native smoke test
run: .github/smoke-native-macos.sh jars 'llama-*-jar-with-dependencies.jar'
report:
name: Report
needs: [package]
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with: { java-version-file: .java-version, distribution: temurin }
- uses: actions/download-artifact@v8
with: { name: jacoco-report, path: target/site/jacoco/ }
continue-on-error: true
# Submits the dependency graph to GitHub. Informational: it says nothing about whether the
# artifacts are correct, but it sits in the `report` job, which the release path needs -- so
# without this flag a third-party action having a bad day can block a publish.
- uses: advanced-security/maven-dependency-submission-action@v6
continue-on-error: true
- name: Coveralls
uses: coverallsapp/github-action@v2
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
file: target/site/jacoco/jacoco.xml
format: jacoco
continue-on-error: true
- name: Codecov
uses: codecov/codecov-action@v7
with:
token: ${{ secrets.CODECOV_TOKEN }}
files: target/site/jacoco/jacoco.xml
continue-on-error: true
check-snapshot:
name: "Check: main branch / SNAPSHOT"
needs: [report]
runs-on: ubuntu-latest
if: >-
(github.event_name == 'push' && github.ref == 'refs/heads/main') ||
(github.event_name == 'workflow_dispatch' && !startsWith(github.ref, 'refs/tags/v'))
steps:
- name: Confirm snapshot ref
run: echo "Confirmed on snapshot ref ${{ github.ref }}"
check-tag:
name: "Check: v* tag"
needs: [report]
runs-on: ubuntu-latest
if: startsWith(github.ref, 'refs/tags/v')
steps:
- name: Confirm tag ref
run: echo "Confirmed on tag ${{ github.ref }}"
publish-snapshot:
name: Publish Snapshot to Central
needs: [check-snapshot, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar, smoke-fatjar-macos, smoke-agent-linux, vmlens, shared-files]
if: needs.check-snapshot.result == 'success' && inputs.publish_to_central
runs-on: ubuntu-latest
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
# Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one
# subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact
# holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is
# claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can
# byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS
# dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named
# outside the glob for that reason.
- uses: actions/download-artifact@v8
with:
pattern: "natives-*"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the natives tree (checked per artifact)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/"
- name: Set up Maven Central Repository
uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: 'temurin'
server-id: central
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
gpg-passphrase: MAVEN_GPG_PASSPHRASE
- name: Guard - require a -SNAPSHOT version
shell: bash
run: |
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
echo "Resolved project version: $VERSION"
case "$VERSION" in
*-SNAPSHOT) echo "OK: -SNAPSHOT version, continuing snapshot deploy." ;;
*) echo "::error::Refusing to publish non-SNAPSHOT version '$VERSION' from the snapshot job. Snapshot publishing requires a -SNAPSHOT version; releases go through the v* tag path."; exit 1 ;;
esac
# One reactor deploy publishes all five Maven artifacts at the same version:
# net.ladenthin:llama-parent (the pom), :llama (the classes jar + natives jars),
# :llama-langchain4j, :llama-kotlin and :llama-platform (a pom). The `release` profile (GPG + Central
# Publishing) is inherited from the parent, so every module — including the
# parent pom — is signed. The Android AARs are published by the separate
# Gradle step below (Maven cannot deploy <packaging>aar</packaging>).
# Informational only (nothing depends on it): logs the effective POM with the same
# profile set as the deploy below, so the resolved central-publishing configuration
# (waitUntil/waitMaxTime etc.) is visible for debugging.
- name: Show effective POM (debug)
run: mvn --batch-mode --no-transfer-progress -P release,natives help:effective-pom
- name: Publish snapshot (reactor - parent + llama + llama-langchain4j + llama-kotlin + llama-platform)
run: mvn --batch-mode --no-transfer-progress -P release,natives -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# The agent's thin jar + pom (not a reactor module, so a deploy of its own). It resolves the
# core and the natives jars llama-platform names from the local repository, where the reactor
# deploy above has just installed them; check-natives.py holds its version to the reactor's.
- name: Publish snapshot (llama-atmosphere-agent)
run: mvn --batch-mode --no-transfer-progress -f llama-atmosphere-agent/pom.xml -P release -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# Android AARs (llama-android, llama-android-opencl): assembled and published by
# the plain-Gradle build in llama-android/ — Maven cannot deploy
# <packaging>aar</packaging>. Natives are already on disk from the artifact
# downloads above; the core jar was just built by the reactor deploy.
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Publish Android AAR snapshots (llama-android + llama-android-opencl)
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/
gradle -p llama-android publishAllPublicationsToCentralSnapshotsRepository
env:
CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
# Runs even when the deploy step failed: a Central publish-poll timeout reds the
# job *after* the bundle was uploaded (and typically published server-side), while
# the signed jars + .asc files already exist in target/ (signing happens at
# verify). Collecting on failure lets the github-snapshot job still attach them.
- name: Collect signed artifacts
if: ${{ !cancelled() }}
run: |
mkdir -p signed-snapshot-assets
cp llama/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-android/build/aar/*.aar signed-snapshot-assets/ 2>/dev/null || true
- uses: actions/upload-artifact@v7
if: ${{ !cancelled() }}
with:
name: signed-snapshot-assets
path: signed-snapshot-assets/
github-snapshot:
name: Update Snapshot Pre-release on GitHub
needs: [publish-snapshot, package-fatjars, test-java-llama-atmosphere-agent]
# Also runs when publish-snapshot FAILED (not when skipped/cancelled): a Central
# publish-poll timeout reds that job after the artifacts were already uploaded —
# the GitHub pre-release assets must not be lost in that case.
if: ${{ !cancelled() && (needs.publish-snapshot.result == 'success' || needs.publish-snapshot.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }}
runs-on: ubuntu-latest
# maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is
# scoped to this environment) for signing the fat jars below. This environment has
# no approval gate (the standalone verify-signing-key jobs use it on every run), so
# declaring it here does not block the release.
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: signed-snapshot-assets
path: snapshot-assets/
# All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub
# download assets only, deliberately NOT deployed to Maven Central. Downloaded
# into the same directory so the one upload glob below picks everything up
# (fat-jar names are disjoint from the signed thin-jar names).
- uses: actions/download-artifact@v8
with:
name: llama-fatjars
path: snapshot-assets/
# The agent jar (+ sha256) — built without the core, run next to one of the fat jars above.
# Same directory, so sign-fatjars.sh signs it and the upload glob attaches it.
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: snapshot-assets/
# GPG-sign the fat jars so each carries a detached .asc signature alongside its
# .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at
# deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity)
# are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because
# only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md.
- name: GPG-sign the fat jars (.asc)
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
run: bash .github/sign-fatjars.sh snapshot-assets
- name: Report unsigned assets (does not block the upload)
# Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even
# when their publish job failed, because a Central publish-poll timeout must not cost the
# GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at
# all. Refusing to attach on a signing failure would defeat exactly that. So annotate here,
# upload regardless, and fail the job afterwards. Assets always land; an unsigned release is
# still loudly red rather than quietly wrong.
# See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red".
id: signatures
run: |
set -uo pipefail
dir="snapshot-assets"
jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort)
if [ -z "$jars" ]; then
echo "::error::no jars in $dir -- the collection step produced nothing"
echo "missing=-1" >> "$GITHUB_OUTPUT"
exit 0
fi
missing=0
for jar in $jars; do
if [ ! -e "$jar.asc" ]; then
echo "::error::unsigned: $(basename "$jar") has no detached .asc"
missing=$((missing + 1))
fi
done
echo "missing=$missing" >> "$GITHUB_OUTPUT"
[ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed"
exit 0
- name: Update snapshot pre-release
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
gh release view snapshot --repo ${{ github.repository }} 2>/dev/null \
|| gh release create snapshot \
--repo ${{ github.repository }} \
--prerelease \
--title "Snapshot (latest)" \
--notes "Latest snapshot build from the main branch."
gh release upload snapshot snapshot-assets/* \
--repo ${{ github.repository }} \
--clobber
- name: Fail if anything was attached unsigned
# After the upload on purpose: the assets must exist even when the signature does not.
if: ${{ always() && steps.signatures.outputs.missing != '0' }}
run: |
echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)"
exit 1
publish-release:
name: Publish Release to Central
if: needs.check-tag.result == 'success' && inputs.publish_to_central
needs: [check-tag, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar, smoke-fatjar-macos, smoke-agent-linux, vmlens, shared-files]
runs-on: ubuntu-latest
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
# Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one
# subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact
# holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is
# claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can
# byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS
# dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named
# outside the glob for that reason.
- uses: actions/download-artifact@v8
with:
pattern: "natives-*"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the natives tree (checked per artifact)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/"
- name: Set up Maven Central Repository
uses: actions/setup-java@v6
with:
java-version-file: .java-version
distribution: 'temurin'
server-id: central
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
gpg-passphrase: MAVEN_GPG_PASSPHRASE
# One reactor deploy publishes all five Maven artifacts at the same version:
# net.ladenthin:llama-parent (the pom), :llama (the classes jar + natives jars),
# :llama-langchain4j, :llama-kotlin and :llama-platform (a pom). The `release` profile (GPG + Central Publishing) is inherited
# from the parent, so every module — including the parent pom — is signed.
# Informational only (nothing depends on it): logs the effective POM with the same
# profile set as the deploy below, so the resolved central-publishing configuration
# (waitUntil/waitMaxTime etc.) is visible for debugging.
- name: Show effective POM (debug)
run: mvn --batch-mode --no-transfer-progress -P release,natives help:effective-pom
- name: Publish release (reactor - parent + llama + llama-langchain4j + llama-kotlin + llama-platform)
run: mvn --batch-mode --no-transfer-progress -P release,natives -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# The agent's thin jar + pom (not a reactor module, so a deploy of its own). It resolves the
# core and the natives jars llama-platform names from the local repository, where the reactor
# deploy above has just installed them; check-natives.py holds its version to the reactor's.
- name: Publish release (llama-atmosphere-agent)
run: mvn --batch-mode --no-transfer-progress -f llama-atmosphere-agent/pom.xml -P release -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# Android AARs (llama-android, llama-android-opencl): Maven cannot deploy
# <packaging>aar</packaging>, so the plain-Gradle build signs + publishes them
# into a local Maven-layout staging repo, which is zipped into a Central
# Portal bundle and uploaded via the Publisher API (publishingType=AUTOMATIC:
# the deployment publishes as soon as portal validation passes). Natives are
# already on disk from the artifact downloads above; the core jar was just
# built by the reactor deploy.
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: ${{ env.GRADLE_VERSION }}
- name: Publish Android AAR release bundle to Central Portal
shell: bash
run: |
set -euo pipefail
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/
cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/
gradle -p llama-android publishAllPublicationsToStagingRepository
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
( cd llama-android/build/staging-repo && zip -r -q ../central-bundle.zip net -x "*maven-metadata*" )
TOKEN=$(printf "%s:%s" "$CENTRAL_USERNAME" "$CENTRAL_PASSWORD" | base64 -w0)
curl --fail-with-body -X POST \
-H "Authorization: Bearer $TOKEN" \
-F "bundle=@llama-android/build/central-bundle.zip" \
"https://central.sonatype.com/api/v1/publisher/upload?publishingType=AUTOMATIC&name=llama-android-$VERSION"
env:
CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
# Runs even when the deploy step failed: a Central publish-poll timeout reds the
# job *after* the bundle was uploaded (and typically published server-side), while
# the signed jars + .asc files already exist in target/ (signing happens at
# verify). Collecting on failure lets the github-release-signed job still attach them.
- name: Collect signed artifacts
if: ${{ !cancelled() }}
run: |
mkdir -p signed-release-assets
cp llama/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-android/build/aar/*.aar signed-release-assets/ 2>/dev/null || true
- uses: actions/upload-artifact@v7
if: ${{ !cancelled() }}
with:
name: signed-release-assets
path: signed-release-assets/
github-release-signed:
name: Attach Signed Binaries to GitHub Release
needs: [publish-release, package-fatjars, test-java-llama-atmosphere-agent]
# Also runs when publish-release FAILED (not when skipped/cancelled): a Central
# publish-poll timeout reds that job after the artifacts were already uploaded —
# the GitHub release assets must not be lost in that case.
if: ${{ !cancelled() && (needs.publish-release.result == 'success' || needs.publish-release.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }}
runs-on: ubuntu-latest
# maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is
# scoped to this environment) for signing the fat jars below. This environment has
# no approval gate (the standalone verify-signing-key jobs use it on every run), so
# declaring it here does not block the release.
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: signed-release-assets
path: release-assets/
# All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub
# download assets only, deliberately NOT deployed to Maven Central. Downloaded
# into the same directory so the one upload glob below picks everything up
# (fat-jar names are disjoint from the signed thin-jar names).
- uses: actions/download-artifact@v8
with:
name: llama-fatjars
path: release-assets/
# The agent jar (+ sha256) — built without the core, run next to one of the fat jars above.
# Same directory, so sign-fatjars.sh signs it and the upload glob attaches it.
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: release-assets/
# GPG-sign the fat jars so each carries a detached .asc signature alongside its
# .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at
# deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity)
# are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because
# only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md.
- name: GPG-sign the fat jars (.asc)
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
run: bash .github/sign-fatjars.sh release-assets
- name: Report unsigned assets (does not block the upload)
# Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even
# when their publish job failed, because a Central publish-poll timeout must not cost the
# GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at
# all. Refusing to attach on a signing failure would defeat exactly that. So annotate here,
# upload regardless, and fail the job afterwards. Assets always land; an unsigned release is
# still loudly red rather than quietly wrong.
# See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red".
id: signatures
run: |
set -uo pipefail
dir="release-assets"
jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort)
if [ -z "$jars" ]; then
echo "::error::no jars in $dir -- the collection step produced nothing"
echo "missing=-1" >> "$GITHUB_OUTPUT"
exit 0
fi
missing=0
for jar in $jars; do
if [ ! -e "$jar.asc" ]; then
echo "::error::unsigned: $(basename "$jar") has no detached .asc"
missing=$((missing + 1))
fi
done
echo "missing=$missing" >> "$GITHUB_OUTPUT"
[ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed"
exit 0
- name: Upload release assets
uses: softprops/action-gh-release@v3
with:
files: release-assets/*
- name: Fail if anything was attached unsigned
# After the upload on purpose: the assets must exist even when the signature does not.
if: ${{ always() && steps.signatures.outputs.missing != '0' }}
run: |
echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)"
exit 1