Skip to content

Publish

Publish #1079

Workflow file for this run

# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
# SPDX-FileCopyrightText: 2023-2025 Konstantin Herud
#
# SPDX-License-Identifier: MIT
name: Publish
on:
push:
branches: [ main ]
tags: ['v*']
pull_request:
workflow_dispatch:
inputs:
publish_to_central:
description: "Deploy to Maven Central (snapshot if -SNAPSHOT, release if a vX.Y.Z tag)"
type: boolean
default: false
use_cache:
description: "Use the shared sccache/Depot compiler cache (faster incremental builds)"
type: boolean
default: true
env:
JAVA_VERSION: '21'
# Model DOWNLOAD URLS live in .github/models.csv (the single source of truth for the
# CI model set; the model cache key is derived from that file's hash). The *_NAME
# vars below are consumer-side wiring only (-Dnet.ladenthin.llama.* test properties)
# and must match the filename column of models.csv.
MODEL_NAME: "codellama-7b.Q2_K.gguf"
RERANKING_MODEL_NAME: "jina-reranker-v1-tiny-en-Q4_0.gguf"
DRAFT_MODEL_NAME: "AMD-Llama-135m-code.Q2_K.gguf"
REASONING_MODEL_NAME: "Qwen3-0.6B-Q4_K_M.gguf"
TOOL_MODEL_NAME: "Qwen2.5-1.5B-Instruct-Q4_K_M.gguf"
NOMIC_EMBED_MODEL_NAME: "nomic-embed-text-v1.5.f16.gguf"
# Vision model + mmproj for MultimodalIntegrationTest.
# SmolVLM-500M is the smallest community vision GGUF that loads reliably
# under the upstream mtmd pipeline.
VISION_MODEL_NAME: "SmolVLM-500M-Instruct-Q8_0.gguf"
VISION_MMPROJ_NAME: "mmproj-SmolVLM-500M-Instruct-Q8_0.gguf"
# Qwen3-TTS backbone + mmproj for TtsIntegrationTest (the OuteTTS+WavTokenizer pair this replaced
# was retired when upstream #26254 replaced the whole TTS pipeline — see
# docs/history/llama-cpp-breaking-changes.md, b10269-b10270). Smallest available quants:
# Q4_K_M backbone (0.96 GiB), Q8_0 mmproj (0.42 GiB; no smaller mmproj quant is published).
TTS_MODEL_NAME: "Qwen3-TTS-12Hz-1.7B-Base-Q4_K_M.gguf"
TTS_MMPROJ_NAME: "mmproj-Qwen3-TTS-12Hz-1.7B-Base-Q8_0.gguf"
# Fine-tuning fixture for LlamaTrainerIntegrationTest. stories260K is 1.19 MB and, unlike every
# other model here, is **F32** — which is the only reason it works: llama_set_param silently skips
# any tensor that is not GGML_TYPE_F32, so a quantized model would "train" nothing but the norm
# weights and still write an output GGUF that the test's assertions accept. Its small trained
# context also suits the test's short corpus; the test pins nCtx itself so that is belt-and-braces.
TRAIN_MODEL_NAME: "stories260K.gguf"
# Test image used by MultimodalIntegrationTest is committed to the repo
# at src/test/resources/images/test-image.jpg (see the README in that
# directory for licensing). No download step is needed; CI just points
# mvn test at the committed path.
VISION_IMAGE_PATH: "llama/src/test/resources/images/test-image.jpg"
# Supersede an in-flight run when a PR branch is pushed again.
#
# Without this every push starts a full parallel pipeline and the older ones keep
# draining -- four were live at once during one session, which makes "what is CI
# saying right now" genuinely ambiguous and wastes a lot of runner time on results
# nobody will read.
#
# cancel-in-progress is deliberately scoped to pull_request ONLY. A push to main or
# to a v* tag is a release path: cancelling one midway could leave a partially
# published set of artifacts.
#
# cancel-in-progress: false is NOT sufficient on its own to protect a release run.
# GitHub cancels a *pending* run whenever a newer run joins the same group behind an
# in-progress one -- that rule is independent of cancel-in-progress. So with a plain
# `workflow-ref` group, a queued `publish_to_central` dispatch on main could be
# silently dropped by a later push to main, both sharing `Publish-refs/heads/main`.
# Giving every non-PR run its own group (via the unique run_id) means such a run is
# never queued behind a sibling and therefore can never be cancelled, while PR runs
# still share a group per ref and supersede each other as intended.
#
# One-time effect when this expression changes: GitHub reads `concurrency` from the
# workflow file at each run's own ref, so a run started before the change sits in the
# old group and a run started after it sits in the new one. They are different groups,
# so the new push does NOT supersede the in-flight old run -- exactly once, on the
# commit that lands this. It self-heals from the next push on. Expect the same overlap
# when porting this to a sibling repo; it is not a sign the expression is wrong.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name == 'pull_request' && 'pr' || github.run_id }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
jobs:
# ---------------------------------------------------------------------------
# Start gate — single cancellable abort window before the pipeline starts.
# The wait duration lives in the `startgate` GitHub Environment (Settings →
# Environments → startgate → Wait timer).
# ---------------------------------------------------------------------------
startgate:
name: Start gate (abort window)
runs-on: ubuntu-latest
environment: startgate
steps:
- run: echo "Start gate elapsed — proceeding with pipeline."
# ---------------------------------------------------------------------------
# GPG signing-key preflight (standalone, no `needs:` — runs in parallel at the
# very start on every trigger). Reproduces what maven-gpg-plugin does at deploy
# time so a bad/expired key or wrong passphrase is caught in ~20s instead of
# failing the publish stage. Declares `environment: maven-central` so it reads
# the SAME GPG_PRIVATE_KEY / GPG_PASSPHRASE secret the publish jobs use.
#
# It is EXPECTED to go RED on refs where the secret is not delivered — fork PRs
# and other contributors' branches (secrets are withheld there). That red is
# the intended signal: "this ref cannot sign a release", not a regression.
#
# SECURITY: this job NEVER prints secret material. It imports the key into an
# ephemeral keyring, prints only PUBLIC key metadata (key id, fingerprint,
# owner UID, algorithm, created/expiry — all of which live on public
# keyservers), and validates the passphrase by producing + verifying a
# throwaway signature. The passphrase is passed on fd 3 (never argv, never a
# log line), `set -x` is deliberately never enabled, and the passphrase is
# additionally `::add-mask::`ed.
# ---------------------------------------------------------------------------
verify-signing-key:
name: Verify GPG signing key (no secrets printed)
runs-on: ubuntu-latest
environment: maven-central
steps:
- name: Import key + run sign/verify self-test (prints only PUBLIC metadata)
shell: bash
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
run: |
set -euo pipefail # NOTE: deliberately NO `set -x` — it would echo the passphrase.
if [ -z "${GPG_PRIVATE_KEY:-}" ]; then
echo "::error::GPG_PRIVATE_KEY is empty for this run. Either the secret is not set, or it is scoped to a different environment/branch than 'maven-central' on this ref. Nothing to verify."
exit 1
fi
# Defensive: even though we never print it, register the passphrase as a
# masked value so any accidental echo downstream is redacted.
if [ -n "${GPG_PASSPHRASE:-}" ]; then echo "::add-mask::${GPG_PASSPHRASE}"; fi
echo "gpg: $(gpg --version | head -n1)"
# Ephemeral, private keyring; removed on exit.
export GNUPGHOME="$(mktemp -d)"
chmod 700 "$GNUPGHOME"
cleanup() { gpgconf --kill gpg-agent >/dev/null 2>&1 || true; rm -rf "$GNUPGHOME"; }
trap cleanup EXIT
echo "== Import private key into an ephemeral keyring (key via stdin, never argv) =="
printf '%s\n' "$GPG_PRIVATE_KEY" | gpg --batch --import
COLONS="$(gpg --list-secret-keys --with-colons --fixed-list-mode)"
SECCOUNT="$(printf '%s\n' "$COLONS" | awk -F: '$1=="sec"{n++} END{print n+0}')"
echo "Secret keys imported: $SECCOUNT"
if [ "$SECCOUNT" -lt 1 ]; then
echo "::error::No secret key was imported — GPG_PRIVATE_KEY is not a valid armored secret key (check that the secret contains the full -----BEGIN PGP PRIVATE KEY BLOCK----- with intact newlines)."
exit 1
fi
KEYID="$(printf '%s\n' "$COLONS" | awk -F: '$1=="sec"{print $5; exit}')"
ALGO="$(printf '%s\n' "$COLONS" | awk -F: '$1=="sec"{print $4; exit}')"
CREATED="$(printf '%s\n' "$COLONS"| awk -F: '$1=="sec"{print $6; exit}')"
EXPIRES="$(printf '%s\n' "$COLONS"| awk -F: '$1=="sec"{print $7; exit}')"
FPR="$(printf '%s\n' "$COLONS" | awk -F: '$1=="fpr"{print $10; exit}')"
echo "== PUBLIC key metadata =="
echo " Key ID (long): $KEYID"
echo " Fingerprint: $FPR"
echo " Pubkey algo id: $ALGO"
echo " Created (UTC): $(date -u -d "@$CREATED" 2>/dev/null || echo "$CREATED")"
echo " Owner UID(s):"
printf '%s\n' "$COLONS" | awk -F: '$1=="uid"{print " - " $10}'
# --- Expiration gate ---
NOW="$(date -u +%s)"
if [ -n "$EXPIRES" ]; then
echo " Expires (UTC): $(date -u -d "@$EXPIRES" 2>/dev/null || echo "$EXPIRES")"
if [ "$EXPIRES" -le "$NOW" ]; then
echo "::error::Signing key is EXPIRED — Maven Central will reject its signatures. Extend the key's expiry and update the GPG_PRIVATE_KEY secret."
exit 1
fi
echo " Days to expiry: $(( (EXPIRES - NOW) / 86400 ))"
if [ "$(( (EXPIRES - NOW) / 86400 ))" -lt 30 ]; then
echo "::warning::Signing key expires in under 30 days — plan to rotate it."
fi
else
echo " Expires (UTC): never"
fi
# --- Signing-capability gate ---
if printf '%s\n' "$COLONS" | awk -F: '($1=="sec"||$1=="ssb"){print $12}' | grep -q 's'; then
echo " Signing capability: present"
else
echo "::error::No signing-capable (sub)key found — this key cannot produce release signatures."
exit 1
fi
# --- Passphrase unlock + sign + verify roundtrip (the exact failure mode) ---
# Passphrase on fd 3 only. Payload is a throwaway nonce; only the signature's
# validity (exit codes) matters — no secret is ever emitted.
echo "== Passphrase unlock + detached-sign + verify self-test =="
WORK="$(mktemp -d)"
printf '%s' "ai-index signing-selftest" > "$WORK/payload.txt"
gpg --batch --yes --pinentry-mode loopback --passphrase-fd 3 \
--local-user "$KEYID" \
--detach-sign --armor --output "$WORK/payload.txt.asc" "$WORK/payload.txt" \
3<<<"${GPG_PASSPHRASE:-}"
echo " Signature produced: $(wc -c < "$WORK/payload.txt.asc") armored bytes"
gpg --batch --verify "$WORK/payload.txt.asc" "$WORK/payload.txt"
rm -rf "$WORK"
echo "RESULT: OK — key imports, is not expired, is signing-capable, and the passphrase successfully unlocked it to produce a VALID signature. maven-gpg-plugin will be able to sign with this key/passphrase."
# ---------------------------------------------------------------------------
# GPG signing-key preflight — GRADLE / BouncyCastle path.
# Companion to the `verify-signing-key` (gpg) job above: that one mirrors
# maven-gpg-plugin (how the Maven artifacts are signed); this one drives
# Gradle's `signing` plugin + `useInMemoryPgpKeys` (BouncyCastle) — the path any
# Gradle-based publish (e.g. an Android AAR) uses to sign. BouncyCastle is a
# STRICTER parser of the armored key than gpg, so it catches key/format problems
# gpg tolerates (e.g. the primary-vs-signing-subkey null-PGPPrivateKey issue).
# It signs a throwaway project (.github/signing-selftest/) — no repo build is
# involved — so this job is IDENTICAL across the sibling repos and validates the
# release key via the Gradle path even in repos that do not publish via Gradle
# yet ("prepared for Gradle"). Standalone (no `needs:`), parallel at pipeline
# start, `environment: maven-central` so it reads the same secret the publish
# uses. Red-by-design where the secret is not delivered (see the gpg job's note).
#
# SECURITY: prints no secret material. Key/passphrase reach Gradle only via env
# (read by System.getenv at runtime); `set -x` is never enabled; the passphrase
# is `::add-mask::`ed; Gradle runs with `--stacktrace` only. Only the produced
# `.asc` (exit code) is asserted. Uses Gradle 9.6.1.
# ---------------------------------------------------------------------------
verify-signing-key-gradle:
name: Verify GPG signing key — Gradle/BouncyCastle path (no secrets printed)
runs-on: ubuntu-latest
environment: maven-central
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: '21'
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: "9.8.0"
- name: Sign a throwaway artifact via useInMemoryPgpKeys (BouncyCastle)
shell: bash
env:
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
run: |
set -euo pipefail # NOTE: deliberately NO `set -x` — it would echo the passphrase.
if [ -z "${MAVEN_GPG_PRIVATE_KEY:-}" ]; then
echo "::error::MAVEN_GPG_PRIVATE_KEY is empty for this run. The maven-central environment did not deliver the secret to this ref (fork PR / other branch). Nothing to verify."
exit 1
fi
if [ -n "${MAVEN_GPG_PASSPHRASE:-}" ]; then echo "::add-mask::${MAVEN_GPG_PASSPHRASE}"; fi
PROJ=".github/signing-selftest"
echo "== Sign a throwaway artifact through Gradle's useInMemoryPgpKeys (BouncyCastle) =="
gradle --no-daemon -p "$PROJ" signMakeArtifact --stacktrace
ASC="$PROJ/build/signing-selftest.zip.asc"
if [ -f "$ASC" ]; then
echo " Detached signature produced: $(wc -c < "$ASC") armored bytes"
echo "RESULT: OK — Gradle's useInMemoryPgpKeys accepted the armored key + passphrase and produced a signature."
else
echo "::error::Gradle signing produced no .asc — useInMemoryPgpKeys could not build a usable signatory from MAVEN_GPG_PRIVATE_KEY / MAVEN_GPG_PASSPHRASE."
exit 1
fi
# ---------------------------------------------------------------------------
# Download + cache the GGUF test models ONCE, upfront, for the whole pipeline.
# Every Java test job (`test-java-*`) and the langchain4j integration job `needs:` this
# job and then only RESTORES the shared cache (key gguf-models-<manifest hash>) — so the ~5 GB model
# set is fetched from HuggingFace at most once per cache lifetime instead of racing to
# download in each job. GGUF is platform-independent, so this single ubuntu job's cache
# is reused by the macOS and Windows jobs too. On a warm cache this job is a no-op
# restore; on a cold cache it downloads all models, validates them, and saves the cache
# at job end (immutable key, so downstream jobs restore-hit). This is the single source
# of the download logic — do not re-add per-job downloads.
# ---------------------------------------------------------------------------
download-models:
name: Download + cache GGUF models (once)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Cache GGUF models (GitHub Actions cache; avoids re-downloading from HuggingFace)
# This job is the ONLY writer of the model cache (every consumer job uses the
# restore-only action). enableCrossOsArchive makes the one ubuntu-built entry
# restorable on macOS AND Windows — without it, cache entries are versioned
# per-OS, and the unreachable Windows-side entry was once found re-saved EMPTY
# (343 B) after an eviction, silently starving the Windows jobs of models.
uses: actions/cache@v6
with:
path: models/
# GGUF is platform-independent, so ubuntu + macOS + Windows share one entry.
# The key is derived from the model manifest, so EDITING models.csv
# automatically creates a fresh complete entry — no manual cache deletion.
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Download missing models (manifest-driven, .github/models.csv)
# The manifest is the single source of truth for the model set; this loop is
# the ONLY place in CI that downloads models. Files already restored from the
# cache are skipped, so a warm cache downloads nothing.
run: |
while IFS=, read -r name url; do
case "$name" in ''|\#*) continue ;; esac
test -f "models/$name" || curl -L --proto =https --proto-redir =https --fail --retry 5 --retry-all-errors "$url" --create-dirs -o "models/$name"
done < .github/models.csv
- name: List files in models directory
run: ls -l models/
- name: Validate model files
run: bash .github/validate-models.sh
# Prove the freshly written cache entry is restorable AND complete on every OS the
# pipeline uses BEFORE any model-consuming job starts: the same restore-only action
# the consumers use (fail-on-cache-miss makes an unrestorable entry fail here, not
# deep inside a test job), followed by the full validate gate. Every model-backed
# job `needs:` this instead of download-models directly.
verify-model-cache:
name: "Verify model cache (${{ matrix.os }})"
needs: download-models
strategy:
fail-fast: false
matrix:
include:
- os: ubuntu-latest
validate: bash .github/validate-models.sh
- os: macos-15
validate: bash .github/validate-models.sh
- os: windows-2025-vs2026
validate: .github\validate-models.bat
runs-on: ${{ matrix.os }}
steps:
- uses: actions/checkout@v7
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
fail-on-cache-miss: true
- name: Validate model files
run: ${{ matrix.validate }}
# ---------------------------------------------------------------------------
# Cross-compile jobs (Docker / dockcross) — produce release artifacts, no testing
# ---------------------------------------------------------------------------
code-style:
name: Code style (spotless) + package graph
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: '21'
distribution: temurin
- name: Spotless check (fail fast on format violations)
run: mvn -B --no-transfer-progress -f llama/pom.xml spotless:check
- name: SpotBugs check (fail fast on static-analysis findings)
run: mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile spotbugs:check
- name: Print internal package dependency graph (jdeps, informational)
continue-on-error: true
run: |
mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile
echo "=== internal package dependency graph (jdeps, bytecode) ==="
jdeps -verbose:package llama/target/classes | grep 'net.ladenthin.llama' || true
# ---------------------------------------------------------------------------
# Sibling module `llama-langchain4j` (LangChain4j adapters). Pure Java, no native
# code and no per-classifier matrix: it compiles against the core's stable Java API
# (identical across every classifier) and the backend is a runtime choice for the
# consumer. This job installs the parent + core into the local repo, then builds + tests
# the module (Java 17; langchain4j 1.x baseline). It runs its mapping unit tests; the
# model-backed integration test self-skips without a GGUF. `verify` also builds the
# javadoc/sources jars so a release-time javadoc break is caught here in PR CI. Version
# lockstep is now guaranteed by construction (both modules inherit the parent's version),
# so the old lockstep guard is gone.
# ---------------------------------------------------------------------------
test-java-llama-langchain4j:
name: Build and Test llama-langchain4j
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- name: Install parent + core net.ladenthin:llama into the local repo (Java only)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Build and test llama-langchain4j
run: mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml verify
# ---------------------------------------------------------------------------
# Model-free unit tests for the Kotlin coroutines facade (llama-kotlin).
# Pure Kotlin/JVM reactor module; its 6 tests fake the Iterable+AutoCloseable
# seam, so no native library and no model are needed.
# ---------------------------------------------------------------------------
test-java-llama-kotlin:
name: Build and Test llama-kotlin
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- name: Install parent + core net.ladenthin:llama into the local repo (Java only)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Build and test llama-kotlin
run: mvn -B --no-transfer-progress -f llama-kotlin/pom.xml verify
# ---------------------------------------------------------------------------
# Model-backed integration for the langchain4j adapters. Reuses the SAME shared GGUF
# cache (populated once by download-models) and the SAME Linux-x86_64 native artifact the
# core Java jobs already use — no extra model download and no duplicated download logic
# (restore-only cache, no curl steps). It exercises the chat / embedding / scoring
# adapters against the already-cached chat (Qwen3-0.6B), nomic-embedding and jina-reranker
# models. The model-backed tests self-skip when a model is absent, so a cold cache degrades
# to a skip, never a failure.
# ---------------------------------------------------------------------------
test-java-llama-langchain4j-integration:
name: Integration Test llama-langchain4j (model-backed)
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Download Linux x86_64 native library (reused, not rebuilt)
uses: actions/download-artifact@v8
with:
name: Linux-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install parent + core net.ladenthin:llama (bundles the downloaded native library)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Run llama-langchain4j model-backed integration tests (reused cached models)
run: >
mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml test
-Dnet.ladenthin.llama.model.path=models/${REASONING_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.embedding.model=models/${NOMIC_EMBED_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.rerank.model=models/${RERANKING_MODEL_NAME}
-Dnet.ladenthin.llama.langchain4j.tool.model=models/${TOOL_MODEL_NAME}
# This job is model-backed and crosses JNI, so a forked test JVM here can abort exactly the
# way the six test-java-* jobs can -- but it had neither of their diagnostics. Same step and
# same path set as those, scoped to this module. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama-langchain4j/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama-langchain4j/target/surefire-reports/*.dumpstream llama-langchain4j/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-langchain4j-integration
path: |
${{ github.workspace }}/llama-langchain4j/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama-langchain4j/*.hprof
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dump
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.txt
${{ github.workspace }}/llama-langchain4j/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
# ---------------------------------------------------------------------------
# Build the llama.cpp WebUI ONCE, from the same pinned tag CMakeLists.txt fetches,
# and share it to every native build as the generated, platform-independent
# ui.cpp/ui.h ("webui-generated" artifact). The native builds embed it into
# libjllama (CMake's "WebUI assets" block); when this job's artifact is absent the
# build falls back to the empty-asset stub. npm runs only here, in one controlled
# job — never in the dockcross cross-compilers (which have no node) or per-platform.
# ---------------------------------------------------------------------------
# ---------------------------------------------------------------------------
# llama-atmosphere-agent: the standalone (non-reactor, never on Maven Central) local
# coding-agent project that wires Atmosphere's built-in OpenAI-compatible agent runtime to
# this project's OpenAiCompatServer. Three jobs, all publish gates:
# - model-free: unit tests + the wire-contract tests, which drive the REAL
# OpenAiCompatServer over a loopback socket with a scripted backend (no native lib,
# no GGUF) and pin the streamed tool_calls / role=tool / multi-round shape — seconds,
# on every PR. It also builds the GitHub Release asset (the agent jar WITHOUT the core).
# - model-backed: the same loop against the cached Qwen2.5-1.5B tool model through the
# downloaded Linux native library (chat, streaming, tool call + result, read/write/read
# loop). A gate since its assertions stopped pinning wording (content checks are limited
# to facts no instruct model gets wrong and to tool results) and it ran green throughout.
# - smoke-agent-linux (further down, after package-fatjars): the release asset itself,
# started next to the real core fat jar.
# The project is built with -Dllama.version=<reactor version> against the core that was just
# installed to the local repo, so it always tests the code of this checkout.
# ---------------------------------------------------------------------------
test-java-llama-atmosphere-agent:
name: Build and Test llama-atmosphere-agent (model-free)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- name: Install parent + core net.ladenthin:llama into the local repo (Java only)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Resolve the reactor version
run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV"
- name: Spotless check
run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" spotless:check
- name: Build and test (unit + model-free wire contract against the real OpenAiCompatServer)
run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" verify
# The GitHub Release asset llama-atmosphere-agent-<core version>-jar-with-dependencies.jar:
# the agent plus Atmosphere/JLine, WITHOUT the core (src/assembly/agent-jar.xml), so it is a
# few MB and the natives are not in the release twice. Never deployed to Maven Central.
# smoke-agent-linux launches it next to the real core fat jar; the attach jobs sign it.
- name: Build the agent release jar (without the core)
run: >
mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}"
-P assembly -DskipTests package
- name: Collect the agent release jar + sha256
run: |
mkdir -p agent-jar
cp "llama-atmosphere-agent/target/llama-atmosphere-agent-${VERSION}-jar-with-dependencies.jar" agent-jar/
(cd agent-jar && for f in *.jar; do sha256sum "$f" > "$f.sha256"; done)
ls -la agent-jar
- name: Upload the agent release jar
uses: actions/upload-artifact@v7
with:
name: llama-atmosphere-agent-jar
path: agent-jar/
compression-level: 0 # jars are already deflated
retention-days: 7
if-no-files-found: error
test-java-llama-atmosphere-agent-integration:
name: Integration Test llama-atmosphere-agent (model-backed)
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Download Linux x86_64 native library (reused, not rebuilt)
uses: actions/download-artifact@v8
with:
name: Linux-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install parent + core net.ladenthin:llama (bundles the downloaded native library)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Resolve the reactor version
run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV"
- name: Run the Atmosphere tool-loop integration test (cached Qwen2.5-1.5B tool model, CPU)
run: >
mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" test
-Dtest=AtmosphereToolLoopIntegrationTest -Dsurefire.failIfNoSpecifiedTests=false
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME}
-Dnet.ladenthin.llama.test.ngl=0
# Model-backed and crossing JNI: same crash diagnostics as the langchain4j integration job.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
for f in llama-atmosphere-agent/hs_err_pid*.log; do
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama-atmosphere-agent/target/surefire-reports/*.dumpstream llama-atmosphere-agent/target/surefire-reports/*.dump; do
echo "===== $f ====="
cat "$f"
done
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-atmosphere-agent-integration
path: |
${{ github.workspace }}/llama-atmosphere-agent/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama-atmosphere-agent/*.hprof
build-webui:
name: Build WebUI assets (shared)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Resolve pinned llama.cpp tag from CMakeLists.txt
id: tag
shell: bash
run: |
TAG=$(grep -oE 'GIT_TAG[[:space:]]+b[0-9]+' llama/CMakeLists.txt | grep -oE 'b[0-9]+' | head -1)
if [ -z "$TAG" ]; then
echo "could not resolve llama.cpp GIT_TAG (b<nnnn>) from CMakeLists.txt" >&2
exit 1
fi
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "Pinned llama.cpp WebUI tag: $TAG"
- name: Checkout llama.cpp tools/ui at the pinned tag
uses: actions/checkout@v7
with:
repository: ggml-org/llama.cpp
ref: ${{ steps.tag.outputs.tag }}
path: llamacpp-ui
# tools/ui holds the npm project AND the ui.cpp.in / ui.h.in templates;
# scripts/ holds ui-assets.cmake, which consumes them. Both are needed
# since upstream #28445 replaced the embed.cpp host tool.
sparse-checkout: |
tools/ui
scripts
sparse-checkout-cone-mode: true
- uses: actions/setup-node@v7
with:
node-version: '24'
cache: npm
cache-dependency-path: llamacpp-ui/tools/ui/package-lock.json
- name: Build WebUI (Svelte/Vite)
working-directory: llamacpp-ui/tools/ui
env:
HF_UI_VERSION: ${{ steps.tag.outputs.tag }}
LLAMA_BUILD_NUMBER: ${{ steps.tag.outputs.tag }}
run: |
npm ci --ignore-scripts
npm run build
test -f dist/index.html
- name: Embed assets into ui.cpp / ui.h (upstream scripts/ui-assets.cmake)
shell: bash
run: |
set -euo pipefail
# Upstream #28445 ("ui : embed assets directly with CMake") deleted the
# tools/ui/embed.cpp host tool this step used to compile, and replaced it
# with scripts/ui-assets.cmake -- a plain `cmake -P` script, no npm and no
# host executable. Priority 1 of its provisioning order is "pre-built
# assets in <UI_SOURCE_DIR>/dist", which is exactly what the npm step
# above produced, so BUILD_UI and HF_ENABLED stay OFF: no second npm run
# and no Hugging Face download happen here. LLAMA_UI_GZIP is upstream's
# own knob and replaces the hand-rolled gzip loop this step used to do.
GEN="${RUNNER_TEMP}/ui-assets"
OUT="${GITHUB_WORKSPACE}/llama/webui-generated"
mkdir -p "$GEN" "$OUT"
cmake \
"-DUI_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui/tools/ui" \
"-DUI_BINARY_DIR=${GEN}" \
"-DLLAMA_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui" \
-DBUILD_UI=OFF \
-DHF_ENABLED=OFF \
-DLLAMA_UI_GZIP=ON \
-P "${GITHUB_WORKSPACE}/llamacpp-ui/scripts/ui-assets.cmake"
# The script also drops a ui-gzip/ working tree next to the generated
# sources; copy only the two files the artifact is defined to carry.
cp "$GEN/ui.cpp" "$GEN/ui.h" "$OUT/"
echo "=== generated WebUI assets ==="
ls -la "$OUT"
# Guard against a silently empty WebUI. A bare `grep LLAMA_UI_HAS_ASSETS`
# does NOT work here and would pass the failure case: upstream's ui.h.in
# emits "/* #undef LLAMA_UI_HAS_ASSETS */" when the table is empty, so the
# token is present either way. (The old embed.cpp emitted no such line at
# all, which is why the naive grep used to be sufficient.) Assert the ACTIVE
# #define and a non-zero asset count instead -- verified against both paths.
N=$(sed -n 's/.*std::array<llama_ui_asset, \([0-9]\+\)>.*/\1/p' "$OUT/ui.h" | head -1)
if grep -qE '^[[:space:]]*#define[[:space:]]+LLAMA_UI_HAS_ASSETS' "$OUT/ui.h" \
&& [ -n "$N" ] && [ "$N" -gt 0 ]; then
echo "LLAMA_UI_HAS_ASSETS: present, $N assets embedded"
else
echo "ERROR: ui-assets.cmake produced an empty asset table (assets=${N:-unknown})" >&2
exit 1
fi
- name: Upload WebUI artifact
uses: actions/upload-artifact@v7
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
retention-days: 1
if-no-files-found: error
crosscompile-linux-x86_64-cuda:
name: Cross-Compile manylinux_2_28 x86_64 (CUDA)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# CUDA cache. build_cuda_linux.sh execs build.sh, so the same sccache probe guards this job.
# build.sh also wraps nvcc (CMAKE_CUDA_COMPILER_LAUNCHER=sccache) for CUDA builds, so the
# per-arch .cu device passes — the dominant cost of this job — cache over Depot alongside the
# gcc host TUs. Verified on a warm run: 100% hit on CUDA / CUBIN / device-code (139 CUDA hits,
# 99.86% overall), cutting the job from ~51 min cold to ~15 min warm. The job therefore always
# builds the FULL CMAKE_CUDA_ARCHITECTURES set (no single-arch shortcut) and leans on the warm
# cache for speed, so every artifact stays release-safe (runs on every GPU generation) on PR /
# push as well as publish. CUDA_FAST_BUILD still exists in build_cuda_linux.sh as a LOCAL-dev
# knob, but CI no longer sets it. The first-run sccache debug diagnostics (SCCACHE_LOG /
# SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that caching is confirmed; build.sh still
# prints the `sccache --show-stats` hit table at the end of every run. Inert without DEPOT_TOKEN
# (fork PRs) or use_cache=false.
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: |
echo "=== Host CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-manylinux_2_28-x64 .github/build_cuda_linux.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: linux-libraries-cuda
path: ${{ github.workspace }}/llama/src/main/resources_linux_cuda/net/ladenthin/llama/
crosscompile-linux-x86_64:
name: Cross-Compile manylinux2014 x86_64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 1, VERIFIED green in CI (PR #245): sccache v0.16.0
# probe passed in-container (devtoolset-10 gcc), cache ON over Depot WebDAV (cold run: 275
# objects stored). Steady-state env below — the first-run diagnostics (SCCACHE_LOG /
# SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that it is proven. Inert without
# DEPOT_TOKEN (fork PRs) or with use_cache=false; a crashing sccache still falls back to a
# green uncached build via the build.sh probe.
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: |
echo "=== Host CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-manylinux2014-x64 .github/build.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
crosscompile-linux-aarch64:
name: Build and Test Linux aarch64
needs: [startgate, build-webui]
# Native ARM64 build on GitHub's free arm64 runner, mirroring upstream llama.cpp's
# `ubuntu-cpu` aarch64 release job (ubuntu-24.04-arm + GCC 14). Replaces the former dockcross
# `linux-arm64-lts` cross-compile (GCC 8.5, glibc 2.17), which can no longer compile llama.cpp
# b9789 — its C++17 CTAD-in-`new` needs GCC >= 12. Building natively also lets us run the C++
# unit suite (ctest) on real ARM hardware for the first time (the cross build ran no tests).
# Trade-off: the glibc floor rises 2.17 -> ~2.39, the same envelope upstream's own ARM binaries
# require. GGML_NATIVE=OFF keeps the artifact portable across ARMv8 CPU generations (no
# build-host -march baked in). The job id is kept (a `needs:` target downstream); only the
# display name changed, so update any branch-protection required-check that pinned the old name.
runs-on: ubuntu-24.04-arm
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install toolchain (GCC 14, mirrors upstream llama.cpp ARM release)
run: |
sudo apt-get update
sudo apt-get install -y gcc-14 g++-14
echo "CC=gcc-14" >> "$GITHUB_ENV"
echo "CXX=g++-14" >> "$GITHUB_ENV"
- name: Display CPU Info
shell: bash
run: |
echo "=== Host CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DOS_NAME=Linux -DOS_ARCH=aarch64 -DGGML_NATIVE=OFF -DBUILD_TESTING=ON"
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-aarch64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-linux-s390x:
name: Build and Test Linux s390x (big-endian, qemu)
needs: [startgate, build-webui]
# Cross-compile for IBM Z (s390x, BIG-ENDIAN) with the GCC cross toolchain, then run the full
# C++ unit suite under qemu-user — a real big-endian correctness gate for our helpers and
# serializers (esp. the little-endian WAV writer, JSON/token/embedding transforms). The BUILD
# is native speed (x86 cross-gcc); only the tiny test binary is emulated. s390x is a DEFAULT-jar
# CPU platform (like aarch64), so the artifact merges via the `*-libraries` glob (no classifier /
# pom profile). Model-backed Java tests are NOT run under emulation (a JVM + GGUF inference under
# qemu-user is slow/flaky); the C++ gate covers the actual byte-order risk since the Java<->JNI
# boundary uses host-native array copies. GGML_OPENMP=OFF avoids cross-libgomp issues (ggml uses
# its own std::thread pool). CMAKE_CROSSCOMPILING_EMULATOR makes ctest run the s390x exe via qemu;
# QEMU_LD_PREFIX lets the emulated binary find the s390x sysroot libs.
runs-on: ubuntu-latest
env:
QEMU_LD_PREFIX: /usr/s390x-linux-gnu
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install s390x cross toolchain + qemu-user
run: |
sudo apt-get update
sudo apt-get install -y gcc-s390x-linux-gnu g++-s390x-linux-gnu qemu-user-static
# NOTE: GGML_NATIVE=OFF is required here (an x86 build host must not bake -march=native into an
# s390x artifact), and it has a non-obvious second effect: ggml declares
# `option(GGML_VXE "ggml: enable vxe" ${GGML_NATIVE})`, so VXE is off too. No `-mvx -mzvector` is
# passed, `__VEC__` stays undefined, and ggml-cpu-impl.h's `#if defined(__s390x__) && defined(__VEC__)`
# never self-defines `__VXE__`/`__VXE2__`. This job therefore builds a SCALAR s390x binary -- which is
# exactly right for what it is (a big-endian correctness gate for our own layer, not a perf target).
# Do NOT "fix" a VXE-related compile error by adding -DGGML_VXE=ON: that define sets __VXE__ and
# __VXE2__ together while -march stays at the toolchain default arch11, so every z14+ builtin is
# rejected. See the "`0013` was dropped at the b10948 bump" note in CLAUDE.md for the measured
# comparison (the patch itself is gone -- upstream merged it as ggml-org/llama.cpp#28775).
- name: Build libraries (cross-compile s390x)
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_NATIVE=OFF -DGGML_OPENMP=OFF -DBUILD_TESTING=ON -DCMAKE_SYSTEM_NAME=Linux -DCMAKE_SYSTEM_PROCESSOR=s390x -DCMAKE_C_COMPILER=s390x-linux-gnu-gcc -DCMAKE_CXX_COMPILER=s390x-linux-gnu-g++ -DCMAKE_CROSSCOMPILING_EMULATOR=/usr/bin/qemu-s390x-static -DOS_NAME=Linux -DOS_ARCH=s390x"
- name: Run C++ unit tests under qemu-s390x (big-endian gate)
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-s390x-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-linux-x86_64-vulkan:
name: Build Linux x86_64 Vulkan
needs: [startgate, build-webui]
# Native ubuntu build (NOT dockcross) — the Vulkan SDK is trivial to apt-install here, and
# upstream llama.cpp builds its ubuntu-vulkan artifact the same way. GPU runtime libvulkan.so.1
# is supplied by the consumer's driver (nothing bundled). GitHub runners have NO GPU, so this
# is a BUILD-ONLY job (no -DBUILD_TESTING/ctest: a Vulkan-linked jllama_test errors enumerating
# devices on a GPU-less runner — same rationale as the Windows GPU jobs). GGML_NATIVE=OFF keeps
# the artifact portable across x86_64 CPU generations. Trade-off vs the manylinux CPU jar: the
# glibc floor rises to the ubuntu-latest baseline (same as the native aarch64 job). build.sh
# self-fetches sccache; the probe guards it (a miss just builds uncached).
runs-on: ubuntu-latest
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install Vulkan SDK (headers + loader + glslc shader compiler)
run: |
sudo apt-get update
sudo apt-get install -y libvulkan-dev glslc glslang-tools spirv-headers
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
if-no-files-found: error
build-linux-aarch64-vulkan:
name: Build Linux aarch64 Vulkan
needs: [startgate, build-webui]
# Native ARM64 Vulkan build on GitHub's free arm64 runner (same runner as the aarch64 CPU job).
# Build-only (GPU-less runner); GGML_NATIVE=OFF for portability across ARMv8 generations; GCC 14
# to match the aarch64 CPU job. Reuses the resources_linux_vulkan tree (arch subdir Linux/aarch64);
# the vulkan-linux-aarch64 Maven profile packages only that subtree.
runs-on: ubuntu-24.04-arm
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install toolchain (GCC 14) + Vulkan SDK
run: |
sudo apt-get update
sudo apt-get install -y gcc-14 g++-14 libvulkan-dev glslc glslang-tools spirv-headers
echo "CC=gcc-14" >> "$GITHUB_ENV"
echo "CXX=g++-14" >> "$GITHUB_ENV"
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=aarch64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-aarch64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
if-no-files-found: error
crosscompile-android-aarch64:
name: Cross-Compile Android aarch64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 4. Same steady-state env as manylinux2014 (job 1);
# the build.sh probe makes it safe to enable without a separate verification run. Inert
# without DEPOT_TOKEN (fork PRs) or use_cache=false.
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: bash
run: |
echo "=== Host CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-arm64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-Android-aarch64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
crosscompile-android-x86_64:
name: Cross-Compile Android x86_64
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Android x86_64 CPU ABI: consumed by the llama-android AAR as jni/x86_64 (so the
# Android emulator — and x86_64 Android devices/Chromebooks — can run the binding)
# and merged into the default JAR's Linux-Android/x86_64 tree via the *-libraries
# glob. Same dockcross + sccache steady-state env as the arm64 job; the wrapper
# pins the same image tag. Fail-loud and in the package/publish needs graphs.
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-x86_64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-Android-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
if-no-files-found: error
crosscompile-android-aarch64-opencl:
name: Cross-Compile Android aarch64 (OpenCL/Adreno)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
# Phase 2 dockcross cache rollout — job 5. build_opencl_android.sh stages the OpenCL
# headers/loader, then delegates the jllama cmake build to build.sh (which owns the
# sccache probe + launcher). Same steady-state env as the other dockcross jobs. Inert
# without DEPOT_TOKEN (fork PRs) or use_cache=false.
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE"
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Build libraries
shell: bash
run: |
.github/dockcross/dockcross-android-arm64 .github/build_opencl_android.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64 -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: android-libraries-opencl
path: ${{ github.workspace }}/llama/src/main/resources_android_opencl/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Android AAR packaging (net.ladenthin:llama-android + llama-android-opencl).
# Plain-Gradle AAR assembly — no AGP and no Android SDK needed to BUILD (an AAR
# is a documented zip; see llama-android/README.md); AGP is only needed to
# CONSUME it, which the consumer smoke test below exercises on the runner's
# preinstalled Android SDK: a minimal app resolves the AAR from mavenLocal and
# runs a full R8 release build (validating AAR format, manifest minSdk merge,
# jni/ packaging, and the shipped consumer proguard rules). Structural checks
# additionally pin the AAR entries and the 16 KB LOAD-segment alignment
# (Google Play requirement for Android 15+ targets). Fail-loud and in the
# publish `needs:` graphs — a broken AAR blocks publishing, same policy as
# every native artifact job.
# ---------------------------------------------------------------------------
package-android-aar:
name: Package + Validate Android AARs
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, crosscompile-android-aarch64-opencl]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0.
gradle-version: "9.8.0"
- name: Build core jar (byte-identical classes payload for the AAR)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true package
- name: Download Android CPU natives (arm64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-aarch64-libraries
path: stage/cpu/
- name: Download Android CPU natives (x86_64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-x86_64-libraries
path: stage/cpu-x86_64/
- name: Download Android OpenCL natives
uses: actions/download-artifact@v8
with:
name: android-libraries-opencl
path: stage/opencl/
- name: Stage natives for the AAR build
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp stage/cpu/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp stage/cpu-x86_64/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
cp stage/opencl/Linux-Android/aarch64/libjllama.so llama-android/natives/opencl/arm64-v8a/
- name: Assemble AARs + publish to mavenLocal
run: gradle -p llama-android aarCpu aarOpencl publishToMavenLocal
- name: Validate AAR structure + 16 KB page-size alignment
shell: bash
run: |
set -euo pipefail
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
for flavor in llama-android llama-android-opencl; do
AAR="llama-android/build/aar/${flavor}-${VERSION}.aar"
echo "== validating $AAR"
test -f "$AAR"
# The CPU AAR is multi-ABI (arm64 devices + x86_64 emulators/devices);
# the OpenCL flavor stays arm64-only (Adreno is Qualcomm ARM hardware).
ABIS="arm64-v8a"
if [ "$flavor" = "llama-android" ]; then ABIS="arm64-v8a x86_64"; fi
ENTRIES="AndroidManifest.xml classes.jar proguard.txt R.txt"
for abi in $ABIS; do ENTRIES="$ENTRIES jni/$abi/libjllama.so"; done
for entry in $ENTRIES; do
unzip -l "$AAR" | grep -q "${entry}$" || { echo "::error::$AAR is missing $entry"; exit 1; }
done
unzip -p "$AAR" AndroidManifest.xml | grep -q 'android:minSdkVersion="28"' \
|| { echo "::error::$AAR manifest lost minSdkVersion 28"; exit 1; }
unzip -p "$AAR" classes.jar > /tmp/aar-classes.jar
unzip -l /tmp/aar-classes.jar | grep -q "net/ladenthin/llama/LlamaModel.class" \
|| { echo "::error::$AAR classes.jar lost LlamaModel"; exit 1; }
if unzip -l /tmp/aar-classes.jar | grep -qE "module-info\.class|net/ladenthin/llama/(Linux|Mac|Windows)/"; then
echo "::error::$AAR classes.jar carries module-info or desktop native resources"; exit 1
fi
rm -rf /tmp/aar-natives && unzip -o -q -d /tmp/aar-natives "$AAR" "jni/*/libjllama.so"
# Google Play 16 KB page-size requirement (Android 15+ targets): every
# LOAD segment must be aligned to a multiple of 16384. CMake pins
# -Wl,-z,max-page-size=16384 for every Android ABI; this asserts it held.
for so in /tmp/aar-natives/jni/*/libjllama.so; do
for align in $(readelf -lW "$so" | awk '$1=="LOAD"{print $NF}'); do
if [ $(( align % 16384 )) -ne 0 ]; then
echo "::error::$so: LOAD alignment $align is not a multiple of 16384 (16 KB page-size regression)"; exit 1
fi
done
# dlopen-ability gate: an app consuming the AAR bundles no other native
# libs, so every DT_NEEDED must be a bionic system library (plus the
# vendor ICD libOpenCL.so for the OpenCL flavor). A stray dependency —
# libomp.so / libc++_shared.so once shipped exactly this way — makes
# System.loadLibrary fail on every device with UnsatisfiedLinkError.
# CMake's Android guard (GGML_OPENMP OFF + -static-libstdc++) keeps the
# list clean; this asserts it held.
ALLOWED="libc.so libm.so libdl.so liblog.so libandroid.so"
if [ "$flavor" = "llama-android-opencl" ]; then ALLOWED="$ALLOWED libOpenCL.so"; fi
for needed in $(readelf -dW "$so" | awk '/\(NEEDED\)/{gsub(/[\[\]]/,"",$NF); print $NF}'); do
case " $ALLOWED " in
*" $needed "*) ;;
*) echo "::error::$so: DT_NEEDED '$needed' is not a bionic system library — dlopen would fail on-device"; exit 1 ;;
esac
done
done
done
- name: AGP consumer smoke test (R8 release build from mavenLocal)
shell: bash
run: |
set -euo pipefail
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
gradle -p .github/android-consumer-test assembleRelease "-PjllamaVersion=${VERSION}"
APK=.github/android-consumer-test/app/build/outputs/apk/release/app-release.apk
test -f "$APK"
for abi in arm64-v8a x86_64; do
unzip -l "$APK" | grep -q "lib/$abi/libjllama.so" \
|| { echo "::error::consumer APK is missing lib/$abi/libjllama.so"; exit 1; }
done
# The AAR's consumer proguard.txt must have carried the binding through R8.
unzip -p "$APK" "classes*.dex" | grep -aq "Lnet/ladenthin/llama/LlamaModel;" \
|| { echo "::error::R8 stripped net.ladenthin.llama.LlamaModel — consumer proguard rules broken"; exit 1; }
- name: Upload AARs
uses: actions/upload-artifact@v7
with:
name: llama-android-aars
path: llama-android/build/aar/*.aar
if-no-files-found: error
# ---------------------------------------------------------------------------
# On-emulator runtime validation of the Android AAR: boots a KVM-accelerated
# x86_64 emulator (GitHub Linux runners have KVM; arm64 images cannot run here,
# which is exactly why the AAR carries the jni/x86_64 ABI), publishes the CPU AAR
# to mavenLocal, adb-pushes the already-cached tiny draft model, and runs the
# consumer fixture's connectedDebugAndroidTest — System.loadLibrary from the APK,
# pure-Java GgufInspector on-device, and real native inference on Android/bionic.
# RELEASE GATE (in both publish needs graphs) since PR #298: the job ran
# flake-free through the PR's validation cycle (boot ~30 s, on-device inference
# green), so a broken on-device runtime now blocks publishing — same fail-loud
# policy as every native artifact job. If emulator-boot flakiness ever appears,
# re-run the job first; demote it from the needs graphs only as a last resort.
# ---------------------------------------------------------------------------
test-android-emulator:
name: Android emulator on-device test (x86_64)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- name: Free up disk space
# The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages"
# (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap
# file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it.
uses: ggml-org/free-disk-space@v1.3.1
with:
android: false
large-packages: false
tool-cache: false
swap-storage: false
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0.
gradle-version: "9.8.0"
- name: Enable KVM group permissions (GitHub-hosted runner)
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build core jar (classes payload for the AAR)
run: >
mvn -B --no-transfer-progress -pl llama -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true package
- name: Download Android CPU natives (arm64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-aarch64-libraries
path: stage/cpu/
- name: Download Android CPU natives (x86_64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-x86_64-libraries
path: stage/cpu-x86_64/
- name: Stage natives + publish the CPU AAR to mavenLocal
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64
cp stage/cpu/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp stage/cpu-x86_64/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
gradle -p llama-android publishLlamaAndroidPublicationToMavenLocal
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Free disk space for the emulator (only the draft model is needed on-device)
# Same guard as test-android-llmservice: the AVD userdata partition needs ~7.4 GB; drop the
# rest of the restored ~10 GB GGUF cache (this job adb-pushes only the tiny draft model) plus
# the few preinstalled toolchains the free-disk-space step leaves, so the emulator can create
# userdata and boot.
run: |
echo "Disk before cleanup:"; df -h / | tail -1
find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true
sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true
echo "Disk after cleanup:"; df -h / | tail -1
- name: Run on-emulator instrumentation (connectedDebugAndroidTest)
uses: reactivecircus/android-emulator-runner@v2
with:
api-level: 30
arch: x86_64
target: default
disable-animations: true
emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim
# One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh,
# so shell control flow must live in the committed helper script.
script: .github/run-android-emulator-test.sh
- name: Upload instrumentation reports (on failure)
if: failure()
uses: actions/upload-artifact@v7
with:
name: android-emulator-test-reports
path: .github/android-consumer-test/app/build/reports/androidTests/
if-no-files-found: ignore
# ---------------------------------------------------------------------------
# The shippable Android app "LLM Service" (android-llmservice) — a KISS, fully-offline
# on-device chat app consuming the llama-android AAR + llama-kotlin facade. Split into TWO jobs
# so the installable artifacts are ALWAYS produced even when the on-device UI test is flaky/slow:
# * build-android-llmservice — builds the signed release AAB (real upload key when the
# ANDROID_UPLOAD_KEYSTORE_BASE64 secret is set, else debug-signed) + the installable APK and
# uploads both. No emulator, so a flaky/slow emulator can NEVER block getting the APK.
# * test-android-llmservice — a SEPARATE, non-gating check that boots the KVM x86_64 emulator
# and runs the app's Compose UI test (type a prompt, tap Send) against real on-device
# inference. It can go red on its own without stopping the build/artifacts.
# Neither is a publish gate (unlike test-android-emulator): a Compose/AGP toolchain hiccup must
# not block a Maven Central release of the library. Keep test-android-llmservice non-required in
# branch protection to keep the emulator test optional (visible-but-non-blocking).
# ---------------------------------------------------------------------------
build-android-llmservice:
name: Build the LLM Service Android app (AAB + APK)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0; also satisfies Kotlin
# 2.4's Gradle-plugin floor. The AAR/lib-only jobs stay on an older Gradle since
# they build no AGP project.
gradle-version: "9.8.0"
- name: Build core jar + install llama-kotlin facade to mavenLocal
# install (not package) so the app's Gradle build resolves llama-kotlin +
# the core POM from mavenLocal. llama-kotlin's core dep is provided-scope, so the
# desktop JAR is never pulled into the APK — the AAR supplies net.ladenthin.llama.*.
run: >
mvn -B --no-transfer-progress -pl llama,llama-kotlin -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Download Android CPU natives (arm64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-aarch64-libraries
path: stage/cpu/
- name: Download Android CPU natives (x86_64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-x86_64-libraries
path: stage/cpu-x86_64/
- name: Stage natives + publish the CPU AAR to mavenLocal
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64
cp stage/cpu/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp stage/cpu-x86_64/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
gradle -p llama-android publishLlamaAndroidPublicationToMavenLocal
- name: Decode upload keystore (if configured)
# Optional Play upload key. Set the ANDROID_UPLOAD_KEYSTORE_BASE64 secret
# (base64 of a PKCS12/JKS upload keystore) plus the three password/alias secrets
# below to sign the release AAB with a real upload key. Without it the release
# build falls back to debug signing (app/build.gradle.kts) so this step is inert
# on forks/PRs where secrets are withheld.
env:
KEYSTORE_B64: ${{ secrets.ANDROID_UPLOAD_KEYSTORE_BASE64 }}
run: |
if [ -n "${KEYSTORE_B64}" ]; then
echo "${KEYSTORE_B64}" | base64 -d > "${RUNNER_TEMP}/upload-keystore.jks"
echo "JLLAMA_UPLOAD_STORE_FILE=${RUNNER_TEMP}/upload-keystore.jks" >> "$GITHUB_ENV"
echo "Upload keystore decoded -> release AAB will be signed with the upload key."
else
echo "No ANDROID_UPLOAD_KEYSTORE_BASE64 secret -> release AAB will be debug-signed."
fi
- name: Build release bundle (AAB) + installable APK
env:
JLLAMA_UPLOAD_STORE_PASSWORD: ${{ secrets.ANDROID_UPLOAD_STORE_PASSWORD }}
JLLAMA_UPLOAD_KEY_ALIAS: ${{ secrets.ANDROID_UPLOAD_KEY_ALIAS }}
JLLAMA_UPLOAD_KEY_PASSWORD: ${{ secrets.ANDROID_UPLOAD_KEY_PASSWORD }}
run: |
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
# bundleRelease -> app-release.aab (Play upload, NOT directly installable);
# assembleRelease -> app-release.apk (sideloadable installable, debug-signed unless the
# upload-key secrets are set). The debug + androidTest APKs are built by the emulator
# test job (test-android-llmservice), not here — this job never touches the emulator.
gradle -p android-llmservice bundleRelease assembleRelease "-PjllamaVersion=${VERSION}"
test -f android-llmservice/app/build/outputs/bundle/release/app-release.aab
test -f android-llmservice/app/build/outputs/apk/release/app-release.apk
- name: Upload release bundle (AAB — for Play upload)
uses: actions/upload-artifact@v7
with:
name: android-llmservice-aab
path: android-llmservice/app/build/outputs/bundle/release/*.aab
if-no-files-found: error
- name: Upload installable APK (release — sideload / adb install)
# Directly installable APK, downloadable from this run's Artifacts (independent of any
# release). Debug-signed unless the ANDROID_UPLOAD_* secrets are set. This is the file to
# grab to try the app on a phone; the .aab above is only for the Play Console.
uses: actions/upload-artifact@v7
with:
name: android-llmservice-apk
path: android-llmservice/app/build/outputs/apk/release/*.apk
if-no-files-found: error
test-android-llmservice:
name: LLM Service app UI test on emulator (non-gating)
needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- name: Free up disk space
# The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages"
# (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap
# file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it.
uses: ggml-org/free-disk-space@v1.3.1
with:
android: false
large-packages: false
tool-cache: false
swap-storage: false
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: temurin
- uses: gradle/actions/setup-gradle@v6
with:
# AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0.
gradle-version: "9.8.0"
- name: Enable KVM group permissions (GitHub-hosted runner)
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build core jar + install llama-kotlin facade to mavenLocal
run: >
mvn -B --no-transfer-progress -pl llama,llama-kotlin -am -DskipTests -Denforcer.skip=true
-Dspotless.check.skip=true -Dspotbugs.skip=true
-Dmaven.javadoc.skip=true -Dmaven.source.skip=true -Dgpg.skip=true install
- name: Download Android CPU natives (arm64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-aarch64-libraries
path: stage/cpu/
- name: Download Android CPU natives (x86_64)
uses: actions/download-artifact@v8
with:
name: Linux-Android-x86_64-libraries
path: stage/cpu-x86_64/
- name: Stage natives + publish the CPU AAR to mavenLocal
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64
cp stage/cpu/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp stage/cpu-x86_64/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
gradle -p llama-android publishLlamaAndroidPublicationToMavenLocal
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Free disk space for the emulator (only the draft model is needed on-device)
# The AVD userdata partition needs ~7.4 GB; the full ~10 GB GGUF cache restore leaves too
# little free, so the emulator FATALs ("Not enough space to create userdata partition") and
# the boot poll loops until timeout. This job only adb-pushes the tiny draft model, so drop
# the rest of the restored cache plus what the free-disk-space step leaves (PowerShell, CodeQL).
run: |
echo "Disk before cleanup:"; df -h / | tail -1
find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true
sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true
echo "Disk after cleanup:"; df -h / | tail -1
- name: Run LLM Service UI test on emulator (connectedDebugAndroidTest)
uses: reactivecircus/android-emulator-runner@v2
with:
api-level: 30
arch: x86_64
target: default
disable-animations: true
emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim
# One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh.
script: .github/run-android-llmservice-test.sh
- name: Upload instrumentation reports (on failure)
if: failure()
uses: actions/upload-artifact@v7
with:
name: android-llmservice-test-reports
path: android-llmservice/app/build/reports/androidTests/
if-no-files-found: ignore
# ---------------------------------------------------------------------------
# Native build jobs — produce release artifacts + run C++ unit tests
# ---------------------------------------------------------------------------
build-macos-arm64-no-metal:
name: Build and Test macOS 15 arm64 (no Metal)
needs: [startgate, build-webui]
runs-on: macos-15
env:
BUILD_JOBS: 2
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DGGML_METAL=OFF -DGGML_NATIVE=OFF -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: macos-15-no-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-macos-arm64-metal:
name: Build and Test macOS 14 arm64 (Metal)
needs: [startgate, build-webui]
runs-on: macos-14
env:
BUILD_JOBS: 2
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: macos-14-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-windows-x86_64-msvc:
name: Build and Test Windows 2025 x86_64 (MSVC / VS 2026, classifier)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Visual Studio 18 2026" -A "x64" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-msvc
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-windows-x86-msvc:
name: Build and Test Windows 2025 x86 (MSVC / VS 2026, classifier)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Visual Studio 18 2026" -A "Win32" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86-msvc
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Windows Ninja Multi-Config + sccache — the DEFAULT Windows CPU natives.
# The Visual Studio generator ignores CMAKE_{C,CXX}_COMPILER_LAUNCHER, so only the
# Ninja Multi-Config generator can front cl.exe with sccache over Depot WebDAV
# (build.bat probe-guards it). Both generators use the same MSVC toolchain (cl.exe,
# static /MT CRT) on the same runner, so the produced jllama.dll/llama.dll/ggml.dll
# are functionally equivalent with identical runtime dependencies — the only delta
# is build-system plumbing + caching. The Ninja build is therefore the default JAR
# (artifacts `Windows-*-libraries`, picked up by the package job's `pattern:
# "*-libraries"`); the MSVC build above is shipped as the `msvc-windows` classifier
# for anyone who wants the Visual-Studio-generator natives. Upstream llama.cpp also
# builds its Windows artifacts with Ninja Multi-Config + MSVC.
# ---------------------------------------------------------------------------
build-windows-x86_64:
name: Build and Test Windows 2025 x86_64 (Ninja Multi-Config + sccache, default)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-windows-x86:
name: Build and Test Windows 2025 x86 (Ninja Multi-Config + sccache, default)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x86)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x86
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- name: Build libraries
shell: cmd
run: |
.github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
build-windows-arm64:
name: Build and Test Windows 11 arm64 (Ninja Multi-Config, default)
needs: [startgate, build-webui]
# Native arm64 build on GitHub's free windows-11-arm runner. Goes into the DEFAULT JAR (no
# classifier): OSInfo maps a Windows-on-ARM JVM (os.arch=aarch64) to Windows/aarch64, the same
# path CMake emits here, and the `*-libraries` glob in the package/publish jobs merges it into
# src/main/resources. sccache: the native aarch64-pc-windows-msvc release, wrapping clang-cl
# (guarded by build.bat's probe + uncached retry like every other Windows job).
#
# Compiler: clang-cl, NOT MSVC cl.exe. ggml's ggml-cpu/CMakeLists.txt aborts with "MSVC is not
# supported for ARM, use clang" via `if (MSVC AND NOT CMAKE_C_COMPILER_ID STREQUAL "Clang")`.
# clang-cl (LLVM's MSVC-compatible driver) satisfies that guard (its compiler id is "Clang")
# while still leaving CMake's MSVC=TRUE, so our static /MT CRT block (CMAKE_MSVC_RUNTIME_LIBRARY
# in CMakeLists.txt) keeps applying and the generator stays Ninja Multi-Config. msvc-dev-cmd
# (arm64) supplies the MSVC headers/libs/linker AND the bundled clang-cl / lld-link under
# VC\Tools\Llvm\ARM64, so no separate LLVM install is needed.
#
# GGML_OPENMP=OFF: with clang-cl, ggml links LLVM's OpenMP (libomp.lib -> needs libomp140.aarch64.dll
# at runtime), which is NOT on PATH like MSVC's ambient vcomp140.dll on x64 — so gtest_discover_tests
# (and any consumer) failed to launch the binary with 0xc0000135 STATUS_DLL_NOT_FOUND. Turning OpenMP
# off makes ggml use its own std::thread threadpool, so the arm64 jllama.dll (and the test exe) are
# self-contained with no libomp dependency to ship. The x86_64/x86 jobs keep OpenMP (MSVC vcomp).
runs-on: windows-11-arm
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (arm64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: arm64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
# sccache ships a native Windows-on-ARM build; it wraps clang-cl here.
# build.bat probes sccache before trusting it and, should a build fail with it as the
# launcher, retries once uncached -- so this can speed the job up but never red it.
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-aarch64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# No mvn compile needed: the JNI header (jllama.h) is committed and the native build
# uses the bundled JNI headers in .github/include, and OS_NAME/OS_ARCH are passed
# explicitly (so the OSInfo-class OS-detection path is skipped) — same as the x86_64 job.
# clang-cl (see the job comment) is required: ggml refuses MSVC cl.exe on ARM.
run: |
.github\build.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DOS_NAME=Windows -DOS_ARCH=aarch64 -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-aarch64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Windows GPU classifiers (x86_64 only) — CUDA, Vulkan, OpenCL.
# All three use the same Ninja Multi-Config + MSVC + sccache toolchain as the
# default CPU build; they differ only by the GGML backend flag (and the build-time
# SDK each needs). CMakeLists.txt routes each backend's output to its own
# src/main/resources_windows_{cuda,vulkan,opencl}/ tree, which the matching Maven
# profile (cuda-windows / vulkan-windows / opencl-windows) turns into a classifier
# JAR. GPU runtime libraries are NOT bundled — the consumer's GPU driver / toolkit
# provides them (CUDA: cudart64_13/cublas64_13 from the CUDA Toolkit; Vulkan:
# vulkan-1.dll from the driver; OpenCL: System32\OpenCL.dll from the driver).
# NOTE: GitHub-hosted Windows runners have NO GPU, so these jobs build + run the
# C++ unit suite (ctest, CPU-only) but cannot run model-backed GPU inference;
# end-to-end GPU validation is local / self-hosted.
# ---------------------------------------------------------------------------
build-windows-x86_64-cuda:
name: Build Windows 2025 x86_64 CUDA (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install CUDA Toolkit 13.4 (NVIDIA redist archives)
# KEEP IN SYNC WITH UPSTREAM: the component set and versions mirror llama.cpp's
# .github/actions/windows-setup-cuda ("Install Cuda Toolkit 13.4 for x64") at the pinned
# GIT_TAG. Jimver/cuda-toolkit (used here up to CUDA 13.3.1) has no 13.4 in any release or
# on master, so the toolkit is assembled from NVIDIA's per-component redist zips instead —
# which is also how upstream builds its own Windows CUDA release. cuda_crt is listed
# explicitly: since 13.x the nvcc crt headers (crt/host_config.h) ship in their own
# archive, and without them CMake's CUDA compiler detection fails at configure.
shell: pwsh
run: |
$ErrorActionPreference = "Stop"
$cuda = "C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v13.4"
$base = "https://developer.download.nvidia.com/compute/cuda/redist"
$parts = @(
"cuda_crt/cuda_crt-windows-x86_64-13.4.59",
"cuda_cudart/cuda_cudart-windows-x86_64-13.4.49",
"cuda_nvcc/cuda_nvcc-windows-x86_64-13.4.59",
"cuda_nvrtc/cuda_nvrtc-windows-x86_64-13.4.59",
"libcublas/libcublas-windows-x86_64-13.7.0.27",
"libnvvm/libnvvm-windows-x86_64-13.4.59",
"cuda_nvtx/cuda_nvtx-windows-x86_64-13.4.49",
"cuda_profiler_api/cuda_profiler_api-windows-x86_64-13.4.49",
"visual_studio_integration/visual_studio_integration-windows-x86_64-13.4.49",
"cccl/cccl-windows-x86_64-13.3.4.2.1"
)
New-Item -ItemType Directory -Force -Path $cuda | Out-Null
foreach ($p in $parts) {
$dir, $name = $p.Split("/")
$url = "$base/$dir/windows-x86_64/$name-archive.zip"
$zip = "$env:RUNNER_TEMP\$name.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile $zip
Expand-Archive -Path $zip -DestinationPath "$env:RUNNER_TEMP\cuda-redist" -Force
Copy-Item -Path "$env:RUNNER_TEMP\cuda-redist\$name-archive\*" -Destination $cuda -Recurse -Force
}
"$cuda\bin" | Out-File -FilePath $env:GITHUB_PATH -Append -Encoding utf8
"CUDA_PATH=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8
"CUDA_PATH_V13_4=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# GPU jobs build the artifact only — no -DBUILD_TESTING / ctest. The C++ unit
# suite is CPU-only and fully covered by the `C++ Tests` job + the CPU Windows
# jobs; a GPU-linked jllama_test.exe cannot be discovered/run on a GPU-less
# GitHub runner (it errors probing for a CUDA device -> ctest *_NOT_BUILT).
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_CUDA=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-cuda
path: ${{ github.workspace }}/llama/src/main/resources_windows_cuda/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-vulkan:
name: Build Windows 2025 x86_64 Vulkan (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install Vulkan SDK
uses: jakoch/install-vulkan-sdk-action@v1.6.0
with:
vulkan_version: 1.4.350.0
cache: true
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# Build the artifact only (see the CUDA job's note: GPU-less runner can't run a
# GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs).
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_VULKAN=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_windows_vulkan/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-opencl:
name: Build Windows 2025 x86_64 OpenCL (Ninja + sccache)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# Build the artifact only (see the CUDA job's note: GPU-less runner can't run a
# GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs).
run: |
.github\build_opencl_windows.bat -G "Ninja Multi-Config" -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
if-no-files-found: error
# ---------------------------------------------------------------------------
# Additional GPU-backend classifiers (fail-loud, same wiring as the CUDA/Vulkan/
# OpenCL jobs): AMD ROCm/HIP, Intel SYCL (oneAPI), Windows-on-ARM OpenCL (Adreno),
# Intel OpenVINO. All BUILD-ONLY (GitHub runners have no AMD/Intel/Adreno GPU, and
# no ctest — a GPU-linked jllama_test can't enumerate a device). GPU runtime libs
# are NOT bundled — the consumer's driver/toolkit supplies them. CMakeLists.txt
# routes each backend to its own src/main/resources_* tree; the matching Maven
# profile turns it into a classifier JAR. Toolchain install steps are first-pass —
# if a vendor URL/version 404s in CI, adjust it (the failure is intentional signal).
# ---------------------------------------------------------------------------
build-linux-x86_64-rocm:
name: Build Linux x86_64 ROCm/HIP (AMD)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The TheRock wheels unpack to several GB; upstream's ubuntu-rocm job frees the runner
# the same way. Runs first, before setup-java puts the JDK into the tool cache.
uses: ggml-org/free-disk-space@v1.3.1
with:
tool-cache: true
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install ROCm/HIP (TheRock wheels)
# KEEP IN SYNC WITH UPSTREAM: ROCM_VERSION tracks the ubuntu-rocm job in llama.cpp's
# .github/workflows/release.yml at the pinned GIT_TAG (GPU targets: see the build step). Since ROCm 7.14
# AMD builds and releases ROCm through TheRock (https://github.com/ROCm/TheRock), replacing
# the monolithic releases behind the repo.radeon.com apt repo this job used before (it was
# pinned at 6.3.4). The wheels carry the HIP runtime + CMake configs ("libraries") and the
# compilers/headers ("devel"); rocm-sdk reports where they landed.
env:
ROCM_VERSION: "10.0.0"
run: |
python3 -m venv "$RUNNER_TEMP/rocm-venv"
source "$RUNNER_TEMP/rocm-venv/bin/activate"
python -m pip install --upgrade pip
python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==${ROCM_VERSION}"
ROCM_PATH=$(rocm-sdk path --root)
echo "ROCM_PATH=$ROCM_PATH" >> "$GITHUB_ENV"
echo "HIP_PATH=$ROCM_PATH" >> "$GITHUB_ENV"
echo "CMAKE_PREFIX_PATH=$(rocm-sdk path --cmake)" >> "$GITHUB_ENV"
echo "LD_LIBRARY_PATH=$ROCM_PATH/lib:${LD_LIBRARY_PATH:-}" >> "$GITHUB_ENV"
echo "$(rocm-sdk path --bin)" >> "$GITHUB_PATH"
echo "$RUNNER_TEMP/rocm-venv/bin" >> "$GITHUB_PATH"
- name: Build libraries
shell: bash
# Native CMake HIP language (upstream's form): the HIP compiler is ROCm's clang, the C/C++
# TUs stay on the runner's gcc. The target list is every Linux target TheRock builds
# (its SUPPORTED_GPUS.md) — a superset of upstream llama.cpp's list, which omits
# gfx900/gfx906/gfx90c/gfx1153 ("build passing" only, not release-ready). Better too many
# than too few, but only while they cost nothing: drop an extra the moment it needs a
# patch or blocks a newer ROCm. Linux alone has the Instinct parts (gfx908/90a/942/950):
# ROCm does not support them on Windows.
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_HIP=ON -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGPU_TARGETS=gfx900;gfx906;gfx908;gfx90a;gfx90c;gfx942;gfx950;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Verify the GPU code is compressed
# ggml-hip is compiled with --offload-compress (llama/CMakeLists.txt); without it the
# library carries every target's kernels uncompressed. Fails on an uncompressed bundle.
run: python3 .github/verify-hip-offload-compressed.py llama/src/main/resources_linux_rocm
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_linux_rocm/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-rocm:
name: Build Windows x86_64 ROCm/HIP (AMD)
needs: [startgate, build-webui]
# windows-2022 (MSVC 14.4x), NOT windows-2025-vs2026 (VS 2026 / MSVC 14.51): ROCm 7.1's
# HIP clang headers (__clang_hip_cmath.h) cannot overload the __host__ __device__
# isgreater/isless/... that the very new MSVC <cmath> declares via _CLANG_BUILTIN2, so the
# device-code compile fails. Upstream llama.cpp builds win-hip on windows-2022 for the same
# reason (it drives ROCm's own clang and relies on the older MSVC STL).
runs-on: windows-2022
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install ROCm/HIP (TheRock wheels)
shell: pwsh
# KEEP IN SYNC WITH UPSTREAM: mirrors llama.cpp's windows-rocm release job
# (.github/actions/windows-setup-rocm at the pinned GIT_TAG) — the same TheRock wheels as
# the Linux job, in place of the former AMD-Software-PRO-Edition HIP SDK installer. The
# venv lives under C:\TheRock\build, the path upstream uses.
env:
ROCM_VERSION: "10.0.0"
run: |
$ErrorActionPreference = "Stop"
New-Item -Path "C:\TheRock\build" -ItemType Directory -Force | Out-Null
python -m venv C:\TheRock\build\.venv
& C:\TheRock\build\.venv\Scripts\Activate.ps1
python -m pip install --upgrade pip
python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==$env:ROCM_VERSION"
if ($LASTEXITCODE -ne 0) { throw "ROCm wheel install failed with exit code $LASTEXITCODE" }
rocm-sdk init
if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" }
$rocm = (rocm-sdk path --root)
if (-not $rocm) { throw "rocm-sdk path --root returned empty - devel package may not be installed" }
$rocm = $rocm.Trim()
"HIP_PATH=$rocm" | Out-File -FilePath $env:GITHUB_ENV -Append
"HIP_DEVICE_LIB_PATH=$rocm\lib\llvm\amdgcn\bitcode" | Out-File -FilePath $env:GITHUB_ENV -Append
"HIP_PLATFORM=amd" | Out-File -FilePath $env:GITHUB_ENV -Append
"LLVM_PATH=$rocm\lib\llvm" | Out-File -FilePath $env:GITHUB_ENV -Append
(rocm-sdk path --bin).Trim() | Out-File -FilePath $env:GITHUB_PATH -Append
"C:\TheRock\build\.venv\Scripts" | Out-File -FilePath $env:GITHUB_PATH -Append
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
# Wraps ROCm's clang (HIP is compiled as C++ on Windows, so the C/CXX launcher covers it).
# build.bat probes sccache before trusting it and, should a build fail with it as the
# launcher, retries once uncached -- so this can speed the job up but never red it.
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# Upstream's compiler wiring for TheRock (clang under lib\llvm\bin, not bin\);
# -Wno-error=incompatible-pointer-types is upstream's too. Targets: every Windows target
# TheRock builds — the Linux list minus the Instinct parts, which ROCm has no Windows
# support for. Same rule as on Linux for the extras upstream omits -- here only
# gfx900/gfx906/gfx90c, since upstream's windows-rocm list already has gfx1153.
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_HIP=ON -DGPU_TARGETS=gfx900;gfx906;gfx90c;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DCMAKE_PREFIX_PATH="%HIP_PATH%" -DHIP_PATH="%HIP_PATH%" -DCMAKE_C_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_CXX_COMPILER="%HIP_PATH%\lib\llvm\bin\clang++.exe" -DCMAKE_HIP_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Verify the GPU code is compressed
# Same check as the Linux ROCm job: an uncompressed bundle means ~1 GB jllama.dll again.
shell: pwsh
run: python .github/verify-hip-offload-compressed.py llama/src/main/resources_windows_rocm
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_windows_rocm/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-sycl-fp16:
name: Build Linux x86_64 SYCL fp16 (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install Intel oneAPI (DPC++ + MKL)
run: |
wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null
echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list
sudo apt-get update
sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel
- name: Build libraries
shell: bash
run: |
source /opt/intel/oneapi/setvars.sh
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-sycl-fp16
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp16/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-sycl-fp32:
name: Build Linux x86_64 SYCL fp32 (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- name: Free up disk space
# The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree;
# same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is
# kept (default) so nothing a later setup-* step relies on is removed.
uses: ggml-org/free-disk-space@v1.3.1
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install Intel oneAPI (DPC++ + MKL)
run: |
wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null
echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list
sudo apt-get update
sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel
- name: Build libraries
shell: bash
run: |
source /opt/intel/oneapi/setvars.sh
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-sycl-fp32
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp32/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-sycl:
name: Build Windows 2025 x86_64 SYCL (Intel oneAPI)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install Intel oneAPI (DPC++ + MKL + oneDNN + TBB)
shell: cmd
# Mirrors upstream llama.cpp's windows-sycl release job: extract the offline
# installer, then run its bootstrapper with the DPC++/MKL/oneDNN/TBB components.
run: |
curl -fSL -o "%RUNNER_TEMP%\oneapi.exe" "https://registrationcenter-download.intel.com/akdlm/IRC_NAS/b60765d1-2b85-4e85-86b6-cb0e9563a699/intel-deep-learning-essentials-2025.3.3.18_offline.exe"
"%RUNNER_TEMP%\oneapi.exe" -s -x -f "%RUNNER_TEMP%\oneapi_extracted" --log "%RUNNER_TEMP%\extract.log"
"%RUNNER_TEMP%\oneapi_extracted\bootstrapper.exe" -s --action install --components=intel.oneapi.win.cpp-dpcpp-common:intel.oneapi.win.mkl.devel:intel.oneapi.win.dnnl:intel.oneapi.win.tbb.devel --eula=accept -p=NEED_VS2022_INTEGRATION=0 --log-dir="%RUNNER_TEMP%"
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
# Wraps cl.exe (C) and icx (C++); if sccache cannot handle icx the retry builds uncached.
# build.bat probes sccache before trusting it and, should a build fail with it as the
# launcher, retries once uncached -- so this can speed the job up but never red it.
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
run: |
call "C:\Program Files (x86)\Intel\oneAPI\setvars.bat" intel64 --force
.github\build.bat -G "Ninja Multi-Config" -DGGML_SYCL=ON -DCMAKE_C_COMPILER=cl -DCMAKE_CXX_COMPILER=icx -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-sycl
path: ${{ github.workspace }}/llama/src/main/resources_windows_sycl/net/ladenthin/llama/
if-no-files-found: error
build-windows-arm64-opencl:
name: Build Windows 11 arm64 OpenCL (Adreno)
needs: [startgate, build-webui]
# Windows-on-ARM OpenCL (Snapdragon X / Adreno). Same clang-cl + GGML_OPENMP=OFF
# toolchain as the arm64 CPU job (ggml refuses MSVC cl.exe on ARM). Reuses the
# resources_windows_opencl tree under Windows/aarch64; the opencl-windows-aarch64
# Maven profile packages only that subtree. build_opencl_windows.bat stages the
# OpenCL headers + ICD loader before delegating to build.bat.
runs-on: windows-11-arm
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (arm64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: arm64
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
# sccache ships a native Windows-on-ARM build; it wraps clang-cl here.
# build.bat probes sccache before trusting it and, should a build fail with it as the
# launcher, retries once uncached -- so this can speed the job up but never red it.
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-aarch64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
run: |
.github\build_opencl_windows.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=aarch64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-aarch64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
if-no-files-found: error
build-linux-x86_64-openvino:
name: Build Linux x86_64 OpenVINO (Intel)
needs: [startgate, build-webui]
runs-on: ubuntu-latest
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Install OpenCL dev + Intel OpenVINO 2026.4 (archive)
run: |
# Intel's OpenVINO APT repo only publishes up to ~2025 (the /openvino/2026 path 404s), and
# 2025.x has the older ov::Allocator API that breaks ggml-openvino's template compile. So use
# the ARCHIVE — exactly what upstream llama.cpp's linux-setup-openvino action does, from the
# same URL template.
#
# KEEP IN SYNC WITH UPSTREAM. The version tracks llama.cpp's own OPENVINO_VERSION_MAJOR /
# OPENVINO_VERSION_FULL (.github/workflows/release.yml at the pinned GIT_TAG); ggml-openvino
# is developed against that pair, so lagging it is what eventually breaks the compile. Both
# OpenVINO jobs here (Linux + Windows) use the same two values — bump them together:
# major = 2026.4 full = 2026.4.0.22959.99c81491cc3
# OpenCL headers (incl. the C++ CL/cl2.hpp via opencl-clhpp-headers) come from Ubuntu's own repos.
sudo apt-get update
sudo apt-get install -y ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd
url="https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/linux/openvino_toolkit_ubuntu24_2026.4.0.22959.99c81491cc3_x86_64.tgz"
sudo mkdir -p /opt/intel/openvino
curl -fSL "$url" | sudo tar -xz --strip-components=1 -C /opt/intel/openvino
echo "OpenVINO_DIR=/opt/intel/openvino/runtime/cmake" >> "$GITHUB_ENV"
- name: Build libraries
shell: bash
run: |
source /opt/intel/openvino/setupvars.sh || true
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh "-DGGML_OPENVINO=ON -DOpenVINO_DIR=$OpenVINO_DIR -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64"
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Linux-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_linux_openvino/net/ladenthin/llama/
if-no-files-found: error
build-windows-x86_64-openvino:
name: Build Windows 2025 x86_64 OpenVINO (Intel)
needs: [startgate, build-webui]
runs-on: windows-2025-vs2026
env:
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- name: Set up MSVC developer environment (x64)
uses: ilammy/msvc-dev-cmd@v1
with:
arch: x64
- name: Install OpenCL headers (vcpkg) + Intel OpenVINO 2026.4
shell: pwsh
# vcpkg's opencl port ships the full C++ headers incl. CL/cl2.hpp that OpenVINO's
# ocl_wrapper.hpp needs (the Khronos OpenCL-Headers dropped cl2.hpp) — same as upstream
# llama.cpp's windows-openvino job. OpenVINO 2026.4 matches ggml-openvino's target API.
# Keep the version in sync with the Linux OpenVINO job above (and with upstream's
# OPENVINO_VERSION_MAJOR / OPENVINO_VERSION_FULL) — see the note there.
run: |
C:\vcpkg\vcpkg install opencl:x64-windows
$url = "https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/windows/openvino_toolkit_windows_2026.4.0.22959.99c81491cc3_x86_64.zip"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\openvino.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\openvino.zip" -DestinationPath "C:\openvino" -Force
# The archive extracts into a nested versioned folder; point OpenVINO_DIR at its runtime/cmake.
$root = (Get-ChildItem "C:\openvino" -Directory | Select-Object -First 1).FullName
"OpenVINO_DIR=$root\runtime\cmake" | Out-File -FilePath $env:GITHUB_ENV -Append
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
shell: pwsh
# Plain cl.exe build, same as the Vulkan/OpenCL jobs.
# build.bat probes sccache before trusting it and, should a build fail with it as the
# launcher, retries once uncached -- so this can speed the job up but never red it.
run: |
$ver = "0.18.0"
$rel = "sccache-v$ver-x86_64-pc-windows-msvc"
$url = "https://github.com/mozilla/sccache/releases/download/v$ver/$rel.zip"
Write-Host "Downloading $url"
Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\sccache.zip"
Expand-Archive -Path "$env:RUNNER_TEMP\sccache.zip" -DestinationPath "$env:RUNNER_TEMP\sccache" -Force
Add-Content -Path $env:GITHUB_PATH -Value "$env:RUNNER_TEMP\sccache\$rel"
- name: Build libraries
shell: cmd
# vcpkg toolchain file wires in the OpenCL (incl. cl2.hpp) that ggml-openvino needs.
run: |
.github\build.bat -G "Ninja Multi-Config" -DGGML_OPENVINO=ON -DOpenVINO_DIR="%OpenVINO_DIR%" -DCMAKE_TOOLCHAIN_FILE=C:\vcpkg\scripts\buildsystems\vcpkg.cmake -DOS_NAME=Windows -DOS_ARCH=x86_64
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: Windows-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
if-no-files-found: error
# ---------------------------------------------------------------------------
# CI-only jobs — no release artifact, purely for test coverage
# ---------------------------------------------------------------------------
test-cpp-linux-x86_64:
name: C++ Tests Ubuntu Latest x86_64
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Display CPU Info
run: |
echo "=== CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- name: Build libraries
run: |
mvn -q --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DBUILD_TESTING=ON
# Every patch has a runnable guard that reds this job if it goes missing (a link error, or
# test_utils.cpp / test_model_split.cpp / test_common_log_callback.cpp). This is the second
# line: it asserts the applier actually ran and nothing reverted the patched tree, which no
# per-patch guard covers directly. Model-free, milliseconds.
- name: Verify llama.cpp patches are applied
run: .github/verify-patches-applied.sh
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
build-macos-arm64-metal-15:
name: Build and Test macOS 15 arm64 (Metal)
needs: [startgate, build-webui]
runs-on: macos-15
env:
BUILD_JOBS: 2
USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }}
SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev
SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }}
steps:
- uses: actions/checkout@v7
- name: Download shared WebUI assets
uses: actions/download-artifact@v8
with:
name: webui-generated
path: ${{ github.workspace }}/llama/webui-generated/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- name: Install sccache (shared compiler cache)
if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != ''
continue-on-error: true
run: brew install sccache
- name: Build libraries
shell: bash
run: |
mvn --no-transfer-progress -f llama/pom.xml compile
.github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DGGML_NATIVE=OFF -DBUILD_TESTING=ON
- name: Run C++ unit tests
run: ctest --test-dir llama/build --output-on-failure
- name: Upload artifacts
uses: actions/upload-artifact@v7
with:
name: macos-15-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# ---------------------------------------------------------------------------
# Java test jobs — download release artifact, run mvn test
# ---------------------------------------------------------------------------
test-java-linux-x86_64:
name: Java Tests Ubuntu Latest x86_64
needs: [crosscompile-linux-x86_64, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
lscpu
echo ""
echo "=== CPU Details from /proc/cpuinfo ==="
cat /proc/cpuinfo
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. GGUF is platform-independent, so ubuntu + macOS + Windows share
# one entry (key gguf-models-<manifest hash>). validate-models is kept as an integrity guard so a
# partial/absent restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: bash .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: free -h
- name: Enable core dumps
run: |
ulimit -c unlimited
echo "${{ github.workspace }}/core.%e.%p" | sudo tee /proc/sys/kernel/core_pattern
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml -P jcstress test \
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME} \
-Dnet.ladenthin.llama.nomic.path=models/${NOMIC_EMBED_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.model=models/${VISION_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.mmproj=models/${VISION_MMPROJ_NAME} \
-Dnet.ladenthin.llama.vision.image=${VISION_IMAGE_PATH} \
-Dnet.ladenthin.llama.tts.model=models/${TTS_MODEL_NAME} \
-Dnet.ladenthin.llama.tts.mmproj=models/${TTS_MMPROJ_NAME} \
-Dnet.ladenthin.llama.train.model=models/${TRAIN_MODEL_NAME}
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- uses: actions/upload-artifact@v7
if: success()
with:
name: jacoco-report
path: llama/target/site/jacoco/jacoco.xml
if-no-files-found: ignore
- name: Run PIT mutation tests
run: mvn --batch-mode --no-transfer-progress -f llama/pom.xml test-compile org.pitest:pitest-maven:mutationCoverage
- name: Extract PIT survivors
if: always()
run: |
echo "=== PIT Survived Mutations ==="
for html_file in $(find llama/target/pit-reports -name "*.html" -type f 2>/dev/null | sort); do
if grep -q "SURVIVED" "$html_file"; then
echo "Found survivors in $html_file:"
grep -B 2 -A 3 "SURVIVED" "$html_file"
echo ""
fi
done
- uses: actions/upload-artifact@v7
if: always()
with: { name: pit-reports, path: llama/target/pit-reports/ }
- name: Memory after tests
if: always()
run: free -h
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-linux-x86_64
path: |
${{ github.workspace }}/llama/hs_err_pid*.log
${{ github.workspace }}/core.*
${{ github.workspace }}/llama/*.hprof
${{ github.workspace }}/llama/target/surefire-reports/*.dump
${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama/target/surefire-reports/*.txt
${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
# ---------------------------------------------------------------------------
# vmlens interleaving analysis — pure-Java, needs no native library or models.
# Staged to a single smoke test for now (see the `vmlens` profile in pom.xml).
# ---------------------------------------------------------------------------
vmlens:
name: Test (vmlens interleavings)
needs: startgate
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
cache: maven
- name: Test under vmlens (interleaving analysis)
# Add each new test in the `vmlens` package to this -Dtest list (surefire
# -Dtest matches simple class names, not package globs; the default suite is
# excluded from the vmlens package via pom.xml managed surefire <excludes>).
run: >-
mvn --batch-mode --no-transfer-progress -f llama/pom.xml -Pvmlens test
-Dtest=VmlensInterleavingSmokeTest,SessionStateInterleavingTest -DfailIfNoTests=false
- uses: actions/upload-artifact@v7
if: always()
with:
name: vmlens-report
path: llama/target/vmlens-report/
if-no-files-found: ignore
test-java-macos-arm64-metal:
name: Java Tests macOS 14 arm64 (Metal)
needs: [build-macos-arm64-metal, verify-model-cache]
runs-on: macos-14
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- uses: actions/download-artifact@v8
with:
name: macos-14-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. GGUF is platform-independent, so ubuntu + macOS + Windows share
# one entry (key gguf-models-<manifest hash>). validate-models is kept as an integrity guard so a
# partial/absent restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: bash .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: vm_stat && sysctl hw.memsize hw.physmem
- name: Enable core dumps
run: ulimit -c unlimited
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml -Dnet.ladenthin.llama.test.ngl=0 test \
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME} \
-Dnet.ladenthin.llama.nomic.path=models/${NOMIC_EMBED_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.model=models/${VISION_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.mmproj=models/${VISION_MMPROJ_NAME} \
-Dnet.ladenthin.llama.vision.image=${VISION_IMAGE_PATH} \
-Dnet.ladenthin.llama.tts.model=models/${TTS_MODEL_NAME} \
-Dnet.ladenthin.llama.tts.mmproj=models/${TTS_MMPROJ_NAME} \
-Dnet.ladenthin.llama.train.model=models/${TRAIN_MODEL_NAME}
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- name: Memory after tests
if: always()
run: vm_stat && sysctl hw.memsize hw.physmem
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-macos-14-metal
path: |
${{ github.workspace }}/llama/hs_err_pid*.log
${{ github.workspace }}/llama/*.hprof
${{ github.workspace }}/llama/target/surefire-reports/*.dump
${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama/target/surefire-reports/*.txt
${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
test-java-macos-arm64-no-metal:
name: Java Tests macOS 15 arm64 (no Metal)
needs: [build-macos-arm64-no-metal, verify-model-cache]
runs-on: macos-15
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- uses: actions/download-artifact@v8
with:
name: macos-15-no-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. GGUF is platform-independent, so ubuntu + macOS + Windows share
# one entry (key gguf-models-<manifest hash>). validate-models is kept as an integrity guard so a
# partial/absent restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: bash .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: vm_stat && sysctl hw.memsize hw.physmem
- name: Enable core dumps
run: ulimit -c unlimited
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml test \
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME} \
-Dnet.ladenthin.llama.nomic.path=models/${NOMIC_EMBED_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.model=models/${VISION_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.mmproj=models/${VISION_MMPROJ_NAME} \
-Dnet.ladenthin.llama.vision.image=${VISION_IMAGE_PATH} \
-Dnet.ladenthin.llama.tts.model=models/${TTS_MODEL_NAME} \
-Dnet.ladenthin.llama.tts.mmproj=models/${TTS_MMPROJ_NAME} \
-Dnet.ladenthin.llama.train.model=models/${TRAIN_MODEL_NAME}
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- name: Memory after tests
if: always()
run: vm_stat && sysctl hw.memsize hw.physmem
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-macos-15-no-metal
path: |
${{ github.workspace }}/llama/hs_err_pid*.log
${{ github.workspace }}/llama/*.hprof
${{ github.workspace }}/llama/target/surefire-reports/*.dump
${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama/target/surefire-reports/*.txt
${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
test-java-macos-arm64-metal-15:
name: Java Tests macOS 15 arm64 (Metal)
needs: [build-macos-arm64-metal-15, verify-model-cache]
runs-on: macos-15
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: bash
run: |
echo "=== CPU Information ==="
sysctl hw.model hw.cachelinesize hw.cpufrequency hw.cachesize hw.physicalcpu hw.logicalcpu hw.packages hw.memsize hw.ncpu 2>/dev/null || true
echo ""
echo "=== Processor Details ==="
system_profiler SPHardwareDataType
- uses: actions/download-artifact@v8
with:
name: macos-15-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. GGUF is platform-independent, so ubuntu + macOS + Windows share
# one entry (key gguf-models-<manifest hash>). validate-models is kept as an integrity guard so a
# partial/absent restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: bash .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: vm_stat && sysctl hw.memsize hw.physmem
- name: Enable core dumps
run: ulimit -c unlimited
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml test \
-Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME} \
-Dnet.ladenthin.llama.nomic.path=models/${NOMIC_EMBED_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.model=models/${VISION_MODEL_NAME} \
-Dnet.ladenthin.llama.vision.mmproj=models/${VISION_MMPROJ_NAME} \
-Dnet.ladenthin.llama.vision.image=${VISION_IMAGE_PATH} \
-Dnet.ladenthin.llama.tts.model=models/${TTS_MODEL_NAME} \
-Dnet.ladenthin.llama.tts.mmproj=models/${TTS_MMPROJ_NAME} \
-Dnet.ladenthin.llama.train.model=models/${TRAIN_MODEL_NAME}
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- name: Memory after tests
if: always()
run: vm_stat && sysctl hw.memsize hw.physmem
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: error-log-macos-15-metal
path: |
${{ github.workspace }}/llama/hs_err_pid*.log
${{ github.workspace }}/llama/*.hprof
${{ github.workspace }}/llama/target/surefire-reports/*.dump
${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream
${{ github.workspace }}/llama/target/surefire-reports/*.txt
${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml
if-no-files-found: warn
test-java-windows-x86_64:
name: Java Tests Windows 2025 x86_64 (default / Ninja)
needs: [build-windows-x86_64, verify-model-cache]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-libraries
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. validate-models is kept as an integrity guard so a partial/absent
# restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github\validate-models.bat
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: Get-CimInstance Win32_OperatingSystem | Select-Object FreePhysicalMemory,TotalVisibleMemorySize | Format-List
shell: pwsh
- name: Enable WER LocalDumps for java.exe
# Windows Error Reporting writes minidumps when java.exe (or any other
# registered process) crashes via __fastfail / abort / unhandled SEH.
# We use it as the Windows analogue of Linux core dumps so that a JVM
# crash inside the JNI layer leaves us a real native callstack instead
# of just surefire's "VM terminated without saying goodbye" line.
# DumpType=2 == MiniDumpWithFullMemory; the workspace dumps/ folder is
# globbed by the failure-upload step below.
shell: pwsh
run: |
$key = 'HKLM:\SOFTWARE\Microsoft\Windows\Windows Error Reporting\LocalDumps\java.exe'
New-Item -Path $key -Force | Out-Null
New-Item -Path "${{ github.workspace }}\dumps" -ItemType Directory -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpFolder' -Value "${{ github.workspace }}\dumps" -PropertyType ExpandString -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpType' -Value 2 -PropertyType DWord -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpCount' -Value 5 -PropertyType DWord -Force | Out-Null
Get-ItemProperty -Path $key | Format-List
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml test `
"-Dnet.ladenthin.llama.tool.model=models/$env:TOOL_MODEL_NAME" `
"-Dnet.ladenthin.llama.nomic.path=models/$env:NOMIC_EMBED_MODEL_NAME" `
"-Dnet.ladenthin.llama.vision.model=models/$env:VISION_MODEL_NAME" `
"-Dnet.ladenthin.llama.vision.mmproj=models/$env:VISION_MMPROJ_NAME" `
"-Dnet.ladenthin.llama.vision.image=$env:VISION_IMAGE_PATH" `
"-Dnet.ladenthin.llama.tts.model=models/$env:TTS_MODEL_NAME" `
"-Dnet.ladenthin.llama.tts.mmproj=models/$env:TTS_MMPROJ_NAME" `
"-Dnet.ladenthin.llama.train.model=models/$env:TRAIN_MODEL_NAME"
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
shell: bash
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- name: Memory after tests
if: always()
run: Get-CimInstance Win32_OperatingSystem | Select-Object FreePhysicalMemory,TotalVisibleMemorySize | Format-List
shell: pwsh
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: windows-output
path: |
${{ github.workspace }}\llama\hs_err_pid*.log
${{ github.workspace }}\llama\*.hprof
${{ github.workspace }}\dumps\*.dmp
${{ github.workspace }}\llama\target\surefire-reports\*.dump
${{ github.workspace }}\llama\target\surefire-reports\*.dumpstream
${{ github.workspace }}\llama\target\surefire-reports\*.txt
${{ github.workspace }}\llama\target\surefire-reports\TEST-*.xml
${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/**/*
if-no-files-found: warn
# Java/inference validation of the MSVC-built x86_64 DLL (the analogue of
# test-java-windows-x86_64 for the default Ninja build). Loads the MSVC jllama.dll
# via JNI and runs the full model-backed suite, so both Windows generators are
# validated end-to-end before the `msvc-windows` classifier JAR ships.
test-java-windows-x86_64-msvc:
name: Java Tests Windows 2025 x86_64 (MSVC classifier)
needs: [build-windows-x86_64-msvc, verify-model-cache]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- name: Display CPU Info
shell: pwsh
run: |
Write-Host "=== CPU Information (Get-CimInstance - All Properties) ==="
Get-CimInstance Win32_Processor | Select-Object * | Format-List
Write-Host ""
Write-Host "=== CPU Information (systeminfo) ==="
systeminfo | Select-String "Processor"
Write-Host ""
Write-Host "=== CPU Information (Get-ComputerInfo) ==="
Get-ComputerInfo -Property "CsProcessors*" 2>$null || Write-Host "Get-ComputerInfo not available"
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-msvc
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
# GGUF models are downloaded + cached ONCE by the upstream `download-models` job
# (this job `needs:` it), so here we only RESTORE the shared cache — no per-job
# download logic. validate-models is kept as an integrity guard so a partial/absent
# restore fails loudly instead of silently self-skipping required tests.
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github\validate-models.bat
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Memory before tests
run: Get-CimInstance Win32_OperatingSystem | Select-Object FreePhysicalMemory,TotalVisibleMemorySize | Format-List
shell: pwsh
- name: Enable WER LocalDumps for java.exe
# Windows Error Reporting writes minidumps when java.exe (or any other
# registered process) crashes via __fastfail / abort / unhandled SEH.
# We use it as the Windows analogue of Linux core dumps so that a JVM
# crash inside the JNI layer leaves us a real native callstack instead
# of just surefire's "VM terminated without saying goodbye" line.
# DumpType=2 == MiniDumpWithFullMemory; the workspace dumps/ folder is
# globbed by the failure-upload step below.
shell: pwsh
run: |
$key = 'HKLM:\SOFTWARE\Microsoft\Windows\Windows Error Reporting\LocalDumps\java.exe'
New-Item -Path $key -Force | Out-Null
New-Item -Path "${{ github.workspace }}\dumps" -ItemType Directory -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpFolder' -Value "${{ github.workspace }}\dumps" -PropertyType ExpandString -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpType' -Value 2 -PropertyType DWord -Force | Out-Null
New-ItemProperty -Path $key -Name 'DumpCount' -Value 5 -PropertyType DWord -Force | Out-Null
Get-ItemProperty -Path $key | Format-List
- name: Run tests
run: |
mvn -e --no-transfer-progress -f llama/pom.xml test `
"-Dnet.ladenthin.llama.tool.model=models/$env:TOOL_MODEL_NAME" `
"-Dnet.ladenthin.llama.nomic.path=models/$env:NOMIC_EMBED_MODEL_NAME" `
"-Dnet.ladenthin.llama.vision.model=models/$env:VISION_MODEL_NAME" `
"-Dnet.ladenthin.llama.vision.mmproj=models/$env:VISION_MMPROJ_NAME" `
"-Dnet.ladenthin.llama.vision.image=$env:VISION_IMAGE_PATH" `
"-Dnet.ladenthin.llama.tts.model=models/$env:TTS_MODEL_NAME" `
"-Dnet.ladenthin.llama.tts.mmproj=models/$env:TTS_MMPROJ_NAME" `
"-Dnet.ladenthin.llama.train.model=models/$env:TRAIN_MODEL_NAME"
# Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that
# fails makes Surefire record tests="0" for that class -- no entries at all, so it is
# invisible to a skip check. That is exactly how every model-gated class stayed silently
# muted on every test-java-* job for months. The floor is deliberately slack; the per-class
# zero check is the sensitive half.
- name: Verify tests actually ran
shell: bash
run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500
- name: Memory after tests
if: always()
run: Get-CimInstance Win32_OperatingSystem | Select-Object FreePhysicalMemory,TotalVisibleMemorySize | Format-List
shell: pwsh
# A forked test JVM that aborts leaves an hs_err_pid log and a surefire
# dumpstream -- both otherwise ONLY inside the artifact uploaded below,
# which is unreachable from anywhere that cannot fetch from Azure Blob
# (a phone, a restricted network, an agent sandbox). Echo them here so the
# aborting frame is readable from the run page itself. See
# ../workspace/policies/ci-test-diagnostics.md section 3.1.
- name: Print crash logs (on failure)
if: failure()
shell: bash
run: |
shopt -s nullglob
found=0
for f in llama/hs_err_pid*.log; do
found=1
echo "===== $f (first 200 lines; full file in the uploaded artifact) ====="
sed -n '1,200p' "$f"
done
for f in llama/target/surefire-reports/*.dumpstream llama/target/surefire-reports/*.dump; do
found=1
echo "===== $f ====="
cat "$f"
done
if [ "$found" = 0 ]; then
echo "No hs_err_pid*.log and no surefire dump/dumpstream was written."
echo
echo "For an ordinary test failure that is EXPECTED, not a finding: this step runs on"
echo "any job failure, and an assertion failure, a timeout or a compile error writes no"
echo "crash log. Read the surefire output above for the real cause."
echo
echo "It points at a JVM-level abort only if the log ALSO shows a fork ending abnormally"
echo "-- 'The forked VM terminated without properly saying goodbye', or an exit with no"
echo "test results. In that case the abort bypassed the JVM error handler (a native"
echo "exit()/terminate() rather than a raised signal), which is why no file was written."
fi
- if: failure()
uses: actions/upload-artifact@v7
with:
name: windows-output-msvc
path: |
${{ github.workspace }}\llama\hs_err_pid*.log
${{ github.workspace }}\llama\*.hprof
${{ github.workspace }}\dumps\*.dmp
${{ github.workspace }}\llama\target\surefire-reports\*.dump
${{ github.workspace }}\llama\target\surefire-reports\*.dumpstream
${{ github.workspace }}\llama\target\surefire-reports\*.txt
${{ github.workspace }}\llama\target\surefire-reports\TEST-*.xml
${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/**/*
if-no-files-found: warn
# ---------------------------------------------------------------------------
# Package and publish
# ---------------------------------------------------------------------------
package:
name: Package JARs
needs:
- crosscompile-linux-x86_64-cuda
- crosscompile-linux-aarch64
- build-linux-s390x
- build-linux-x86_64-vulkan
- build-linux-aarch64-vulkan
- crosscompile-android-aarch64
- crosscompile-android-x86_64
- crosscompile-android-aarch64-opencl
- build-windows-x86_64
- build-windows-x86
- build-windows-arm64
- build-windows-x86_64-msvc
- build-windows-x86-msvc
- build-windows-x86_64-cuda
- build-windows-x86_64-vulkan
- build-windows-x86_64-opencl
- build-linux-x86_64-rocm
- build-windows-x86_64-rocm
- build-linux-x86_64-sycl-fp16
- build-linux-x86_64-sycl-fp32
- build-windows-x86_64-sycl
- build-windows-arm64-opencl
- build-linux-x86_64-openvino
- build-windows-x86_64-openvino
- test-cpp-linux-x86_64
- build-macos-arm64-metal-15
- test-java-linux-x86_64
- test-java-macos-arm64-metal
- test-java-macos-arm64-no-metal
- test-java-macos-arm64-metal-15
- test-java-windows-x86_64
- test-java-windows-x86_64-msvc
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
# Downloaded UNMERGED (one subdirectory per artifact name) on purpose: `merge-multiple: true`
# would silently overwrite — and can byte-level interleave — two artifacts that carry the same
# relative path, which is how the corrupt macOS dylib shipped. merge-native-artifacts.sh does
# the merge instead, and fails the job if any {OS}/{ARCH} path is claimed by more than one
# `*-libraries` artifact. A post-merge check cannot catch this (the collision leaves exactly
# one file on the path — a corrupt one), so the check has to happen before the merge.
- uses: actions/download-artifact@v8
with:
pattern: "*-libraries"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the default resource tree (collision-checked)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/"
# All three macOS arm64 build jobs emit the same path Mac/aarch64/libjllama.dylib, so their
# artifact names are kept outside the "*-libraries" glob above — merging them would drop
# three different dylibs onto one file. The variant that ships is selected explicitly:
# macos-15-metal is the only one built with both Metal and GGML_NATIVE=OFF (portable across
# CPU generations); macos-14-metal and macos-15-no-metal remain test-only.
- uses: actions/download-artifact@v8
with:
name: macos-15-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: linux-libraries-cuda
path: ${{ github.workspace }}/llama/src/main/resources_linux_cuda/net/ladenthin/llama/
# Linux Vulkan classifiers (x86_64 + aarch64) share one tree; the two Maven profiles
# split it by arch subdir into one single-arch classifier JAR each.
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-aarch64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: android-libraries-opencl
path: ${{ github.workspace }}/llama/src/main/resources_android_opencl/net/ladenthin/llama/
# MSVC-built Windows natives -> `msvc-windows` classifier tree. The default JAR
# now ships the Ninja `*-libraries` natives merged above (default flip).
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
# Windows GPU classifiers (x86_64 only) -> one tree each.
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-cuda
path: ${{ github.workspace }}/llama/src/main/resources_windows_cuda/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_windows_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_linux_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_windows_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp16
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp16/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp32
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp32/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-sycl
path: ${{ github.workspace }}/llama/src/main/resources_windows_sycl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-aarch64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_linux_openvino/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
# Runtime dependencies of every shipped native library, read from the file itself (ELF
# DT_NEEDED, PE imports incl. Windows arm64, Mach-O load commands). The default tree must
# match an exact per-{OS}/{ARCH} allowlist -- a new dependency is a load failure on every
# machine that lacks it. The classifier trees legitimately need their vendor runtime, so
# there only the denylist applies. The case this was written for: ggml-rpc's RDMA
# transport, which upstream enables whenever the build host has libibverbs/librdma
# (llama/CMakeLists.txt forces it off).
- name: Verify native runtime dependencies
run: |
python3 .github/verify-native-deps.py --default llama/src/main/resources
python3 .github/verify-native-deps.py --deny llama/src/main/resources_*
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Build JARs
# `assembly` additionally produces the fat jar-with-dependencies uber JAR
# (llama-<version>-jar-with-dependencies.jar: library classes + Java runtime deps +
# default-platform native libs in one drop-on-classpath JAR, runnable via its
# ServerLauncher Main-Class). It lands in target/ and is uploaded in the `llama-jars`
# artifact below - never a Maven Central asset; the `package-fatjars` job downstream
# combines it with the classifier jars into the GitHub-Release fat-jar assets.
# Windows classifier JARs: `windows-msvc` (MSVC-built CPU natives) plus the GPU
# backends `cuda-windows` / `vulkan-windows` / `opencl-windows`. The default JAR's
# Windows natives are the Ninja `*-libraries` merged into src/main/resources/ above.
run: mvn --batch-mode --no-transfer-progress -P release,cuda,vulkan-linux,vulkan-linux-aarch64,opencl-android,windows-msvc,cuda-windows,vulkan-windows,opencl-windows,rocm-linux,rocm-windows,sycl-fp16-linux,sycl-fp32-linux,sycl-windows,opencl-windows-aarch64,openvino-linux,openvino-windows,assembly -Dmaven.test.skip=true -Dgpg.skip=true package
# Class-file floor, checked on every jar this job just built (all 16 classifier
# jars plus the default fat jar). Production code targets Java 8, so anything a
# consumer's JVM can load must be major 52 or lower: a single Java 11 class kills
# the process with UnsupportedClassVersionError before any of our code runs, which
# is exactly what shipped when logback's LogbackServiceProvider was the binding.
# module-info.class and META-INF/versions/** are skipped unconditionally because a
# classpath JVM never loads them. Kept byte-identical across all four sibling repos.
- name: Verify Java 8 bytecode (no class newer than major 52)
run: .github/verify-bytecode-version.sh --max-major 52 llama/target
- name: Upload JARs
uses: actions/upload-artifact@v7
with:
name: llama-jars
path: llama/target/*.jar
package-fatjars:
name: Package all-backends fat jars
needs: [package]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-jars
path: jars/
# One self-contained multi-backend server fat jar per OS/arch
# (llama-<version>-all-<os>-<arch>-jar-with-dependencies.jar): the default
# jar-with-dependencies plus every GPU backend of that OS/arch in backend
# subdirectories, described by the jllama-backends.txt priority manifest that
# LlamaLoader reads at runtime (first loadable backend wins, CPU fallback).
# These are GitHub-Release download assets ONLY - never deployed to Maven
# Central (the deploy jobs run without the `assembly` profile and are untouched).
# The script enumerates the classifiers from llama/pom.xml and fails loud on any
# mismatch with the built classifier jars, so a new classifier cannot be silently
# skipped.
- name: Assemble all-backends fat jars
run: bash .github/package-fatjars.sh jars fatjars llama/pom.xml
- name: Upload fat jars
uses: actions/upload-artifact@v7
with:
name: llama-fatjars
path: fatjars/
compression-level: 0 # jars are already deflated
retention-days: 7 # multi-GB artifact; release jobs consume it within the same run
if-no-files-found: error
# Small single-jar artifacts so the smoke jobs don't download the multi-GB set.
- name: Upload Linux smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-linux
path: fatjars/llama-*-all-linux-x86-64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
- name: Upload Windows smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-windows
path: fatjars/llama-*-all-windows-x86-64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
# The two aarch64 fat jars are release assets exactly like the x86-64 pair, so they get the
# same treatment: package-fatjars emits four, and for a long time only the two x86-64 ones
# were ever launched. GitHub's free ARM runners (ubuntu-24.04-arm / windows-11-arm) are
# already used by the aarch64 build jobs, so there is no reason to leave these unrun.
- name: Upload Linux aarch64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-linux-aarch64
path: fatjars/llama-*-all-linux-aarch64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
- name: Upload Windows arm64 smoke jar
uses: actions/upload-artifact@v7
with:
name: llama-fatjar-smoke-windows-arm64
path: fatjars/llama-*-all-windows-aarch64-jar-with-dependencies.jar
compression-level: 0
retention-days: 7
if-no-files-found: error
# GPU-less runners: every manifest backend must fail its load cleanly (missing vendor
# runtimes) and the server must come up on the default CPU natives — this exercises
# manifest parsing, per-backend extraction, and the fallback chain end-to-end through
# a real `java -jar`, then proves the server serves /health + /v1/chat/completions.
smoke-fatjar-linux:
name: Smoke test all-backends fat jar (Linux)
needs: [package-fatjars, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-linux
path: fatjars/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
# Same floor, re-checked on the ASSEMBLED release asset rather than the module
# jars the `package` job verified: package-fatjars rewrites the zip (backend native
# trees + the jllama-backends.txt manifest), and this is the artifact users download.
- name: Verify Java 8 bytecode (no class newer than major 52)
run: .github/verify-bytecode-version.sh --max-major 52 fatjars
- name: Run fat-jar server smoke test
run: .github/smoke-test-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
# RPC over two JVMs from the same release asset: RpcServer in one, the default NativeServer
# with --rpc in the other. Requires a chat completion, the model buffer on the RPC endpoint in
# the load log (the layers really went over RPC), an accepted client on the server, and a clean
# non-SIGABRT failure naming the endpoint when --rpc names a server nobody runs.
- name: Run fat-jar RPC smoke test (two JVMs)
run: .github/smoke-rpc-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: fatjar-smoke-linux-logs
path: |
server-out.log
server-err.log
rpc-server.log
rpc-client-out.log
rpc-client-err.log
rpc-unreachable.log
if-no-files-found: warn
# The agent release asset, launched the way the README tells a user to: `java -jar` on the agent
# jar lying next to the real all-backends Linux fat jar, so the core is found only through the
# agent manifest's Class-Path. Proves the two assets fit together (matching version in the file
# names, nothing missing on either side), that the agent jar carries no core, and that a
# one-shot answer and a read_file tool round work through the shipped jars.
smoke-agent-linux:
name: Smoke test the agent release jar (Linux)
needs: [test-java-llama-atmosphere-agent, package-fatjars, verify-model-cache]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: agent-assets/
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-linux
path: agent-assets/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
# The agent is Java 21 (Atmosphere's floor), unlike the Java 8 core: its ceiling is 65.
# One waiver: JLine ships its FFM terminal provider (org/jline/terminal/impl/ffm, 25 classes)
# as Java 22 bytecode. It is discovered through META-INF/jline/providers/ffm, not loaded
# eagerly: on Java 21 JLine picks its JNI provider, and a forced FFM provider falls back to a
# dumb terminal instead of failing (verified on 21.0.10). Only that package is waived, so
# any other class above 65 still fails the gate.
- name: Verify Java 21 bytecode (no class newer than major 65)
run: >
.github/verify-bytecode-version.sh --max-major 65
--allow 'llama-atmosphere-agent-*:org/jline/terminal/impl/ffm/*'
agent-assets/llama-atmosphere-agent-*-jar-with-dependencies.jar
- name: Run the agent release-jar smoke test
run: .github/smoke-agent-jar.sh agent-assets "models/${TOOL_MODEL_NAME}"
- name: Upload agent logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: agent-smoke-linux-logs
path: agent-*.log
if-no-files-found: warn
smoke-fatjar-windows:
name: Smoke test all-backends fat jar (Windows)
needs: [package-fatjars, verify-model-cache]
runs-on: windows-2025-vs2026
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-windows
path: fatjars/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so Windows/macOS restore the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github\validate-models.bat
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Run fat-jar server smoke test
shell: pwsh
run: .github/smoke-test-fatjar.ps1 -JarDir fatjars -JarGlob 'llama-*-all-windows-x86-64-jar-with-dependencies.jar' -Model "models/$env:DRAFT_MODEL_NAME"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: fatjar-smoke-windows-logs
path: |
server-out.log
server-err.log
if-no-files-found: warn
# The aarch64 halves of the same convention. Identical in shape to the two jobs above — the only
# differences are the runner and the jar glob — and they exist because `all-linux-aarch64` and
# `all-windows-aarch64` were built, GPG-signed and attached to every release without CI ever
# launching them, which is exactly what workspace/policies/fat-jar-release-assets.md forbids
# ("No release asset is attached that CI has not run"). That rule exists because a corrupt macOS
# dylib shipped in three releases under a fully green pipeline; these two jars were the remaining
# assets in the same blind spot.
smoke-fatjar-linux-aarch64:
name: Smoke test all-backends fat jar (Linux aarch64)
needs: [package-fatjars, verify-model-cache]
runs-on: ubuntu-24.04-arm
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-linux-aarch64
path: fatjars/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: consumer jobs can NEVER write the cache, so a job running on a
# cache miss cannot re-save an empty/partial entry under the immutable key.
# download-models is the single writer; enableCrossOsArchive matches its
# cross-OS entry version so every consumer restores the same ubuntu-built entry.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github/validate-models.sh
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
# Same floor, re-checked on this assembled release asset: package-fatjars rewrites the zip
# per OS/arch, so the aarch64 jar is a different artifact from the one smoke-fatjar-linux
# verifies even though the classes are identical.
- name: Verify Java 8 bytecode (no class newer than major 52)
run: .github/verify-bytecode-version.sh --max-major 52 fatjars
- name: Run fat-jar server smoke test
run: .github/smoke-test-fatjar.sh fatjars 'llama-*-all-linux-aarch64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: fatjar-smoke-linux-aarch64-logs
path: |
server-out.log
server-err.log
if-no-files-found: warn
smoke-fatjar-windows-arm64:
name: Smoke test all-backends fat jar (Windows arm64)
needs: [package-fatjars, verify-model-cache]
runs-on: windows-11-arm
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-fatjar-smoke-windows-arm64
path: fatjars/
- name: Restore shared GGUF model cache (populated by download-models; no re-download)
# Restore-only: see the note on the Linux job above.
uses: actions/cache/restore@v6
with:
path: models/
key: gguf-models-${{ hashFiles('.github/models.csv') }}
enableCrossOsArchive: true
- name: Validate model files
run: .github\validate-models.bat
# temurin publishes a Windows/AArch64 JDK for this java-version; the build-windows-arm64
# job already resolves it on this same runner.
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Run fat-jar server smoke test
shell: pwsh
run: .github/smoke-test-fatjar.ps1 -JarDir fatjars -JarGlob 'llama-*-all-windows-aarch64-jar-with-dependencies.jar' -Model "models/$env:DRAFT_MODEL_NAME"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
with:
name: fatjar-smoke-windows-arm64-logs
path: |
server-out.log
server-err.log
if-no-files-found: warn
# ---------------------------------------------------------------------------
# macOS member of the cross-repo "no release asset is attached that CI has not run" convention
# (workspace/policies/fat-jar-release-assets.md; BitcoinAddressFinder and srcmorph run the shared
# .github/smoke-fatjar-cli.sh in the same job shape). It closes the gap that let a corrupt dylib
# ship: the three macOS Java test jobs each test the dylib THEIR OWN build job produced, so until
# now nothing ever loaded the one that goes into the published jar — see CLAUDE.md, "macOS arm64:
# three build jobs, one shipped dylib".
#
# macOS-specific in two ways. It targets the DEFAULT fat jar from `llama-jars`, because there is
# no `all-macos-*` fat jar to target: macOS has no GPU classifier (Metal ships in the default
# jar), so package-fatjars builds no macOS variant. And it asserts native loadability rather than
# a CLI exit code — this jar's Main-Class is a server that never returns, and `codesign --strict`
# is what actually detects a dylib assembled from two builds. Deliberately model-free (no GGUF,
# no cache restore, ~1 min): a full model-backed macOS server smoke would be strictly more, but
# this catches the failure class that shipped and is cheap enough to always run.
# ---------------------------------------------------------------------------
smoke-fatjar-macos:
name: Smoke test packaged natives (macOS)
needs: [package]
runs-on: macos-15
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: llama-jars
path: jars/
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
java-version: ${{ env.JAVA_VERSION }}
- name: Run packaged-native smoke test
run: .github/smoke-native-macos.sh jars 'llama-*-jar-with-dependencies.jar'
report:
name: Report
needs: [package]
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/setup-java@v6
with: { java-version: '${{ env.JAVA_VERSION }}', distribution: temurin }
- uses: actions/download-artifact@v8
with: { name: jacoco-report, path: target/site/jacoco/ }
continue-on-error: true
# Submits the dependency graph to GitHub. Informational: it says nothing about whether the
# artifacts are correct, but it sits in the `report` job, which the release path needs -- so
# without this flag a third-party action having a bad day can block a publish.
- uses: advanced-security/maven-dependency-submission-action@v6
continue-on-error: true
- name: Coveralls
uses: coverallsapp/github-action@v2
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
file: target/site/jacoco/jacoco.xml
format: jacoco
continue-on-error: true
- name: Codecov
uses: codecov/codecov-action@v7
with:
token: ${{ secrets.CODECOV_TOKEN }}
files: target/site/jacoco/jacoco.xml
continue-on-error: true
check-snapshot:
name: "Check: main branch / SNAPSHOT"
needs: [report]
runs-on: ubuntu-latest
if: >-
(github.event_name == 'push' && github.ref == 'refs/heads/main') ||
(github.event_name == 'workflow_dispatch' && !startsWith(github.ref, 'refs/tags/v'))
steps:
- name: Confirm snapshot ref
run: echo "Confirmed on snapshot ref ${{ github.ref }}"
check-tag:
name: "Check: v* tag"
needs: [report]
runs-on: ubuntu-latest
if: startsWith(github.ref, 'refs/tags/v')
steps:
- name: Confirm tag ref
run: echo "Confirmed on tag ${{ github.ref }}"
publish-snapshot:
name: Publish Snapshot to Central
needs: [check-snapshot, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos, smoke-agent-linux]
if: needs.check-snapshot.result == 'success' && inputs.publish_to_central
runs-on: ubuntu-latest
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
# Downloaded UNMERGED (one subdirectory per artifact name) on purpose: `merge-multiple: true`
# would silently overwrite — and can byte-level interleave — two artifacts that carry the same
# relative path, which is how the corrupt macOS dylib shipped. merge-native-artifacts.sh does
# the merge instead, and fails the job if any {OS}/{ARCH} path is claimed by more than one
# `*-libraries` artifact. A post-merge check cannot catch this (the collision leaves exactly
# one file on the path — a corrupt one), so the check has to happen before the merge.
- uses: actions/download-artifact@v8
with:
pattern: "*-libraries"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the default resource tree (collision-checked)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/"
# All three macOS arm64 build jobs emit the same path Mac/aarch64/libjllama.dylib, so their
# artifact names are kept outside the "*-libraries" glob above — merging them would drop
# three different dylibs onto one file. The variant that ships is selected explicitly:
# macos-15-metal is the only one built with both Metal and GGML_NATIVE=OFF (portable across
# CPU generations); macos-14-metal and macos-15-no-metal remain test-only.
- uses: actions/download-artifact@v8
with:
name: macos-15-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: linux-libraries-cuda
path: ${{ github.workspace }}/llama/src/main/resources_linux_cuda/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-aarch64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: android-libraries-opencl
path: ${{ github.workspace }}/llama/src/main/resources_android_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-cuda
path: ${{ github.workspace }}/llama/src/main/resources_windows_cuda/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_windows_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_linux_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_windows_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp16
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp16/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp32
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp32/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-sycl
path: ${{ github.workspace }}/llama/src/main/resources_windows_sycl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-aarch64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_linux_openvino/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
- name: Set up Maven Central Repository
uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: 'temurin'
server-id: central
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
gpg-passphrase: MAVEN_GPG_PASSPHRASE
- name: Guard - require a -SNAPSHOT version
shell: bash
run: |
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
echo "Resolved project version: $VERSION"
case "$VERSION" in
*-SNAPSHOT) echo "OK: -SNAPSHOT version, continuing snapshot deploy." ;;
*) echo "::error::Refusing to publish non-SNAPSHOT version '$VERSION' from the snapshot job. Snapshot publishing requires a -SNAPSHOT version; releases go through the v* tag path."; exit 1 ;;
esac
# One reactor deploy publishes all four Maven artifacts at the same version:
# net.ladenthin:llama-parent (the pom), :llama (the core jar + classifiers),
# :llama-langchain4j, and :llama-kotlin. The `release` profile (GPG + Central
# Publishing) is inherited from the parent, so every module — including the
# parent pom — is signed. The Android AARs are published by the separate
# Gradle step below (Maven cannot deploy <packaging>aar</packaging>).
# Informational only (nothing depends on it): logs the effective POM with the same
# profile set as the deploy below, so the resolved central-publishing configuration
# (waitUntil/waitMaxTime etc.) is visible for debugging.
- name: Show effective POM (debug)
run: mvn --batch-mode --no-transfer-progress -P release,cuda,vulkan-linux,vulkan-linux-aarch64,opencl-android,windows-msvc,cuda-windows,vulkan-windows,opencl-windows,rocm-linux,rocm-windows,sycl-fp16-linux,sycl-fp32-linux,sycl-windows,opencl-windows-aarch64,openvino-linux,openvino-windows help:effective-pom
- name: Publish snapshot (reactor - parent + llama + llama-langchain4j + llama-kotlin)
run: mvn --batch-mode --no-transfer-progress -P release,cuda,vulkan-linux,vulkan-linux-aarch64,opencl-android,windows-msvc,cuda-windows,vulkan-windows,opencl-windows,rocm-linux,rocm-windows,sycl-fp16-linux,sycl-fp32-linux,sycl-windows,opencl-windows-aarch64,openvino-linux,openvino-windows -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# Android AARs (llama-android, llama-android-opencl): assembled and published by
# the plain-Gradle build in llama-android/ — Maven cannot deploy
# <packaging>aar</packaging>. Natives are already on disk from the artifact
# downloads above; the core jar was just built by the reactor deploy.
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: "9.8.0"
- name: Publish Android AAR snapshots (llama-android + llama-android-opencl)
run: |
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp llama/src/main/resources/net/ladenthin/llama/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp llama/src/main/resources/net/ladenthin/llama/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
cp llama/src/main/resources_android_opencl/net/ladenthin/llama/Linux-Android/aarch64/libjllama.so llama-android/natives/opencl/arm64-v8a/
gradle -p llama-android publishAllPublicationsToCentralSnapshotsRepository
env:
CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
# Runs even when the deploy step failed: a Central publish-poll timeout reds the
# job *after* the bundle was uploaded (and typically published server-side), while
# the signed jars + .asc files already exist in target/ (signing happens at
# verify). Collecting on failure lets the github-snapshot job still attach them.
- name: Collect signed artifacts
if: ${{ !cancelled() }}
run: |
mkdir -p signed-snapshot-assets
cp llama/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar signed-snapshot-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true
cp llama-android/build/aar/*.aar signed-snapshot-assets/ 2>/dev/null || true
- uses: actions/upload-artifact@v7
if: ${{ !cancelled() }}
with:
name: signed-snapshot-assets
path: signed-snapshot-assets/
github-snapshot:
name: Update Snapshot Pre-release on GitHub
needs: [publish-snapshot, package-fatjars, test-java-llama-atmosphere-agent]
# Also runs when publish-snapshot FAILED (not when skipped/cancelled): a Central
# publish-poll timeout reds that job after the artifacts were already uploaded —
# the GitHub pre-release assets must not be lost in that case.
if: ${{ !cancelled() && (needs.publish-snapshot.result == 'success' || needs.publish-snapshot.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }}
runs-on: ubuntu-latest
# maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is
# scoped to this environment) for signing the fat jars below. This environment has
# no approval gate (the standalone verify-signing-key jobs use it on every run), so
# declaring it here does not block the release.
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: signed-snapshot-assets
path: snapshot-assets/
# All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub
# download assets only, deliberately NOT deployed to Maven Central. Downloaded
# into the same directory so the one upload glob below picks everything up
# (fat-jar names are disjoint from the signed thin-jar names).
- uses: actions/download-artifact@v8
with:
name: llama-fatjars
path: snapshot-assets/
# The agent jar (+ sha256) — built without the core, run next to one of the fat jars above.
# Same directory, so sign-fatjars.sh signs it and the upload glob attaches it.
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: snapshot-assets/
# GPG-sign the fat jars so each carries a detached .asc signature alongside its
# .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at
# deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity)
# are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because
# only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md.
- name: GPG-sign the fat jars (.asc)
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
run: bash .github/sign-fatjars.sh snapshot-assets
- name: Report unsigned assets (does not block the upload)
# Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even
# when their publish job failed, because a Central publish-poll timeout must not cost the
# GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at
# all. Refusing to attach on a signing failure would defeat exactly that. So annotate here,
# upload regardless, and fail the job afterwards. Assets always land; an unsigned release is
# still loudly red rather than quietly wrong.
# See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red".
id: signatures
run: |
set -uo pipefail
dir="snapshot-assets"
jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort)
if [ -z "$jars" ]; then
echo "::error::no jars in $dir -- the collection step produced nothing"
echo "missing=-1" >> "$GITHUB_OUTPUT"
exit 0
fi
missing=0
for jar in $jars; do
if [ ! -e "$jar.asc" ]; then
echo "::error::unsigned: $(basename "$jar") has no detached .asc"
missing=$((missing + 1))
fi
done
echo "missing=$missing" >> "$GITHUB_OUTPUT"
[ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed"
exit 0
- name: Update snapshot pre-release
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
gh release view snapshot --repo ${{ github.repository }} 2>/dev/null \
|| gh release create snapshot \
--repo ${{ github.repository }} \
--prerelease \
--title "Snapshot (latest)" \
--notes "Latest snapshot build from the main branch."
gh release upload snapshot snapshot-assets/* \
--repo ${{ github.repository }} \
--clobber
- name: Fail if anything was attached unsigned
# After the upload on purpose: the assets must exist even when the signature does not.
if: ${{ always() && steps.signatures.outputs.missing != '0' }}
run: |
echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)"
exit 1
publish-release:
name: Publish Release to Central
if: needs.check-tag.result == 'success' && inputs.publish_to_central
needs: [check-tag, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos, smoke-agent-linux]
runs-on: ubuntu-latest
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
# Downloaded UNMERGED (one subdirectory per artifact name) on purpose: `merge-multiple: true`
# would silently overwrite — and can byte-level interleave — two artifacts that carry the same
# relative path, which is how the corrupt macOS dylib shipped. merge-native-artifacts.sh does
# the merge instead, and fails the job if any {OS}/{ARCH} path is claimed by more than one
# `*-libraries` artifact. A post-merge check cannot catch this (the collision leaves exactly
# one file on the path — a corrupt one), so the check has to happen before the merge.
- uses: actions/download-artifact@v8
with:
pattern: "*-libraries"
path: ${{ github.workspace }}/native-artifacts/
- name: Merge native libraries into the default resource tree (collision-checked)
run: |
bash .github/merge-native-artifacts.sh \
"${{ github.workspace }}/native-artifacts" \
"${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/"
# All three macOS arm64 build jobs emit the same path Mac/aarch64/libjllama.dylib, so their
# artifact names are kept outside the "*-libraries" glob above — merging them would drop
# three different dylibs onto one file. The variant that ships is selected explicitly:
# macos-15-metal is the only one built with both Metal and GGML_NATIVE=OFF (portable across
# CPU generations); macos-14-metal and macos-15-no-metal remain test-only.
- uses: actions/download-artifact@v8
with:
name: macos-15-metal
path: ${{ github.workspace }}/llama/src/main/resources/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: linux-libraries-cuda
path: ${{ github.workspace }}/llama/src/main/resources_linux_cuda/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-aarch64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_linux_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: android-libraries-opencl
path: ${{ github.workspace }}/llama/src/main/resources_android_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86-msvc
path: ${{ github.workspace }}/llama/src/main/resources_windows_msvc/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-cuda
path: ${{ github.workspace }}/llama/src/main/resources_windows_cuda/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-vulkan
path: ${{ github.workspace }}/llama/src/main/resources_windows_vulkan/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_linux_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-rocm
path: ${{ github.workspace }}/llama/src/main/resources_windows_rocm/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp16
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp16/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-sycl-fp32
path: ${{ github.workspace }}/llama/src/main/resources_linux_sycl_fp32/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-sycl
path: ${{ github.workspace }}/llama/src/main/resources_windows_sycl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-aarch64-opencl
path: ${{ github.workspace }}/llama/src/main/resources_windows_opencl/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Linux-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_linux_openvino/net/ladenthin/llama/
- uses: actions/download-artifact@v8
with:
name: Windows-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
- name: Set up Maven Central Repository
uses: actions/setup-java@v6
with:
java-version: ${{ env.JAVA_VERSION }}
distribution: 'temurin'
server-id: central
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
gpg-passphrase: MAVEN_GPG_PASSPHRASE
# One reactor deploy publishes all three artifacts at the same version:
# net.ladenthin:llama-parent (the pom), :llama (the core jar + classifiers), and
# :llama-langchain4j. The `release` profile (GPG + Central Publishing) is inherited
# from the parent, so every module — including the parent pom — is signed.
# Informational only (nothing depends on it): logs the effective POM with the same
# profile set as the deploy below, so the resolved central-publishing configuration
# (waitUntil/waitMaxTime etc.) is visible for debugging.
- name: Show effective POM (debug)
run: mvn --batch-mode --no-transfer-progress -P release,cuda,vulkan-linux,vulkan-linux-aarch64,opencl-android,windows-msvc,cuda-windows,vulkan-windows,opencl-windows,rocm-linux,rocm-windows,sycl-fp16-linux,sycl-fp32-linux,sycl-windows,opencl-windows-aarch64,openvino-linux,openvino-windows help:effective-pom
- name: Publish release (reactor - parent + llama + llama-langchain4j + llama-kotlin)
run: mvn --batch-mode --no-transfer-progress -P release,cuda,vulkan-linux,vulkan-linux-aarch64,opencl-android,windows-msvc,cuda-windows,vulkan-windows,opencl-windows,rocm-linux,rocm-windows,sycl-fp16-linux,sycl-fp32-linux,sycl-windows,opencl-windows-aarch64,openvino-linux,openvino-windows -Dmaven.test.skip=true deploy
env:
MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
# Android AARs (llama-android, llama-android-opencl): Maven cannot deploy
# <packaging>aar</packaging>, so the plain-Gradle build signs + publishes them
# into a local Maven-layout staging repo, which is zipped into a Central
# Portal bundle and uploaded via the Publisher API (publishingType=AUTOMATIC:
# the deployment publishes as soon as portal validation passes). Natives are
# already on disk from the artifact downloads above; the core jar was just
# built by the reactor deploy.
- uses: gradle/actions/setup-gradle@v6
with:
gradle-version: "9.8.0"
- name: Publish Android AAR release bundle to Central Portal
shell: bash
run: |
set -euo pipefail
mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a
cp llama/src/main/resources/net/ladenthin/llama/Linux-Android/aarch64/libjllama.so llama-android/natives/cpu/arm64-v8a/
cp llama/src/main/resources/net/ladenthin/llama/Linux-Android/x86_64/libjllama.so llama-android/natives/cpu/x86_64/
cp llama/src/main/resources_android_opencl/net/ladenthin/llama/Linux-Android/aarch64/libjllama.so llama-android/natives/opencl/arm64-v8a/
gradle -p llama-android publishAllPublicationsToStagingRepository
VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)
( cd llama-android/build/staging-repo && zip -r -q ../central-bundle.zip net -x "*maven-metadata*" )
TOKEN=$(printf "%s:%s" "$CENTRAL_USERNAME" "$CENTRAL_PASSWORD" | base64 -w0)
curl --fail-with-body -X POST \
-H "Authorization: Bearer $TOKEN" \
-F "bundle=@llama-android/build/central-bundle.zip" \
"https://central.sonatype.com/api/v1/publisher/upload?publishingType=AUTOMATIC&name=llama-android-$VERSION"
env:
CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }}
CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }}
MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }}
# Runs even when the deploy step failed: a Central publish-poll timeout reds the
# job *after* the bundle was uploaded (and typically published server-side), while
# the signed jars + .asc files already exist in target/ (signing happens at
# verify). Collecting on failure lets the github-release-signed job still attach them.
- name: Collect signed artifacts
if: ${{ !cancelled() }}
run: |
mkdir -p signed-release-assets
cp llama/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama-langchain4j/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar signed-release-assets/ 2>/dev/null || true
cp llama-kotlin/target/*.jar.asc signed-release-assets/ 2>/dev/null || true
cp llama-android/build/aar/*.aar signed-release-assets/ 2>/dev/null || true
- uses: actions/upload-artifact@v7
if: ${{ !cancelled() }}
with:
name: signed-release-assets
path: signed-release-assets/
github-release-signed:
name: Attach Signed Binaries to GitHub Release
needs: [publish-release, package-fatjars, test-java-llama-atmosphere-agent]
# Also runs when publish-release FAILED (not when skipped/cancelled): a Central
# publish-poll timeout reds that job after the artifacts were already uploaded —
# the GitHub release assets must not be lost in that case.
if: ${{ !cancelled() && (needs.publish-release.result == 'success' || needs.publish-release.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }}
runs-on: ubuntu-latest
# maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is
# scoped to this environment) for signing the fat jars below. This environment has
# no approval gate (the standalone verify-signing-key jobs use it on every run), so
# declaring it here does not block the release.
environment: maven-central
permissions:
contents: write
steps:
- uses: actions/checkout@v7
- uses: actions/download-artifact@v8
with:
name: signed-release-assets
path: release-assets/
# All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub
# download assets only, deliberately NOT deployed to Maven Central. Downloaded
# into the same directory so the one upload glob below picks everything up
# (fat-jar names are disjoint from the signed thin-jar names).
- uses: actions/download-artifact@v8
with:
name: llama-fatjars
path: release-assets/
# The agent jar (+ sha256) — built without the core, run next to one of the fat jars above.
# Same directory, so sign-fatjars.sh signs it and the upload glob attaches it.
- uses: actions/download-artifact@v8
with:
name: llama-atmosphere-agent-jar
path: release-assets/
# GPG-sign the fat jars so each carries a detached .asc signature alongside its
# .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at
# deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity)
# are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because
# only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md.
- name: GPG-sign the fat jars (.asc)
env:
GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }}
GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
run: bash .github/sign-fatjars.sh release-assets
- name: Report unsigned assets (does not block the upload)
# Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even
# when their publish job failed, because a Central publish-poll timeout must not cost the
# GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at
# all. Refusing to attach on a signing failure would defeat exactly that. So annotate here,
# upload regardless, and fail the job afterwards. Assets always land; an unsigned release is
# still loudly red rather than quietly wrong.
# See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red".
id: signatures
run: |
set -uo pipefail
dir="release-assets"
jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort)
if [ -z "$jars" ]; then
echo "::error::no jars in $dir -- the collection step produced nothing"
echo "missing=-1" >> "$GITHUB_OUTPUT"
exit 0
fi
missing=0
for jar in $jars; do
if [ ! -e "$jar.asc" ]; then
echo "::error::unsigned: $(basename "$jar") has no detached .asc"
missing=$((missing + 1))
fi
done
echo "missing=$missing" >> "$GITHUB_OUTPUT"
[ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed"
exit 0
- name: Upload release assets
uses: softprops/action-gh-release@v3
with:
files: release-assets/*
- name: Fail if anything was attached unsigned
# After the upload on purpose: the assets must exist even when the signature does not.
if: ${{ always() && steps.signatures.outputs.missing != '0' }}
run: |
echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)"
exit 1