llama.cpp b11320, shared plugin versions in the parent, agent on Maven Central #1084
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com> | |
| # | |
| # SPDX-License-Identifier: MIT OR Apache-2.0 | |
| name: Publish | |
| on: | |
| push: | |
| branches: [ main ] | |
| tags: ['v*'] | |
| pull_request: | |
| workflow_dispatch: | |
| inputs: | |
| publish_to_central: | |
| description: "Deploy to Maven Central (snapshot if -SNAPSHOT, release if a vX.Y.Z tag)" | |
| type: boolean | |
| default: false | |
| use_cache: | |
| description: "Use the shared sccache/Depot compiler cache (faster incremental builds)" | |
| type: boolean | |
| default: true | |
| env: | |
| # The JDK every job builds with is .java-version (setup-java's java-version-file), so the | |
| # reusable workflows read the same one without an input. | |
| # The Gradle builds (Android AAR, the consumer fixture, the llmservice app, the signing self-test). | |
| # AGP 9.4.x needs Gradle >= 9.6.0 (android-llmservice/CLAUDE.md). Not seen by Dependabot: bump by hand, | |
| # together with the literal in verify-signing-key-gradle (a job kept identical in all four sibling | |
| # repositories, so it names its version itself). | |
| GRADLE_VERSION: '9.8.0' | |
| # The sccache/Depot compiler cache (CLAUDE.md, "CI build cache"). Inert without the token, which | |
| # stays per build job (SCCACHE_WEBDAV_TOKEN): at workflow level it would reach every job, | |
| # including those running third-party actions. | |
| USE_CACHE: ${{ github.event_name != 'workflow_dispatch' || inputs.use_cache }} | |
| SCCACHE_WEBDAV_ENDPOINT: https://cache.depot.dev | |
| # The CI model set is .github/models.csv (URLs, and the model cache key is its hash). The Java | |
| # tests default to exactly that set (TestConstants; TestConstantsTest asserts both directions), | |
| # so the test jobs pass no model properties. The names below are only for the consumers outside | |
| # the llama module's tests -- the smoke scripts, the Android emulator jobs and the langchain4j / | |
| # agent integration jobs; check-natives.py fails when one is not a filename of models.csv. | |
| RERANKING_MODEL_NAME: "jina-reranker-v1-tiny-en-Q4_0.gguf" | |
| DRAFT_MODEL_NAME: "AMD-Llama-135m-code.Q2_K.gguf" | |
| REASONING_MODEL_NAME: "Qwen3-0.6B-Q4_K_M.gguf" | |
| TOOL_MODEL_NAME: "Qwen2.5-1.5B-Instruct-Q4_K_M.gguf" | |
| NOMIC_EMBED_MODEL_NAME: "nomic-embed-text-v1.5.f16.gguf" | |
| # Supersede an in-flight run when a PR branch is pushed again. | |
| # | |
| # Without this every push starts a full parallel pipeline and the older ones keep | |
| # draining -- four were live at once during one session, which makes "what is CI | |
| # saying right now" genuinely ambiguous and wastes a lot of runner time on results | |
| # nobody will read. | |
| # | |
| # cancel-in-progress is deliberately scoped to pull_request ONLY. A push to main or | |
| # to a v* tag is a release path: cancelling one midway could leave a partially | |
| # published set of artifacts. | |
| # | |
| # cancel-in-progress: false is NOT sufficient on its own to protect a release run. | |
| # GitHub cancels a *pending* run whenever a newer run joins the same group behind an | |
| # in-progress one -- that rule is independent of cancel-in-progress. So with a plain | |
| # `workflow-ref` group, a queued `publish_to_central` dispatch on main could be | |
| # silently dropped by a later push to main, both sharing `Publish-refs/heads/main`. | |
| # Giving every non-PR run its own group (via the unique run_id) means such a run is | |
| # never queued behind a sibling and therefore can never be cancelled, while PR runs | |
| # still share a group per ref and supersede each other as intended. | |
| # | |
| # One-time effect when this expression changes: GitHub reads `concurrency` from the | |
| # workflow file at each run's own ref, so a run started before the change sits in the | |
| # old group and a run started after it sits in the new one. They are different groups, | |
| # so the new push does NOT supersede the in-flight old run -- exactly once, on the | |
| # commit that lands this. It self-heals from the next push on. Expect the same overlap | |
| # when porting this to a sibling repo; it is not a sign the expression is wrong. | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name == 'pull_request' && 'pr' || github.run_id }} | |
| cancel-in-progress: ${{ github.event_name == 'pull_request' }} | |
| permissions: | |
| contents: read | |
| jobs: | |
| # --------------------------------------------------------------------------- | |
| # Start gate — single cancellable abort window before the pipeline starts. | |
| # The wait duration lives in the `startgate` GitHub Environment (Settings → | |
| # Environments → startgate → Wait timer). | |
| # --------------------------------------------------------------------------- | |
| startgate: | |
| name: Start gate (abort window) | |
| runs-on: ubuntu-latest | |
| environment: startgate | |
| steps: | |
| - run: echo "Start gate elapsed — proceeding with pipeline." | |
| # --------------------------------------------------------------------------- | |
| # Files kept byte-identical with the sibling repositories (java-llama.cpp, | |
| # BitcoinAddressFinder, srcmorph, streambuffer), listed in .github/shared-files.sha256. | |
| # A copy changed here alone fails; one the siblings list with another hash warns. | |
| # Also runs the tests of the shared build checks and the release-gate check. This | |
| # job is itself identical in all four repositories. | |
| # --------------------------------------------------------------------------- | |
| shared-files: | |
| name: Shared files + build checks | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| persist-credentials: false | |
| - name: Shared files match .github/shared-files.sha256 (siblings compared, warnings only) | |
| run: python3 .github/check-shared-files.py | |
| - name: Build-check tests | |
| run: python3 -m unittest discover -s .github/buildcheck/tests -t .github | |
| - name: Release gate (every job gates both publish jobs, or release-gate-exemptions.txt says why) | |
| run: python3 .github/check-release-gate.py | |
| - name: Run scripts parse (bash -n over every bash run script of the workflows and actions) | |
| run: python3 .github/check-run-scripts.py | |
| - name: Maven versions (warns where a dependency or plugin differs from a sibling repository) | |
| run: python3 .github/check-versions.py | |
| # --------------------------------------------------------------------------- | |
| # GPG signing-key preflight (standalone, no `needs:` — runs in parallel at the | |
| # very start on every trigger). Reproduces what maven-gpg-plugin does at deploy | |
| # time so a bad/expired key or wrong passphrase is caught in ~20s instead of | |
| # failing the publish stage. Declares `environment: maven-central` so it reads | |
| # the SAME GPG_PRIVATE_KEY / GPG_PASSPHRASE secret the publish jobs use. | |
| # | |
| # It is EXPECTED to go RED on refs where the secret is not delivered — fork PRs | |
| # and other contributors' branches (secrets are withheld there). That red is | |
| # the intended signal: "this ref cannot sign a release", not a regression. | |
| # | |
| # SECURITY: this job NEVER prints secret material. It imports the key into an | |
| # ephemeral keyring, prints only PUBLIC key metadata (key id, fingerprint, | |
| # owner UID, algorithm, created/expiry — all of which live on public | |
| # keyservers), and validates the passphrase by producing + verifying a | |
| # throwaway signature. The passphrase is passed on fd 3 (never argv, never a | |
| # log line), `set -x` is deliberately never enabled, and the passphrase is | |
| # additionally `::add-mask::`ed. | |
| # --------------------------------------------------------------------------- | |
| verify-signing-key: | |
| name: Verify GPG signing key (no secrets printed) | |
| runs-on: ubuntu-latest | |
| environment: maven-central | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| persist-credentials: false | |
| - name: Import key + run sign/verify self-test (prints only PUBLIC metadata) | |
| env: | |
| GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| # Shared, byte-identical in all four repos; the security notes are in the script. | |
| run: bash .github/verify-signing-key.sh | |
| # --------------------------------------------------------------------------- | |
| # GPG signing-key preflight — GRADLE / BouncyCastle path. | |
| # Companion to the `verify-signing-key` (gpg) job above: that one mirrors | |
| # maven-gpg-plugin (how the Maven artifacts are signed); this one drives | |
| # Gradle's `signing` plugin + `useInMemoryPgpKeys` (BouncyCastle) — the path any | |
| # Gradle-based publish (e.g. an Android AAR) uses to sign. BouncyCastle is a | |
| # STRICTER parser of the armored key than gpg, so it catches key/format problems | |
| # gpg tolerates (e.g. the primary-vs-signing-subkey null-PGPPrivateKey issue). | |
| # It signs a throwaway project (.github/signing-selftest/) — no repo build is | |
| # involved — so this job is IDENTICAL across the sibling repos and validates the | |
| # release key via the Gradle path even in repos that do not publish via Gradle | |
| # yet ("prepared for Gradle"). Standalone (no `needs:`), parallel at pipeline | |
| # start, `environment: maven-central` so it reads the same secret the publish | |
| # uses. Red-by-design where the secret is not delivered (see the gpg job's note). | |
| # | |
| # SECURITY: prints no secret material. Key/passphrase reach Gradle only via env | |
| # (read by System.getenv at runtime); `set -x` is never enabled; the passphrase | |
| # is `::add-mask::`ed; Gradle runs with `--stacktrace` only. Only the produced | |
| # `.asc` (exit code) is asserted. Uses Gradle 9.6.1. | |
| # --------------------------------------------------------------------------- | |
| verify-signing-key-gradle: | |
| name: Verify GPG signing key — Gradle/BouncyCastle path (no secrets printed) | |
| runs-on: ubuntu-latest | |
| environment: maven-central | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| gradle-version: "9.8.0" | |
| - name: Sign a throwaway artifact via useInMemoryPgpKeys (BouncyCastle) | |
| shell: bash | |
| env: | |
| MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }} | |
| run: | | |
| set -euo pipefail # NOTE: deliberately NO `set -x` — it would echo the passphrase. | |
| if [ -z "${MAVEN_GPG_PRIVATE_KEY:-}" ]; then | |
| echo "::error::MAVEN_GPG_PRIVATE_KEY is empty for this run. The maven-central environment did not deliver the secret to this ref (fork PR / other branch). Nothing to verify." | |
| exit 1 | |
| fi | |
| if [ -n "${MAVEN_GPG_PASSPHRASE:-}" ]; then echo "::add-mask::${MAVEN_GPG_PASSPHRASE}"; fi | |
| PROJ=".github/signing-selftest" | |
| echo "== Sign a throwaway artifact through Gradle's useInMemoryPgpKeys (BouncyCastle) ==" | |
| gradle --no-daemon -p "$PROJ" signMakeArtifact --stacktrace | |
| ASC="$PROJ/build/signing-selftest.zip.asc" | |
| if [ -f "$ASC" ]; then | |
| echo " Detached signature produced: $(wc -c < "$ASC") armored bytes" | |
| echo "RESULT: OK — Gradle's useInMemoryPgpKeys accepted the armored key + passphrase and produced a signature." | |
| else | |
| echo "::error::Gradle signing produced no .asc — useInMemoryPgpKeys could not build a usable signatory from MAVEN_GPG_PRIVATE_KEY / MAVEN_GPG_PASSPHRASE." | |
| exit 1 | |
| fi | |
| # --------------------------------------------------------------------------- | |
| # Download + cache the GGUF test models ONCE, upfront, for the whole pipeline. | |
| # Every Java test job (`test-java-*`) and the langchain4j integration job `needs:` this | |
| # job and then only RESTORES the shared cache (key gguf-models-<manifest hash>) — so the ~5 GB model | |
| # set is fetched from HuggingFace at most once per cache lifetime instead of racing to | |
| # download in each job. GGUF is platform-independent, so this single ubuntu job's cache | |
| # is reused by the macOS and Windows jobs too. On a warm cache this job is a no-op | |
| # restore; on a cold cache it downloads all models, validates them, and saves the cache | |
| # at job end (immutable key, so downstream jobs restore-hit). This is the single source | |
| # of the download logic — do not re-add per-job downloads. | |
| # --------------------------------------------------------------------------- | |
| download-models: | |
| name: Download + cache GGUF models (once) | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Cache GGUF models (GitHub Actions cache; avoids re-downloading from HuggingFace) | |
| # This job is the ONLY writer of the model cache (every consumer job uses the | |
| # restore-only action). enableCrossOsArchive makes the one ubuntu-built entry | |
| # restorable on macOS AND Windows — without it, cache entries are versioned | |
| # per-OS, and the unreachable Windows-side entry was once found re-saved EMPTY | |
| # (343 B) after an eviction, silently starving the Windows jobs of models. | |
| uses: actions/cache@v6 | |
| with: | |
| path: models/ | |
| # GGUF is platform-independent, so ubuntu + macOS + Windows share one entry. | |
| # The key is derived from the model manifest, so EDITING models.csv | |
| # automatically creates a fresh complete entry — no manual cache deletion. | |
| key: gguf-models-${{ hashFiles('.github/models.csv') }} | |
| enableCrossOsArchive: true | |
| - name: Download missing models (manifest-driven, .github/models.csv) | |
| # The manifest is the single source of truth for the model set; this loop is | |
| # the ONLY place in CI that downloads models. Files already restored from the | |
| # cache are skipped, so a warm cache downloads nothing. | |
| run: | | |
| while IFS=, read -r name url; do | |
| case "$name" in ''|\#*) continue ;; esac | |
| test -f "models/$name" || curl -L --proto =https --proto-redir =https --fail --retry 5 --retry-all-errors "$url" --create-dirs -o "models/$name" | |
| done < .github/models.csv | |
| - name: List files in models directory | |
| run: ls -l models/ | |
| - name: Validate model files | |
| run: bash .github/validate-models.sh | |
| # Prove the freshly written cache entry is restorable AND complete on every OS the | |
| # pipeline uses BEFORE any model-consuming job starts: the same restore-only action | |
| # the consumers use (fail-on-cache-miss makes an unrestorable entry fail here, not | |
| # deep inside a test job), followed by the full validate gate. Every model-backed | |
| # job `needs:` this instead of download-models directly. | |
| verify-model-cache: | |
| name: "Verify model cache (${{ matrix.os }})" | |
| needs: download-models | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| os: [ubuntu-latest, macos-15, windows-2025-vs2026] | |
| runs-on: ${{ matrix.os }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: ./.github/actions/restore-models | |
| with: | |
| fail-on-cache-miss: true | |
| # --------------------------------------------------------------------------- | |
| # Cross-compile jobs (Docker / dockcross) — produce release artifacts, no testing | |
| # --------------------------------------------------------------------------- | |
| code-style: | |
| name: Code style (spotless) + package graph | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| # .github/buildcheck/natives.py; its unit tests run in the shared-files job. | |
| - name: Natives list check (everything that names a natives jar agrees with natives.csv) | |
| run: python3 .github/check-natives.py | |
| - name: Spotless check (fail fast on format violations) | |
| run: mvn -B --no-transfer-progress -f llama/pom.xml spotless:check | |
| - name: SpotBugs check (fail fast on static-analysis findings) | |
| run: mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile spotbugs:check | |
| - name: Print internal package dependency graph (jdeps, informational) | |
| continue-on-error: true | |
| run: | | |
| mvn -B --no-transfer-progress -f llama/pom.xml -DskipTests -Denforcer.skip=true compile | |
| echo "=== internal package dependency graph (jdeps, bytecode) ===" | |
| jdeps -verbose:package llama/target/classes | grep 'net.ladenthin.llama' || true | |
| # --------------------------------------------------------------------------- | |
| # Sibling module `llama-langchain4j` (LangChain4j adapters). Pure Java, no native | |
| # code and no per-classifier matrix: it compiles against the core's stable Java API | |
| # (identical across every classifier) and the backend is a runtime choice for the | |
| # consumer. This job installs the parent + core into the local repo, then builds + tests | |
| # the module (Java 17; langchain4j 1.x baseline). It runs its mapping unit tests; the | |
| # model-backed integration test self-skips without a GGUF. `verify` also builds the | |
| # javadoc/sources jars so a release-time javadoc break is caught here in PR CI. Version | |
| # lockstep is now guaranteed by construction (both modules inherit the parent's version), | |
| # so the old lockstep guard is gone. | |
| # --------------------------------------------------------------------------- | |
| test-java-llama-langchain4j: | |
| name: Build and Test llama-langchain4j | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - name: Install parent + core net.ladenthin:llama into the local repo (Java only) | |
| uses: ./.github/actions/build-core | |
| - name: Build and test llama-langchain4j | |
| run: mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml verify | |
| # --------------------------------------------------------------------------- | |
| # Model-free unit tests for the Kotlin coroutines facade (llama-kotlin). | |
| # Pure Kotlin/JVM reactor module; its 6 tests fake the Iterable+AutoCloseable | |
| # seam, so no native library and no model are needed. | |
| # --------------------------------------------------------------------------- | |
| test-java-llama-kotlin: | |
| name: Build and Test llama-kotlin | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - name: Install parent + core net.ladenthin:llama into the local repo (Java only) | |
| uses: ./.github/actions/build-core | |
| - name: Build and test llama-kotlin | |
| run: mvn -B --no-transfer-progress -f llama-kotlin/pom.xml verify | |
| # --------------------------------------------------------------------------- | |
| # Model-backed integration for the langchain4j adapters. Reuses the SAME shared GGUF | |
| # cache (populated once by download-models) and the SAME Linux-x86_64 native artifact the | |
| # core Java jobs already use — no extra model download and no duplicated download logic | |
| # (restore-only cache, no curl steps). It exercises the chat / embedding / scoring | |
| # adapters against the already-cached chat (Qwen3-0.6B), nomic-embedding and jina-reranker | |
| # models. The model-backed tests self-skip when a model is absent, so a cold cache degrades | |
| # to a skip, never a failure. | |
| # --------------------------------------------------------------------------- | |
| test-java-llama-langchain4j-integration: | |
| name: Integration Test llama-langchain4j (model-backed) | |
| needs: [crosscompile-linux-x86_64, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download Linux x86_64 native library (reused, not rebuilt) | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-cpu-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| - uses: ./.github/actions/restore-models | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install parent + core net.ladenthin:llama (classes; the test JVM loads the downloaded native library via lib.path) | |
| uses: ./.github/actions/build-core | |
| - name: Run llama-langchain4j model-backed integration tests (reused cached models) | |
| run: > | |
| mvn -B --no-transfer-progress -f llama-langchain4j/pom.xml test | |
| -Dnet.ladenthin.llama.model.path=models/${REASONING_MODEL_NAME} | |
| -Dnet.ladenthin.llama.langchain4j.embedding.model=models/${NOMIC_EMBED_MODEL_NAME} | |
| -Dnet.ladenthin.llama.langchain4j.rerank.model=models/${RERANKING_MODEL_NAME} | |
| -Dnet.ladenthin.llama.langchain4j.tool.model=models/${TOOL_MODEL_NAME} | |
| -Dnet.ladenthin.llama.lib.path=${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/Linux/x86_64/cpu | |
| # Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1). | |
| - name: Print crash logs (on failure) | |
| if: failure() | |
| shell: bash | |
| run: bash .github/print-crash-logs.sh llama-langchain4j | |
| - if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: error-log-langchain4j-integration | |
| path: | | |
| ${{ github.workspace }}/llama-langchain4j/hs_err_pid*.log | |
| ${{ github.workspace }}/core.* | |
| ${{ github.workspace }}/llama-langchain4j/*.hprof | |
| ${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dump | |
| ${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.dumpstream | |
| ${{ github.workspace }}/llama-langchain4j/target/surefire-reports/*.txt | |
| ${{ github.workspace }}/llama-langchain4j/target/surefire-reports/TEST-*.xml | |
| if-no-files-found: warn | |
| # --------------------------------------------------------------------------- | |
| # Build the llama.cpp WebUI ONCE, from the same pinned tag CMakeLists.txt fetches, | |
| # and share it to every native build as the generated, platform-independent | |
| # ui.cpp/ui.h ("webui-generated" artifact). The native builds embed it into | |
| # libjllama (CMake's "WebUI assets" block); when this job's artifact is absent the | |
| # build falls back to the empty-asset stub. npm runs only here, in one controlled | |
| # job — never in the dockcross cross-compilers (which have no node) or per-platform. | |
| # --------------------------------------------------------------------------- | |
| # --------------------------------------------------------------------------- | |
| # llama-atmosphere-agent: the standalone (non-reactor) local | |
| # coding-agent project that wires Atmosphere's built-in OpenAI-compatible agent runtime to | |
| # this project's OpenAiCompatServer. Three jobs, all publish gates: | |
| # - model-free: unit tests + the wire-contract tests, which drive the REAL | |
| # OpenAiCompatServer over a loopback socket with a scripted backend (no native lib, | |
| # no GGUF) and pin the streamed tool_calls / role=tool / multi-round shape — seconds, | |
| # on every PR. It also builds the GitHub Release asset (the agent jar WITHOUT the core). | |
| # - model-backed: the same loop against the cached Qwen2.5-1.5B tool model through the | |
| # downloaded Linux native library (chat, streaming, tool call + result, read/write/read | |
| # loop). A gate since its assertions stopped pinning wording (content checks are limited | |
| # to facts no instruct model gets wrong and to tool results) and it ran green throughout. | |
| # - smoke-agent-linux (further down, after package-fatjars): the release asset itself, | |
| # started next to the real core fat jar. | |
| # The project is built with -Dllama.version=<reactor version> against the core that was just | |
| # installed to the local repo, so it always tests the code of this checkout. Only the classes and | |
| # the llama-platform pom are installed; -Dllama.natives=none keeps Maven from resolving the natives | |
| # jars that pom names (they exist only after the package job). The agent itself is published to | |
| # Maven Central by the two publish jobs, after the reactor deploy. | |
| # --------------------------------------------------------------------------- | |
| test-java-llama-atmosphere-agent: | |
| name: Build and Test llama-atmosphere-agent (model-free) | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - name: Install parent + core net.ladenthin:llama + the llama-platform pom into the local repo (Java only) | |
| uses: ./.github/actions/build-core | |
| with: | |
| modules: llama,llama-platform | |
| - name: Resolve the reactor version | |
| run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV" | |
| - name: Spotless check | |
| run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none spotless:check | |
| - name: Build and test (unit + model-free wire contract against the real OpenAiCompatServer) | |
| run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none verify | |
| # The GitHub Release asset llama-atmosphere-agent-<core version>-jar-with-dependencies.jar: | |
| # the agent plus Atmosphere/JLine, WITHOUT the core (src/assembly/agent-jar.xml), so it is a | |
| # few MB and the natives are not in the release twice. Never deployed to Maven Central (the | |
| # thin jar is; see the publish jobs). | |
| # smoke-agent-linux launches it next to the real core fat jar; the attach jobs sign it. | |
| - name: Build the agent release jar (without the core) | |
| run: > | |
| mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none | |
| -P assembly -DskipTests package | |
| - name: Collect the agent release jar + sha256 | |
| run: | | |
| mkdir -p agent-jar | |
| cp "llama-atmosphere-agent/target/llama-atmosphere-agent-${VERSION}-jar-with-dependencies.jar" agent-jar/ | |
| (cd agent-jar && for f in *.jar; do sha256sum "$f" > "$f.sha256"; done) | |
| ls -la agent-jar | |
| - name: Upload the agent release jar | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-atmosphere-agent-jar | |
| path: agent-jar/ | |
| compression-level: 0 # jars are already deflated | |
| retention-days: 7 | |
| if-no-files-found: error | |
| test-java-llama-atmosphere-agent-integration: | |
| name: Integration Test llama-atmosphere-agent (model-backed) | |
| needs: [crosscompile-linux-x86_64, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download Linux x86_64 native library (reused, not rebuilt) | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-cpu-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| - uses: ./.github/actions/restore-models | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install parent + core net.ladenthin:llama + the llama-platform pom (classes; the test JVM loads the downloaded native library via lib.path) | |
| uses: ./.github/actions/build-core | |
| with: | |
| modules: llama,llama-platform | |
| - name: Resolve the reactor version | |
| run: echo "VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1)" >> "$GITHUB_ENV" | |
| - name: Run the Atmosphere tool-loop integration test (cached Qwen2.5-1.5B tool model, CPU) | |
| run: > | |
| mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" -Dllama.natives=none test | |
| -Dtest=AtmosphereToolLoopIntegrationTest -Dsurefire.failIfNoSpecifiedTests=false | |
| -Dnet.ladenthin.llama.tool.model=models/${TOOL_MODEL_NAME} | |
| -Dnet.ladenthin.llama.test.ngl=0 | |
| -Dnet.ladenthin.llama.lib.path=${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/Linux/x86_64/cpu | |
| # Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1). | |
| - name: Print crash logs (on failure) | |
| if: failure() | |
| shell: bash | |
| run: bash .github/print-crash-logs.sh llama-atmosphere-agent | |
| - if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: error-log-atmosphere-agent-integration | |
| path: | | |
| ${{ github.workspace }}/llama-atmosphere-agent/hs_err_pid*.log | |
| ${{ github.workspace }}/core.* | |
| ${{ github.workspace }}/llama-atmosphere-agent/*.hprof | |
| build-webui: | |
| name: Build WebUI assets (shared) | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Resolve pinned llama.cpp tag from CMakeLists.txt | |
| id: tag | |
| shell: bash | |
| run: | | |
| TAG=$(grep -oE 'GIT_TAG[[:space:]]+b[0-9]+' llama/CMakeLists.txt | grep -oE 'b[0-9]+' | head -1) | |
| if [ -z "$TAG" ]; then | |
| echo "could not resolve llama.cpp GIT_TAG (b<nnnn>) from CMakeLists.txt" >&2 | |
| exit 1 | |
| fi | |
| echo "tag=$TAG" >> "$GITHUB_OUTPUT" | |
| echo "Pinned llama.cpp WebUI tag: $TAG" | |
| - name: Checkout llama.cpp tools/ui at the pinned tag | |
| uses: actions/checkout@v7 | |
| with: | |
| repository: ggml-org/llama.cpp | |
| ref: ${{ steps.tag.outputs.tag }} | |
| path: llamacpp-ui | |
| # tools/ui holds the npm project AND the ui.cpp.in / ui.h.in templates; | |
| # scripts/ holds ui-assets.cmake, which consumes them. Both are needed | |
| # since upstream #28445 replaced the embed.cpp host tool. | |
| sparse-checkout: | | |
| tools/ui | |
| scripts | |
| sparse-checkout-cone-mode: true | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: '24' | |
| cache: npm | |
| cache-dependency-path: llamacpp-ui/tools/ui/package-lock.json | |
| - name: Build WebUI (Svelte/Vite) | |
| working-directory: llamacpp-ui/tools/ui | |
| env: | |
| HF_UI_VERSION: ${{ steps.tag.outputs.tag }} | |
| LLAMA_BUILD_NUMBER: ${{ steps.tag.outputs.tag }} | |
| run: | | |
| npm ci --ignore-scripts | |
| npm run build | |
| test -f dist/index.html | |
| - name: Embed assets into ui.cpp / ui.h (upstream scripts/ui-assets.cmake) | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| # Upstream #28445 ("ui : embed assets directly with CMake") deleted the | |
| # tools/ui/embed.cpp host tool this step used to compile, and replaced it | |
| # with scripts/ui-assets.cmake -- a plain `cmake -P` script, no npm and no | |
| # host executable. Priority 1 of its provisioning order is "pre-built | |
| # assets in <UI_SOURCE_DIR>/dist", which is exactly what the npm step | |
| # above produced, so BUILD_UI and HF_ENABLED stay OFF: no second npm run | |
| # and no Hugging Face download happen here. LLAMA_UI_GZIP is upstream's | |
| # own knob and replaces the hand-rolled gzip loop this step used to do. | |
| GEN="${RUNNER_TEMP}/ui-assets" | |
| OUT="${GITHUB_WORKSPACE}/llama/webui-generated" | |
| mkdir -p "$GEN" "$OUT" | |
| cmake \ | |
| "-DUI_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui/tools/ui" \ | |
| "-DUI_BINARY_DIR=${GEN}" \ | |
| "-DLLAMA_SOURCE_DIR=${GITHUB_WORKSPACE}/llamacpp-ui" \ | |
| -DBUILD_UI=OFF \ | |
| -DHF_ENABLED=OFF \ | |
| -DLLAMA_UI_GZIP=ON \ | |
| -P "${GITHUB_WORKSPACE}/llamacpp-ui/scripts/ui-assets.cmake" | |
| # The script also drops a ui-gzip/ working tree next to the generated | |
| # sources; copy only the two files the artifact is defined to carry. | |
| cp "$GEN/ui.cpp" "$GEN/ui.h" "$OUT/" | |
| echo "=== generated WebUI assets ===" | |
| ls -la "$OUT" | |
| # Guard against a silently empty WebUI. A bare `grep LLAMA_UI_HAS_ASSETS` | |
| # does NOT work here and would pass the failure case: upstream's ui.h.in | |
| # emits "/* #undef LLAMA_UI_HAS_ASSETS */" when the table is empty, so the | |
| # token is present either way. (The old embed.cpp emitted no such line at | |
| # all, which is why the naive grep used to be sufficient.) Assert the ACTIVE | |
| # #define and a non-zero asset count instead -- verified against both paths. | |
| N=$(sed -n 's/.*std::array<llama_ui_asset, \([0-9]\+\)>.*/\1/p' "$OUT/ui.h" | head -1) | |
| if grep -qE '^[[:space:]]*#define[[:space:]]+LLAMA_UI_HAS_ASSETS' "$OUT/ui.h" \ | |
| && [ -n "$N" ] && [ "$N" -gt 0 ]; then | |
| echo "LLAMA_UI_HAS_ASSETS: present, $N assets embedded" | |
| else | |
| echo "ERROR: ui-assets.cmake produced an empty asset table (assets=${N:-unknown})" >&2 | |
| exit 1 | |
| fi | |
| - name: Upload WebUI artifact | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| retention-days: 1 | |
| if-no-files-found: error | |
| crosscompile-linux-x86_64-cuda: | |
| name: Cross-Compile manylinux_2_28 x86_64 (CUDA) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| # CUDA cache. build_cuda_linux.sh execs build.sh, so the same sccache probe guards this job. | |
| # build.sh also wraps nvcc (CMAKE_CUDA_COMPILER_LAUNCHER=sccache) for CUDA builds, so the | |
| # per-arch .cu device passes — the dominant cost of this job — cache over Depot alongside the | |
| # gcc host TUs. Verified on a warm run: 100% hit on CUDA / CUBIN / device-code (139 CUDA hits, | |
| # 99.86% overall), cutting the job from ~51 min cold to ~15 min warm. The job therefore always | |
| # builds the FULL CMAKE_CUDA_ARCHITECTURES set (no single-arch shortcut) and leans on the warm | |
| # cache for speed, so every artifact stays release-safe (runs on every GPU generation) on PR / | |
| # push as well as publish. CUDA_FAST_BUILD still exists in build_cuda_linux.sh as a LOCAL-dev | |
| # knob, but CI no longer sets it. The first-run sccache debug diagnostics (SCCACHE_LOG / | |
| # SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that caching is confirmed; build.sh still | |
| # prints the `sccache --show-stats` hit table at the end of every run. Inert without DEPOT_TOKEN | |
| # (fork PRs) or use_cache=false. | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE" | |
| steps: | |
| - name: Free up disk space | |
| # The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree; | |
| # same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is | |
| # kept (default) so nothing a later setup-* step relies on is removed. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| .github/dockcross/dockcross-manylinux_2_28-x64 .github/build_cuda_linux.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cuda13-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| crosscompile-linux-x86_64: | |
| name: Cross-Compile manylinux2014 x86_64 | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| # Phase 2 dockcross cache rollout — job 1, VERIFIED green in CI (PR #245): sccache v0.16.0 | |
| # probe passed in-container (devtoolset-10 gcc), cache ON over Depot WebDAV (cold run: 275 | |
| # objects stored). Steady-state env below — the first-run diagnostics (SCCACHE_LOG / | |
| # SCCACHE_ERROR_LOG / RUST_BACKTRACE) were dropped now that it is proven. Inert without | |
| # DEPOT_TOKEN (fork PRs) or with use_cache=false; a crashing sccache still falls back to a | |
| # green uncached build via the build.sh probe. | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE" | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| .github/dockcross/dockcross-manylinux2014-x64 .github/build.sh "-DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| crosscompile-linux-aarch64: | |
| name: Build and Test Linux aarch64 | |
| needs: [startgate, build-webui] | |
| # Native ARM64 build on GitHub's free arm64 runner, mirroring upstream llama.cpp's | |
| # `ubuntu-cpu` aarch64 release job (ubuntu-24.04-arm + GCC 14). Replaces the former dockcross | |
| # `linux-arm64-lts` cross-compile (GCC 8.5, glibc 2.17), which can no longer compile llama.cpp | |
| # b9789 — its C++17 CTAD-in-`new` needs GCC >= 12. Building natively also lets us run the C++ | |
| # unit suite (ctest) on real ARM hardware for the first time (the cross build ran no tests). | |
| # Trade-off: the glibc floor rises 2.17 -> ~2.39, the same envelope upstream's own ARM binaries | |
| # require. GGML_NATIVE=OFF keeps the artifact portable across ARMv8 CPU generations (no | |
| # build-host -march baked in). The job id is kept (a `needs:` target downstream); only the | |
| # display name changed, so update any branch-protection required-check that pinned the old name. | |
| runs-on: ubuntu-24.04-arm | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install toolchain (GCC 14, mirrors upstream llama.cpp ARM release) | |
| run: | | |
| sudo apt-get update | |
| sudo apt-get install -y gcc-14 g++-14 | |
| echo "CC=gcc-14" >> "$GITHUB_ENV" | |
| echo "CXX=g++-14" >> "$GITHUB_ENV" | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DOS_NAME=Linux -DOS_ARCH=aarch64 -DGGML_NATIVE=OFF -DBUILD_TESTING=ON" | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-linux-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-linux-s390x: | |
| name: Build and Test Linux s390x (big-endian, qemu) | |
| needs: [startgate, build-webui] | |
| # Cross-compile for IBM Z (s390x, BIG-ENDIAN) with the GCC cross toolchain, then run the full | |
| # C++ unit suite under qemu-user — a real big-endian correctness gate for our helpers and | |
| # serializers (esp. the little-endian WAV writer, JSON/token/embedding transforms). The BUILD | |
| # is native speed (x86 cross-gcc); only the tiny test binary is emulated. s390x is a CPU | |
| # platform like aarch64 and ships as the cpu-linux-s390x natives jar. Model-backed Java tests | |
| # are NOT run under emulation (a JVM + GGUF inference under qemu-user is slow/flaky); the | |
| # C++ gate covers the actual byte-order risk since the Java<->JNI boundary uses host-native | |
| # array copies. GGML_OPENMP=OFF avoids cross-libgomp issues (ggml uses | |
| # its own std::thread pool). CMAKE_CROSSCOMPILING_EMULATOR makes ctest run the s390x exe via qemu; | |
| # QEMU_LD_PREFIX lets the emulated binary find the s390x sysroot libs. | |
| runs-on: ubuntu-latest | |
| env: | |
| QEMU_LD_PREFIX: /usr/s390x-linux-gnu | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install s390x cross toolchain + qemu-user | |
| run: | | |
| sudo apt-get update | |
| sudo apt-get install -y gcc-s390x-linux-gnu g++-s390x-linux-gnu qemu-user-static | |
| # NOTE: GGML_NATIVE=OFF is required here (an x86 build host must not bake -march=native into an | |
| # s390x artifact), and it has a non-obvious second effect: ggml declares | |
| # `option(GGML_VXE "ggml: enable vxe" ${GGML_NATIVE})`, so VXE is off too. No `-mvx -mzvector` is | |
| # passed, `__VEC__` stays undefined, and ggml-cpu-impl.h's `#if defined(__s390x__) && defined(__VEC__)` | |
| # never self-defines `__VXE__`/`__VXE2__`. This job therefore builds a SCALAR s390x binary -- which is | |
| # exactly right for what it is (a big-endian correctness gate for our own layer, not a perf target). | |
| # Do NOT "fix" a VXE-related compile error by adding -DGGML_VXE=ON: that define sets __VXE__ and | |
| # __VXE2__ together while -march stays at the toolchain default arch11, so every z14+ builtin is | |
| # rejected. See the `0013` note in docs/history/dropped-llama-patches.md for the measured | |
| # comparison (the patch itself is gone -- upstream merged it as ggml-org/llama.cpp#28775). | |
| - name: Build libraries (cross-compile s390x) | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_NATIVE=OFF -DGGML_OPENMP=OFF -DBUILD_TESTING=ON -DCMAKE_SYSTEM_NAME=Linux -DCMAKE_SYSTEM_PROCESSOR=s390x -DCMAKE_C_COMPILER=s390x-linux-gnu-gcc -DCMAKE_CXX_COMPILER=s390x-linux-gnu-g++ -DCMAKE_CROSSCOMPILING_EMULATOR=/usr/bin/qemu-s390x-static -DOS_NAME=Linux -DOS_ARCH=s390x" | |
| - name: Run C++ unit tests under qemu-s390x (big-endian gate) | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-linux-s390x | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-linux-x86_64-vulkan: | |
| name: Build Linux x86_64 Vulkan | |
| needs: [startgate, build-webui] | |
| # Native ubuntu build (NOT dockcross) — the Vulkan SDK is trivial to apt-install here, and | |
| # upstream llama.cpp builds its ubuntu-vulkan artifact the same way. GPU runtime libvulkan.so.1 | |
| # is supplied by the consumer's driver (nothing bundled). GitHub runners have NO GPU, so this | |
| # is a BUILD-ONLY job (no -DBUILD_TESTING/ctest: a Vulkan-linked jllama_test errors enumerating | |
| # devices on a GPU-less runner — same rationale as the Windows GPU jobs). GGML_NATIVE=OFF keeps | |
| # the artifact portable across x86_64 CPU generations. Trade-off vs the manylinux CPU jar: the | |
| # glibc floor rises to the ubuntu-latest baseline (same as the native aarch64 job). build.sh | |
| # self-fetches sccache; the probe guards it (a miss just builds uncached). | |
| runs-on: ubuntu-latest | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install Vulkan SDK (headers + loader + glslc shader compiler) | |
| run: | | |
| sudo apt-get update | |
| sudo apt-get install -y libvulkan-dev glslc glslang-tools spirv-headers | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-vulkan-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-linux-aarch64-vulkan: | |
| name: Build Linux aarch64 Vulkan | |
| needs: [startgate, build-webui] | |
| # Native ARM64 Vulkan build on GitHub's free arm64 runner (same runner as the aarch64 CPU job). | |
| # Build-only (GPU-less runner); GGML_NATIVE=OFF for portability across ARMv8 generations; GCC 14 | |
| # to match the aarch64 CPU job. Ships as the vulkan-linux-aarch64 natives jar. | |
| runs-on: ubuntu-24.04-arm | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install toolchain (GCC 14) + Vulkan SDK | |
| run: | | |
| sudo apt-get update | |
| sudo apt-get install -y gcc-14 g++-14 libvulkan-dev glslc glslang-tools spirv-headers | |
| echo "CC=gcc-14" >> "$GITHUB_ENV" | |
| echo "CXX=g++-14" >> "$GITHUB_ENV" | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_VULKAN=ON -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=aarch64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-vulkan-linux-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| crosscompile-android-aarch64: | |
| name: Cross-Compile Android aarch64 | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| # Phase 2 dockcross cache rollout — job 4. Same steady-state env as manylinux2014 (job 1); | |
| # the build.sh probe makes it safe to enable without a separate verification run. Inert | |
| # without DEPOT_TOKEN (fork PRs) or use_cache=false. | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE" | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| .github/dockcross/dockcross-android-arm64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-android-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| crosscompile-android-x86_64: | |
| name: Cross-Compile Android x86_64 | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| # Android x86_64 CPU ABI: consumed by the llama-android AAR as jni/x86_64 (so the | |
| # Android emulator — and x86_64 Android devices/Chromebooks — can run the binding) | |
| # and shipped as the cpu-android-x86-64 natives jar. Same dockcross + sccache steady-state env as the arm64 job; the wrapper | |
| # pins the same image tag. Fail-loud and in the package/publish needs graphs. | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE" | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| .github/dockcross/dockcross-android-x86_64 .github/build.sh "-DOS_NAME=Linux-Android -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-android-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| crosscompile-android-aarch64-opencl: | |
| name: Cross-Compile Android aarch64 (OpenCL/Adreno) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| # Phase 2 dockcross cache rollout — job 5. build_opencl_android.sh stages the OpenCL | |
| # headers/loader, then delegates the jllama cmake build to build.sh (which owns the | |
| # sccache probe + launcher). Same steady-state env as the other dockcross jobs. Inert | |
| # without DEPOT_TOKEN (fork PRs) or use_cache=false. | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| DOCKCROSS_ARGS: "-e SCCACHE_WEBDAV_ENDPOINT -e SCCACHE_WEBDAV_TOKEN -e USE_CACHE" | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| .github/dockcross/dockcross-android-arm64 .github/build_opencl_android.sh "-DOS_NAME=Linux-Android -DOS_ARCH=aarch64 -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-opencl-android-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| # --------------------------------------------------------------------------- | |
| # Android AAR packaging (net.ladenthin:llama-android + llama-android-opencl). | |
| # Plain-Gradle AAR assembly — no AGP and no Android SDK needed to BUILD (an AAR | |
| # is a documented zip; see llama-android/README.md); AGP is only needed to | |
| # CONSUME it, which the consumer smoke test below exercises on the runner's | |
| # preinstalled Android SDK: a minimal app resolves the AAR from mavenLocal and | |
| # runs a full R8 release build (validating AAR format, manifest minSdk merge, | |
| # jni/ packaging, and the shipped consumer proguard rules). Structural checks | |
| # additionally pin the AAR entries and the 16 KB LOAD-segment alignment | |
| # (Google Play requirement for Android 15+ targets). Fail-loud and in the | |
| # publish `needs:` graphs — a broken AAR blocks publishing, same policy as | |
| # every native artifact job. | |
| # --------------------------------------------------------------------------- | |
| package-android-aar: | |
| name: Package + Validate Android AARs | |
| needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, crosscompile-android-aarch64-opencl] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| # AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0. | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Build core jar (byte-identical classes payload for the AAR) | |
| uses: ./.github/actions/build-core | |
| with: | |
| goal: package | |
| - name: Download Android CPU natives (arm64) | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-cpu-android-aarch64 | |
| path: stage/cpu/ | |
| - name: Download Android CPU natives (x86_64) | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-cpu-android-x86-64 | |
| path: stage/cpu-x86_64/ | |
| - name: Download Android OpenCL natives | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-opencl-android-aarch64 | |
| path: stage/opencl/ | |
| - name: Stage natives for the AAR build | |
| run: | | |
| mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a | |
| cp stage/cpu/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/ | |
| cp stage/cpu-x86_64/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/ | |
| cp stage/opencl/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/ | |
| - name: Assemble AARs + publish to mavenLocal | |
| run: gradle -p llama-android aarCpu aarOpencl publishToMavenLocal | |
| - name: Validate AAR structure | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1) | |
| for flavor in llama-android llama-android-opencl; do | |
| AAR="llama-android/build/aar/${flavor}-${VERSION}.aar" | |
| echo "== validating $AAR" | |
| test -f "$AAR" | |
| # The CPU AAR is multi-ABI (arm64 devices + x86_64 emulators/devices); | |
| # the OpenCL flavor stays arm64-only (Adreno is Qualcomm ARM hardware). | |
| ABIS="arm64-v8a" | |
| if [ "$flavor" = "llama-android" ]; then ABIS="arm64-v8a x86_64"; fi | |
| ENTRIES="AndroidManifest.xml classes.jar proguard.txt R.txt" | |
| for abi in $ABIS; do ENTRIES="$ENTRIES jni/$abi/libjllama.so"; done | |
| for entry in $ENTRIES; do | |
| unzip -l "$AAR" | grep -q "${entry}$" || { echo "::error::$AAR is missing $entry"; exit 1; } | |
| done | |
| unzip -p "$AAR" AndroidManifest.xml | grep -q 'android:minSdkVersion="28"' \ | |
| || { echo "::error::$AAR manifest lost minSdkVersion 28"; exit 1; } | |
| unzip -p "$AAR" classes.jar > /tmp/aar-classes.jar | |
| unzip -l /tmp/aar-classes.jar | grep -q "net/ladenthin/llama/LlamaModel.class" \ | |
| || { echo "::error::$AAR classes.jar lost LlamaModel"; exit 1; } | |
| if unzip -l /tmp/aar-classes.jar | grep -qE "module-info\.class|net/ladenthin/llama/(Linux|Mac|Windows)/"; then | |
| echo "::error::$AAR classes.jar carries module-info or desktop native resources"; exit 1 | |
| fi | |
| # The libraries themselves (bionic-only DT_NEEDED + libOpenCL.so for the OpenCL | |
| # flavour, every LOAD segment 16 KB aligned for Google Play) are checked on the staged | |
| # tree below, by the same allowlists the package job applies to every natives jar. | |
| cmp -s <(unzip -p "$AAR" jni/arm64-v8a/libjllama.so) \ | |
| "llama-android/natives/$([ "$flavor" = llama-android ] && echo cpu || echo opencl)/arm64-v8a/libjllama.so" \ | |
| || { echo "::error::$AAR jni/arm64-v8a/libjllama.so is not the staged library"; exit 1; } | |
| done | |
| - name: Android libraries -- dependency allowlist + 16 KB page alignment (buildcheck/nativedeps.py) | |
| run: python3 .github/verify-native-deps.py stage | |
| - name: AGP consumer smoke test (R8 release build from mavenLocal) | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1) | |
| gradle -p .github/android-consumer-test assembleRelease "-PjllamaVersion=${VERSION}" | |
| APK=.github/android-consumer-test/app/build/outputs/apk/release/app-release.apk | |
| test -f "$APK" | |
| for abi in arm64-v8a x86_64; do | |
| unzip -l "$APK" | grep -q "lib/$abi/libjllama.so" \ | |
| || { echo "::error::consumer APK is missing lib/$abi/libjllama.so"; exit 1; } | |
| done | |
| # The AAR's consumer proguard.txt must have carried the binding through R8. | |
| unzip -p "$APK" "classes*.dex" | grep -aq "Lnet/ladenthin/llama/LlamaModel;" \ | |
| || { echo "::error::R8 stripped net.ladenthin.llama.LlamaModel — consumer proguard rules broken"; exit 1; } | |
| - name: Upload AARs | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-android-aars | |
| path: llama-android/build/aar/*.aar | |
| if-no-files-found: error | |
| # --------------------------------------------------------------------------- | |
| # On-emulator runtime validation of the Android AAR: boots a KVM-accelerated | |
| # x86_64 emulator (GitHub Linux runners have KVM; arm64 images cannot run here, | |
| # which is exactly why the AAR carries the jni/x86_64 ABI), publishes the CPU AAR | |
| # to mavenLocal, adb-pushes the already-cached tiny draft model, and runs the | |
| # consumer fixture's connectedDebugAndroidTest — System.loadLibrary from the APK, | |
| # pure-Java GgufInspector on-device, and real native inference on Android/bionic. | |
| # RELEASE GATE (in both publish needs graphs) since PR #298: the job ran | |
| # flake-free through the PR's validation cycle (boot ~30 s, on-device inference | |
| # green), so a broken on-device runtime now blocks publishing — same fail-loud | |
| # policy as every native artifact job. If emulator-boot flakiness ever appears, | |
| # re-run the job first; demote it from the needs graphs only as a last resort. | |
| # --------------------------------------------------------------------------- | |
| test-android-emulator: | |
| name: Android emulator on-device test (x86_64) | |
| needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Free up disk space | |
| # The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages" | |
| # (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap | |
| # file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| with: | |
| android: false | |
| large-packages: false | |
| tool-cache: false | |
| swap-storage: false | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| # AGP 9.4.0 (the .github/android-consumer-test fixture) requires Gradle >= 9.6.0. | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Enable KVM group permissions (GitHub-hosted runner) | |
| run: | | |
| echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules | |
| sudo udevadm control --reload-rules | |
| sudo udevadm trigger --name-match=kvm | |
| - name: Build core jar (classes payload for the AAR) | |
| uses: ./.github/actions/build-core | |
| with: | |
| goal: package | |
| - uses: ./.github/actions/publish-cpu-aar-local | |
| - uses: ./.github/actions/restore-models | |
| - name: Free disk space for the emulator (only the draft model is needed on-device) | |
| # Same guard as test-android-llmservice: the AVD userdata partition needs ~7.4 GB; drop the | |
| # rest of the restored ~10 GB GGUF cache (this job adb-pushes only the tiny draft model) plus | |
| # the few preinstalled toolchains the free-disk-space step leaves, so the emulator can create | |
| # userdata and boot. | |
| run: | | |
| echo "Disk before cleanup:"; df -h / | tail -1 | |
| find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true | |
| sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true | |
| echo "Disk after cleanup:"; df -h / | tail -1 | |
| - name: Run on-emulator instrumentation (connectedDebugAndroidTest) | |
| uses: reactivecircus/android-emulator-runner@v2 | |
| with: | |
| api-level: 30 | |
| arch: x86_64 | |
| target: default | |
| disable-animations: true | |
| emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim | |
| # One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh, | |
| # so shell control flow must live in the committed helper script. | |
| script: .github/run-android-emulator-test.sh | |
| - name: Upload instrumentation reports (on failure) | |
| if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: android-emulator-test-reports | |
| path: .github/android-consumer-test/app/build/reports/androidTests/ | |
| if-no-files-found: ignore | |
| # --------------------------------------------------------------------------- | |
| # The shippable Android app "LLM Service" (android-llmservice) — a KISS, fully-offline | |
| # on-device chat app consuming the llama-android AAR + llama-kotlin facade. Split into TWO jobs | |
| # so the installable artifacts are ALWAYS produced even when the on-device UI test is flaky/slow: | |
| # * build-android-llmservice — builds the signed release AAB (real upload key when the | |
| # ANDROID_UPLOAD_KEYSTORE_BASE64 secret is set, else debug-signed) + the installable APK and | |
| # uploads both. No emulator, so a flaky/slow emulator can NEVER block getting the APK. | |
| # * test-android-llmservice — a SEPARATE, non-gating check that boots the KVM x86_64 emulator | |
| # and runs the app's Compose UI test (type a prompt, tap Send) against real on-device | |
| # inference. It can go red on its own without stopping the build/artifacts. | |
| # Neither is a publish gate (unlike test-android-emulator): a Compose/AGP toolchain hiccup must | |
| # not block a Maven Central release of the library. Keep test-android-llmservice non-required in | |
| # branch protection to keep the emulator test optional (visible-but-non-blocking). | |
| # --------------------------------------------------------------------------- | |
| build-android-llmservice: | |
| name: Build the LLM Service Android app (AAB + APK) | |
| needs: [crosscompile-android-aarch64, crosscompile-android-x86_64] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| # AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0; also satisfies Kotlin | |
| # 2.4's Gradle-plugin floor. The AAR/lib-only jobs stay on an older Gradle since | |
| # they build no AGP project. | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Build core jar + install llama-kotlin facade to mavenLocal | |
| # install (not package) so the app's Gradle build resolves llama-kotlin + | |
| # the core POM from mavenLocal. llama-kotlin's core dep is provided-scope, so the | |
| # desktop JAR is never pulled into the APK — the AAR supplies net.ladenthin.llama.*. | |
| uses: ./.github/actions/build-core | |
| with: | |
| modules: llama,llama-kotlin | |
| - uses: ./.github/actions/publish-cpu-aar-local | |
| - name: Decode upload keystore (if configured) | |
| # Optional Play upload key. Set the ANDROID_UPLOAD_KEYSTORE_BASE64 secret | |
| # (base64 of a PKCS12/JKS upload keystore) plus the three password/alias secrets | |
| # below to sign the release AAB with a real upload key. Without it the release | |
| # build falls back to debug signing (app/build.gradle.kts) so this step is inert | |
| # on forks/PRs where secrets are withheld. | |
| env: | |
| KEYSTORE_B64: ${{ secrets.ANDROID_UPLOAD_KEYSTORE_BASE64 }} | |
| run: | | |
| if [ -n "${KEYSTORE_B64}" ]; then | |
| echo "${KEYSTORE_B64}" | base64 -d > "${RUNNER_TEMP}/upload-keystore.jks" | |
| echo "JLLAMA_UPLOAD_STORE_FILE=${RUNNER_TEMP}/upload-keystore.jks" >> "$GITHUB_ENV" | |
| echo "Upload keystore decoded -> release AAB will be signed with the upload key." | |
| else | |
| echo "No ANDROID_UPLOAD_KEYSTORE_BASE64 secret -> release AAB will be debug-signed." | |
| fi | |
| - name: Build release bundle (AAB) + installable APK | |
| env: | |
| JLLAMA_UPLOAD_STORE_PASSWORD: ${{ secrets.ANDROID_UPLOAD_STORE_PASSWORD }} | |
| JLLAMA_UPLOAD_KEY_ALIAS: ${{ secrets.ANDROID_UPLOAD_KEY_ALIAS }} | |
| JLLAMA_UPLOAD_KEY_PASSWORD: ${{ secrets.ANDROID_UPLOAD_KEY_PASSWORD }} | |
| run: | | |
| VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1) | |
| # bundleRelease -> app-release.aab (Play upload, NOT directly installable); | |
| # assembleRelease -> app-release.apk (sideloadable installable, debug-signed unless the | |
| # upload-key secrets are set). The debug + androidTest APKs are built by the emulator | |
| # test job (test-android-llmservice), not here — this job never touches the emulator. | |
| gradle -p android-llmservice bundleRelease assembleRelease "-PjllamaVersion=${VERSION}" | |
| test -f android-llmservice/app/build/outputs/bundle/release/app-release.aab | |
| test -f android-llmservice/app/build/outputs/apk/release/app-release.apk | |
| - name: Upload release bundle (AAB — for Play upload) | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: android-llmservice-aab | |
| path: android-llmservice/app/build/outputs/bundle/release/*.aab | |
| if-no-files-found: error | |
| - name: Upload installable APK (release — sideload / adb install) | |
| # Directly installable APK, downloadable from this run's Artifacts (independent of any | |
| # release). Debug-signed unless the ANDROID_UPLOAD_* secrets are set. This is the file to | |
| # grab to try the app on a phone; the .aab above is only for the Play Console. | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: android-llmservice-apk | |
| path: android-llmservice/app/build/outputs/apk/release/*.apk | |
| if-no-files-found: error | |
| test-android-llmservice: | |
| name: LLM Service app UI test on emulator (non-gating) | |
| needs: [crosscompile-android-aarch64, crosscompile-android-x86_64, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Free up disk space | |
| # The AVD userdata partition needs ~7.4 GB. Android SDK and the apt "large packages" | |
| # (it removes libgl1-mesa-dri among others) are kept for the emulator, and so is the swap | |
| # file it may lean on; the tool cache is kept because setup-java/setup-gradle install into it. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| with: | |
| android: false | |
| large-packages: false | |
| tool-cache: false | |
| swap-storage: false | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: temurin | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| # AGP 9.4.0 (android-llmservice) requires Gradle >= 9.6.0. | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Enable KVM group permissions (GitHub-hosted runner) | |
| run: | | |
| echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules | |
| sudo udevadm control --reload-rules | |
| sudo udevadm trigger --name-match=kvm | |
| - name: Build core jar + install llama-kotlin facade to mavenLocal | |
| uses: ./.github/actions/build-core | |
| with: | |
| modules: llama,llama-kotlin | |
| - uses: ./.github/actions/publish-cpu-aar-local | |
| - uses: ./.github/actions/restore-models | |
| - name: Free disk space for the emulator (only the draft model is needed on-device) | |
| # The AVD userdata partition needs ~7.4 GB; the full ~10 GB GGUF cache restore leaves too | |
| # little free, so the emulator FATALs ("Not enough space to create userdata partition") and | |
| # the boot poll loops until timeout. This job only adb-pushes the tiny draft model, so drop | |
| # the rest of the restored cache plus what the free-disk-space step leaves (PowerShell, CodeQL). | |
| run: | | |
| echo "Disk before cleanup:"; df -h / | tail -1 | |
| find models -type f ! -name "${DRAFT_MODEL_NAME}" -delete 2>/dev/null || true | |
| sudo rm -rf /usr/local/share/powershell /opt/hostedtoolcache/CodeQL 2>/dev/null || true | |
| echo "Disk after cleanup:"; df -h / | tail -1 | |
| - name: Run LLM Service UI test on emulator (connectedDebugAndroidTest) | |
| uses: reactivecircus/android-emulator-runner@v2 | |
| with: | |
| api-level: 30 | |
| arch: x86_64 | |
| target: default | |
| disable-animations: true | |
| emulator-options: -no-snapshot -no-window -gpu swiftshader_indirect -noaudio -no-boot-anim | |
| # One line on purpose: the emulator-runner executes `script:` LINE BY LINE via sh. | |
| script: .github/run-android-llmservice-test.sh | |
| - name: Upload instrumentation reports (on failure) | |
| if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: android-llmservice-test-reports | |
| path: android-llmservice/app/build/reports/androidTests/ | |
| if-no-files-found: ignore | |
| # --------------------------------------------------------------------------- | |
| # Native build jobs — produce release artifacts + run C++ unit tests | |
| # --------------------------------------------------------------------------- | |
| build-macos-arm64-no-metal: | |
| name: Build and Test macOS 15 arm64 (no Metal) | |
| needs: [startgate, build-webui] | |
| runs-on: macos-15 | |
| env: | |
| BUILD_JOBS: 2 | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| run: brew install sccache | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh -DGGML_METAL=OFF -DGGML_NATIVE=OFF -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: macos-15-no-metal | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-macos-arm64-metal: | |
| name: Build and Test macOS 14 arm64 (Metal) | |
| needs: [startgate, build-webui] | |
| runs-on: macos-14 | |
| env: | |
| BUILD_JOBS: 2 | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| run: brew install sccache | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: macos-14-metal | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-windows-x86_64-msvc: | |
| name: Build and Test Windows 2025 x86_64 (MSVC / VS 2026, classifier) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| .github\build.bat -G "Visual Studio 18 2026" -A "x64" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-msvc-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-windows-x86-msvc: | |
| name: Build and Test Windows 2025 x86 (MSVC / VS 2026, classifier) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| .github\build.bat -G "Visual Studio 18 2026" -A "Win32" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-msvc-windows-x86 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| # --------------------------------------------------------------------------- | |
| # Windows Ninja Multi-Config + sccache — the DEFAULT Windows CPU natives. | |
| # The Visual Studio generator ignores CMAKE_{C,CXX}_COMPILER_LAUNCHER, so only the | |
| # Ninja Multi-Config generator can front cl.exe with sccache over Depot WebDAV | |
| # (build.bat probe-guards it). Both generators use the same MSVC toolchain (cl.exe, | |
| # static /MT CRT) on the same runner, so the produced jllama.dll/llama.dll/ggml.dll | |
| # are functionally equivalent with identical runtime dependencies — the only delta | |
| # is build-system plumbing + caching. The Ninja build therefore ships as the | |
| # cpu-windows-* natives jars; the MSVC build above as msvc-windows-*, for anyone who | |
| # wants the Visual-Studio-generator natives. Upstream llama.cpp also | |
| # builds its Windows artifacts with Ninja Multi-Config + MSVC. | |
| # --------------------------------------------------------------------------- | |
| build-windows-x86_64: | |
| name: Build and Test Windows 2025 x86_64 (Ninja Multi-Config + sccache, default) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86_64 -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-windows-x86: | |
| name: Build and Test Windows 2025 x86 (Ninja Multi-Config + sccache, default) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x86) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x86 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DOS_NAME=Windows -DOS_ARCH=x86 -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-windows-x86 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| build-windows-arm64: | |
| name: Build and Test Windows 11 arm64 (Ninja Multi-Config, default) | |
| needs: [startgate, build-webui] | |
| # Native arm64 build on GitHub's free windows-11-arm runner; ships as the cpu-windows-aarch64 | |
| # natives jar: OSInfo maps a Windows-on-ARM JVM (os.arch=aarch64) to Windows/aarch64, the same | |
| # path CMake emits here. sccache: the native aarch64-pc-windows-msvc release, wrapping clang-cl | |
| # (guarded by build.bat's probe + uncached retry like every other Windows job). | |
| # | |
| # Compiler: clang-cl, NOT MSVC cl.exe. ggml's ggml-cpu/CMakeLists.txt aborts with "MSVC is not | |
| # supported for ARM, use clang" via `if (MSVC AND NOT CMAKE_C_COMPILER_ID STREQUAL "Clang")`. | |
| # clang-cl (LLVM's MSVC-compatible driver) satisfies that guard (its compiler id is "Clang") | |
| # while still leaving CMake's MSVC=TRUE, so our static /MT CRT block (CMAKE_MSVC_RUNTIME_LIBRARY | |
| # in CMakeLists.txt) keeps applying and the generator stays Ninja Multi-Config. msvc-dev-cmd | |
| # (arm64) supplies the MSVC headers/libs/linker AND the bundled clang-cl / lld-link under | |
| # VC\Tools\Llvm\ARM64, so no separate LLVM install is needed. | |
| # | |
| # GGML_OPENMP=OFF: with clang-cl, ggml links LLVM's OpenMP (libomp.lib -> needs libomp140.aarch64.dll | |
| # at runtime), which is NOT on PATH like MSVC's ambient vcomp140.dll on x64 — so gtest_discover_tests | |
| # (and any consumer) failed to launch the binary with 0xc0000135 STATUS_DLL_NOT_FOUND. Turning OpenMP | |
| # off makes ggml use its own std::thread threadpool, so the arm64 jllama.dll (and the test exe) are | |
| # self-contained with no libomp dependency to ship. The x86_64/x86 jobs keep OpenMP (MSVC vcomp). | |
| runs-on: windows-11-arm | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (arm64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: arm64 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # No mvn compile needed: the JNI header (jllama.h) is committed and the native build | |
| # uses the bundled JNI headers in .github/include, and OS_NAME/OS_ARCH are passed | |
| # explicitly (so the OSInfo-class OS-detection path is skipped) — same as the x86_64 job. | |
| # clang-cl (see the job comment) is required: ggml refuses MSVC cl.exe on ARM. | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DOS_NAME=Windows -DOS_ARCH=aarch64 -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cpu-windows-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| # --------------------------------------------------------------------------- | |
| # Windows GPU classifiers (x86_64 only) — CUDA, Vulkan, OpenCL. | |
| # All three use the same Ninja Multi-Config + MSVC + sccache toolchain as the | |
| # default CPU build; they differ only by the GGML backend flag (and the build-time | |
| # SDK each needs). CMakeLists.txt routes each backend's output to its own | |
| # src/main/natives/.../Windows/x86_64/<backend>/ directory, which the natives Maven | |
| # profile turns into a natives jar. GPU runtime libraries are NOT bundled — the | |
| # consumer's GPU driver / toolkit provides them (CUDA: cudart64_13/cublas64_13 from the CUDA Toolkit; Vulkan: | |
| # vulkan-1.dll from the driver; OpenCL: System32\OpenCL.dll from the driver). | |
| # NOTE: GitHub-hosted Windows runners have NO GPU, so these jobs build + run the | |
| # C++ unit suite (ctest, CPU-only) but cannot run model-backed GPU inference; | |
| # end-to-end GPU validation is local / self-hosted. | |
| # --------------------------------------------------------------------------- | |
| build-windows-x86_64-cuda: | |
| name: Build Windows 2025 x86_64 CUDA (Ninja + sccache) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install CUDA Toolkit 13.4 (NVIDIA redist archives) | |
| # KEEP IN SYNC WITH UPSTREAM: the component set and versions mirror llama.cpp's | |
| # .github/actions/windows-setup-cuda ("Install Cuda Toolkit 13.4 for x64") at the pinned | |
| # GIT_TAG. Jimver/cuda-toolkit (used here up to CUDA 13.3.1) has no 13.4 in any release or | |
| # on master, so the toolkit is assembled from NVIDIA's per-component redist zips instead — | |
| # which is also how upstream builds its own Windows CUDA release. cuda_crt is listed | |
| # explicitly: since 13.x the nvcc crt headers (crt/host_config.h) ship in their own | |
| # archive, and without them CMake's CUDA compiler detection fails at configure. | |
| shell: pwsh | |
| run: | | |
| $ErrorActionPreference = "Stop" | |
| $cuda = "C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v13.4" | |
| $base = "https://developer.download.nvidia.com/compute/cuda/redist" | |
| $parts = @( | |
| "cuda_crt/cuda_crt-windows-x86_64-13.4.59", | |
| "cuda_cudart/cuda_cudart-windows-x86_64-13.4.49", | |
| "cuda_nvcc/cuda_nvcc-windows-x86_64-13.4.59", | |
| "cuda_nvrtc/cuda_nvrtc-windows-x86_64-13.4.59", | |
| "libcublas/libcublas-windows-x86_64-13.7.0.27", | |
| "libnvvm/libnvvm-windows-x86_64-13.4.59", | |
| "cuda_nvtx/cuda_nvtx-windows-x86_64-13.4.49", | |
| "cuda_profiler_api/cuda_profiler_api-windows-x86_64-13.4.49", | |
| "visual_studio_integration/visual_studio_integration-windows-x86_64-13.4.49", | |
| "cccl/cccl-windows-x86_64-13.3.4.2.1" | |
| ) | |
| New-Item -ItemType Directory -Force -Path $cuda | Out-Null | |
| foreach ($p in $parts) { | |
| $dir, $name = $p.Split("/") | |
| $url = "$base/$dir/windows-x86_64/$name-archive.zip" | |
| $zip = "$env:RUNNER_TEMP\$name.zip" | |
| Write-Host "Downloading $url" | |
| Invoke-WebRequest -Uri $url -OutFile $zip | |
| Expand-Archive -Path $zip -DestinationPath "$env:RUNNER_TEMP\cuda-redist" -Force | |
| Copy-Item -Path "$env:RUNNER_TEMP\cuda-redist\$name-archive\*" -Destination $cuda -Recurse -Force | |
| } | |
| "$cuda\bin" | Out-File -FilePath $env:GITHUB_PATH -Append -Encoding utf8 | |
| "CUDA_PATH=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8 | |
| "CUDA_PATH_V13_4=$cuda" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # GPU jobs build the artifact only — no -DBUILD_TESTING / ctest. The C++ unit | |
| # suite is CPU-only and fully covered by the `C++ Tests` job + the CPU Windows | |
| # jobs; a GPU-linked jllama_test.exe cannot be discovered/run on a GPU-less | |
| # GitHub runner (it errors probing for a CUDA device -> ctest *_NOT_BUILT). | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DGGML_CUDA=ON -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-cuda13-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-x86_64-vulkan: | |
| name: Build Windows 2025 x86_64 Vulkan (Ninja + sccache) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install Vulkan SDK | |
| uses: jakoch/install-vulkan-sdk-action@v1.6.0 | |
| with: | |
| vulkan_version: 1.4.350.0 | |
| cache: true | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # Build the artifact only (see the CUDA job's note: GPU-less runner can't run a | |
| # GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs). | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DGGML_VULKAN=ON -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-vulkan-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-x86_64-opencl: | |
| name: Build Windows 2025 x86_64 OpenCL (Ninja + sccache) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # Build the artifact only (see the CUDA job's note: GPU-less runner can't run a | |
| # GPU-linked jllama_test; the C++ unit suite is covered by the CPU jobs). | |
| run: | | |
| .github\build_opencl_windows.bat -G "Ninja Multi-Config" -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-opencl-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| # --------------------------------------------------------------------------- | |
| # Additional GPU-backend classifiers (fail-loud, same wiring as the CUDA/Vulkan/ | |
| # OpenCL jobs): AMD ROCm/HIP, Intel SYCL (oneAPI), Windows-on-ARM OpenCL (Adreno), | |
| # Intel OpenVINO. All BUILD-ONLY (GitHub runners have no AMD/Intel/Adreno GPU, and | |
| # no ctest — a GPU-linked jllama_test can't enumerate a device). GPU runtime libs | |
| # are NOT bundled — the consumer's driver/toolkit supplies them. CMakeLists.txt | |
| # routes each backend to its own src/main/natives/<OS>/<ARCH>/<backend>/ directory; the | |
| # natives Maven profile turns it into a natives jar. Toolchain install steps are first-pass — | |
| # if a vendor URL/version 404s in CI, adjust it (the failure is intentional signal). | |
| # --------------------------------------------------------------------------- | |
| build-linux-x86_64-rocm: | |
| name: Build Linux x86_64 ROCm/HIP (AMD) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - name: Free up disk space | |
| # The TheRock wheels unpack to several GB; upstream's ubuntu-rocm job frees the runner | |
| # the same way. Runs first, before setup-java puts the JDK into the tool cache. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| with: | |
| tool-cache: true | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install ROCm/HIP (TheRock wheels) | |
| # KEEP IN SYNC WITH UPSTREAM: ROCM_VERSION tracks the ubuntu-rocm job in llama.cpp's | |
| # .github/workflows/release.yml at the pinned GIT_TAG (GPU targets: see the build step). Since ROCm 7.14 | |
| # AMD builds and releases ROCm through TheRock (https://github.com/ROCm/TheRock), replacing | |
| # the monolithic releases behind the repo.radeon.com apt repo this job used before (it was | |
| # pinned at 6.3.4). The wheels carry the HIP runtime + CMake configs ("libraries") and the | |
| # compilers/headers ("devel"); rocm-sdk reports where they landed. | |
| env: | |
| ROCM_VERSION: "10.0.0" | |
| run: | | |
| python3 -m venv "$RUNNER_TEMP/rocm-venv" | |
| source "$RUNNER_TEMP/rocm-venv/bin/activate" | |
| python -m pip install --upgrade pip | |
| python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==${ROCM_VERSION}" | |
| ROCM_PATH=$(rocm-sdk path --root) | |
| echo "ROCM_PATH=$ROCM_PATH" >> "$GITHUB_ENV" | |
| echo "HIP_PATH=$ROCM_PATH" >> "$GITHUB_ENV" | |
| echo "CMAKE_PREFIX_PATH=$(rocm-sdk path --cmake)" >> "$GITHUB_ENV" | |
| echo "LD_LIBRARY_PATH=$ROCM_PATH/lib:${LD_LIBRARY_PATH:-}" >> "$GITHUB_ENV" | |
| echo "$(rocm-sdk path --bin)" >> "$GITHUB_PATH" | |
| echo "$RUNNER_TEMP/rocm-venv/bin" >> "$GITHUB_PATH" | |
| - name: Build libraries | |
| shell: bash | |
| # Native CMake HIP language (upstream's form): the HIP compiler is ROCm's clang, the C/C++ | |
| # TUs stay on the runner's gcc. The target list is every Linux target TheRock builds | |
| # (its SUPPORTED_GPUS.md) — a superset of upstream llama.cpp's list, which omits | |
| # gfx900/gfx906/gfx90c/gfx1153 ("build passing" only, not release-ready). Better too many | |
| # than too few, but only while they cost nothing: drop an extra the moment it needs a | |
| # patch or blocks a newer ROCm. Linux alone has the Instinct parts (gfx908/90a/942/950): | |
| # ROCm does not support them on Windows. | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_HIP=ON -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGPU_TARGETS=gfx900;gfx906;gfx908;gfx90a;gfx90c;gfx942;gfx950;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Verify the GPU code is compressed | |
| # ggml-hip is compiled with --offload-compress (llama/CMakeLists.txt); without it the | |
| # library carries every target's kernels uncompressed. Fails on an uncompressed bundle. | |
| run: python3 .github/verify-hip-offload-compressed.py llama/src/main/natives | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-rocm-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-x86_64-rocm: | |
| name: Build Windows x86_64 ROCm/HIP (AMD) | |
| needs: [startgate, build-webui] | |
| # windows-2022 (MSVC 14.4x), NOT windows-2025-vs2026 (VS 2026 / MSVC 14.51): ROCm 7.1's | |
| # HIP clang headers (__clang_hip_cmath.h) cannot overload the __host__ __device__ | |
| # isgreater/isless/... that the very new MSVC <cmath> declares via _CLANG_BUILTIN2, so the | |
| # device-code compile fails. Upstream llama.cpp builds win-hip on windows-2022 for the same | |
| # reason (it drives ROCm's own clang and relies on the older MSVC STL). | |
| runs-on: windows-2022 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install ROCm/HIP (TheRock wheels) | |
| shell: pwsh | |
| # KEEP IN SYNC WITH UPSTREAM: mirrors llama.cpp's windows-rocm release job | |
| # (.github/actions/windows-setup-rocm at the pinned GIT_TAG) — the same TheRock wheels as | |
| # the Linux job, in place of the former AMD-Software-PRO-Edition HIP SDK installer. The | |
| # venv lives under C:\TheRock\build, the path upstream uses. | |
| env: | |
| ROCM_VERSION: "10.0.0" | |
| run: | | |
| $ErrorActionPreference = "Stop" | |
| New-Item -Path "C:\TheRock\build" -ItemType Directory -Force | Out-Null | |
| python -m venv C:\TheRock\build\.venv | |
| & C:\TheRock\build\.venv\Scripts\Activate.ps1 | |
| python -m pip install --upgrade pip | |
| python -m pip install --index-url https://stable.repo.amd.com/rocm/whl-next/ "rocm[libraries,devel]==$env:ROCM_VERSION" | |
| if ($LASTEXITCODE -ne 0) { throw "ROCm wheel install failed with exit code $LASTEXITCODE" } | |
| rocm-sdk init | |
| if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" } | |
| $rocm = (rocm-sdk path --root) | |
| if (-not $rocm) { throw "rocm-sdk path --root returned empty - devel package may not be installed" } | |
| $rocm = $rocm.Trim() | |
| "HIP_PATH=$rocm" | Out-File -FilePath $env:GITHUB_ENV -Append | |
| "HIP_DEVICE_LIB_PATH=$rocm\lib\llvm\amdgcn\bitcode" | Out-File -FilePath $env:GITHUB_ENV -Append | |
| "HIP_PLATFORM=amd" | Out-File -FilePath $env:GITHUB_ENV -Append | |
| "LLVM_PATH=$rocm\lib\llvm" | Out-File -FilePath $env:GITHUB_ENV -Append | |
| (rocm-sdk path --bin).Trim() | Out-File -FilePath $env:GITHUB_PATH -Append | |
| "C:\TheRock\build\.venv\Scripts" | Out-File -FilePath $env:GITHUB_PATH -Append | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # Upstream's compiler wiring for TheRock (clang under lib\llvm\bin, not bin\); | |
| # -Wno-error=incompatible-pointer-types is upstream's too. Targets: every Windows target | |
| # TheRock builds — the Linux list minus the Instinct parts, which ROCm has no Windows | |
| # support for. Same rule as on Linux for the extras upstream omits -- here only | |
| # gfx900/gfx906/gfx90c, since upstream's windows-rocm list already has gfx1153. | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DGGML_HIP=ON -DGPU_TARGETS=gfx900;gfx906;gfx90c;gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201 -DCMAKE_PREFIX_PATH="%HIP_PATH%" -DHIP_PATH="%HIP_PATH%" -DCMAKE_C_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_CXX_COMPILER="%HIP_PATH%\lib\llvm\bin\clang++.exe" -DCMAKE_HIP_COMPILER="%HIP_PATH%\lib\llvm\bin\clang.exe" -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Verify the GPU code is compressed | |
| # Same check as the Linux ROCm job: an uncompressed bundle means ~1 GB jllama.dll again. | |
| shell: pwsh | |
| run: python .github/verify-hip-offload-compressed.py llama/src/main/natives | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-rocm-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-linux-x86_64-sycl-fp16: | |
| name: Build Linux x86_64 SYCL fp16 (Intel oneAPI) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - name: Free up disk space | |
| # The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree; | |
| # same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is | |
| # kept (default) so nothing a later setup-* step relies on is removed. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install Intel oneAPI (DPC++ + MKL) | |
| run: | | |
| wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null | |
| echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list | |
| sudo apt-get update | |
| sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| source /opt/intel/oneapi/setvars.sh | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-sycl-fp16-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-linux-x86_64-sycl-fp32: | |
| name: Build Linux x86_64 SYCL fp32 (Intel oneAPI) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - name: Free up disk space | |
| # The GPU toolkit this job installs runs to several GB on top of the llama.cpp build tree; | |
| # same guard upstream llama.cpp's CUDA/ROCm jobs use. Linux-only action; the tool cache is | |
| # kept (default) so nothing a later setup-* step relies on is removed. | |
| uses: ggml-org/free-disk-space@v1.3.1 | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install Intel oneAPI (DPC++ + MKL) | |
| run: | | |
| wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB | gpg --dearmor | sudo tee /usr/share/keyrings/oneapi-archive-keyring.gpg > /dev/null | |
| echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" | sudo tee /etc/apt/sources.list.d/oneAPI.list | |
| sudo apt-get update | |
| sudo apt-get install -y intel-oneapi-compiler-dpcpp-cpp intel-oneapi-mkl-devel | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| source /opt/intel/oneapi/setvars.sh | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-sycl-fp32-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-x86_64-sycl: | |
| name: Build Windows 2025 x86_64 SYCL (Intel oneAPI) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install Intel oneAPI (DPC++ + MKL + oneDNN + TBB) | |
| shell: cmd | |
| # Mirrors upstream llama.cpp's windows-sycl release job: extract the offline | |
| # installer, then run its bootstrapper with the DPC++/MKL/oneDNN/TBB components. | |
| run: | | |
| curl -fSL -o "%RUNNER_TEMP%\oneapi.exe" "https://registrationcenter-download.intel.com/akdlm/IRC_NAS/b60765d1-2b85-4e85-86b6-cb0e9563a699/intel-deep-learning-essentials-2025.3.3.18_offline.exe" | |
| "%RUNNER_TEMP%\oneapi.exe" -s -x -f "%RUNNER_TEMP%\oneapi_extracted" --log "%RUNNER_TEMP%\extract.log" | |
| "%RUNNER_TEMP%\oneapi_extracted\bootstrapper.exe" -s --action install --components=intel.oneapi.win.cpp-dpcpp-common:intel.oneapi.win.mkl.devel:intel.oneapi.win.dnnl:intel.oneapi.win.tbb.devel --eula=accept -p=NEED_VS2022_INTEGRATION=0 --log-dir="%RUNNER_TEMP%" | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| call "C:\Program Files (x86)\Intel\oneAPI\setvars.bat" intel64 --force | |
| .github\build.bat -G "Ninja Multi-Config" -DGGML_SYCL=ON -DCMAKE_C_COMPILER=cl -DCMAKE_CXX_COMPILER=icx -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-sycl-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-arm64-opencl: | |
| name: Build Windows 11 arm64 OpenCL (Adreno) | |
| needs: [startgate, build-webui] | |
| # Windows-on-ARM OpenCL (Snapdragon X / Adreno). Same clang-cl + GGML_OPENMP=OFF | |
| # toolchain as the arm64 CPU job (ggml refuses MSVC cl.exe on ARM). Ships as the | |
| # opencl-windows-aarch64 natives jar. build_opencl_windows.bat stages the | |
| # OpenCL headers + ICD loader before delegating to build.bat. | |
| runs-on: windows-11-arm | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (arm64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: arm64 | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| run: | | |
| .github\build_opencl_windows.bat -G "Ninja Multi-Config" -DCMAKE_C_COMPILER=clang-cl -DCMAKE_CXX_COMPILER=clang-cl -DGGML_OPENMP=OFF -DGGML_OPENCL=ON -DGGML_OPENCL_EMBED_KERNELS=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON -DOS_NAME=Windows -DOS_ARCH=aarch64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-opencl-windows-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-linux-x86_64-openvino: | |
| name: Build Linux x86_64 OpenVINO (Intel) | |
| needs: [startgate, build-webui] | |
| runs-on: ubuntu-latest | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Install OpenCL dev + Intel OpenVINO 2026.4 (archive) | |
| run: | | |
| # Intel's OpenVINO APT repo only publishes up to ~2025 (the /openvino/2026 path 404s), and | |
| # 2025.x has the older ov::Allocator API that breaks ggml-openvino's template compile. So use | |
| # the ARCHIVE — exactly what upstream llama.cpp's linux-setup-openvino action does, from the | |
| # same URL template. | |
| # | |
| # KEEP IN SYNC WITH UPSTREAM. The version tracks llama.cpp's own OPENVINO_VERSION_MAJOR / | |
| # OPENVINO_VERSION_FULL (.github/workflows/release.yml at the pinned GIT_TAG); ggml-openvino | |
| # is developed against that pair, so lagging it is what eventually breaks the compile. Both | |
| # OpenVINO jobs here (Linux + Windows) use the same two values — bump them together: | |
| # major = 2026.4 full = 2026.4.0.22959.99c81491cc3 | |
| # OpenCL headers (incl. the C++ CL/cl2.hpp via opencl-clhpp-headers) come from Ubuntu's own repos. | |
| sudo apt-get update | |
| sudo apt-get install -y ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd | |
| url="https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/linux/openvino_toolkit_ubuntu24_2026.4.0.22959.99c81491cc3_x86_64.tgz" | |
| sudo mkdir -p /opt/intel/openvino | |
| curl -fSL "$url" | sudo tar -xz --strip-components=1 -C /opt/intel/openvino | |
| echo "OpenVINO_DIR=/opt/intel/openvino/runtime/cmake" >> "$GITHUB_ENV" | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| source /opt/intel/openvino/setupvars.sh || true | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh "-DGGML_OPENVINO=ON -DOpenVINO_DIR=$OpenVINO_DIR -DGGML_NATIVE=OFF -DOS_NAME=Linux -DOS_ARCH=x86_64" | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-openvino-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| build-windows-x86_64-openvino: | |
| name: Build Windows 2025 x86_64 OpenVINO (Intel) | |
| needs: [startgate, build-webui] | |
| runs-on: windows-2025-vs2026 | |
| env: | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - name: Set up MSVC developer environment (x64) | |
| uses: ilammy/msvc-dev-cmd@v1 | |
| with: | |
| arch: x64 | |
| - name: Install OpenCL headers (vcpkg) + Intel OpenVINO 2026.4 | |
| shell: pwsh | |
| # vcpkg's opencl port ships the full C++ headers incl. CL/cl2.hpp that OpenVINO's | |
| # ocl_wrapper.hpp needs (the Khronos OpenCL-Headers dropped cl2.hpp) — same as upstream | |
| # llama.cpp's windows-openvino job. OpenVINO 2026.4 matches ggml-openvino's target API. | |
| # Keep the version in sync with the Linux OpenVINO job above (and with upstream's | |
| # OPENVINO_VERSION_MAJOR / OPENVINO_VERSION_FULL) — see the note there. | |
| run: | | |
| C:\vcpkg\vcpkg install opencl:x64-windows | |
| $url = "https://storage.openvinotoolkit.org/repositories/openvino/packages/2026.4/windows/openvino_toolkit_windows_2026.4.0.22959.99c81491cc3_x86_64.zip" | |
| Invoke-WebRequest -Uri $url -OutFile "$env:RUNNER_TEMP\openvino.zip" | |
| Expand-Archive -Path "$env:RUNNER_TEMP\openvino.zip" -DestinationPath "C:\openvino" -Force | |
| # The archive extracts into a nested versioned folder; point OpenVINO_DIR at its runtime/cmake. | |
| $root = (Get-ChildItem "C:\openvino" -Directory | Select-Object -First 1).FullName | |
| "OpenVINO_DIR=$root\runtime\cmake" | Out-File -FilePath $env:GITHUB_ENV -Append | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| uses: ./.github/actions/install-sccache-windows | |
| - name: Build libraries | |
| shell: cmd | |
| # vcpkg toolchain file wires in the OpenCL (incl. cl2.hpp) that ggml-openvino needs. | |
| run: | | |
| .github\build.bat -G "Ninja Multi-Config" -DGGML_OPENVINO=ON -DOpenVINO_DIR="%OpenVINO_DIR%" -DCMAKE_TOOLCHAIN_FILE=C:\vcpkg\scripts\buildsystems\vcpkg.cmake -DOS_NAME=Windows -DOS_ARCH=x86_64 | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-openvino-windows-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| if-no-files-found: error | |
| # --------------------------------------------------------------------------- | |
| # CI-only jobs — no release artifact, purely for test coverage | |
| # --------------------------------------------------------------------------- | |
| test-cpp-linux-x86_64: | |
| name: C++ Tests Ubuntu Latest x86_64 | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Build libraries | |
| run: | | |
| mvn -q --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh -DBUILD_TESTING=ON | |
| # Every patch has a runnable guard that reds this job if it goes missing (a link error, or | |
| # test_utils.cpp / test_model_split.cpp / test_common_log_callback.cpp). This is the second | |
| # line: it asserts the applier actually ran and nothing reverted the patched tree, which no | |
| # per-patch guard covers directly. Model-free, milliseconds. | |
| - name: Verify llama.cpp patches are applied | |
| run: .github/verify-patches-applied.sh | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| build-macos-arm64-metal-15: | |
| name: Build and Test macOS 15 arm64 (Metal) | |
| needs: [startgate, build-webui] | |
| runs-on: macos-15 | |
| env: | |
| BUILD_JOBS: 2 | |
| SCCACHE_WEBDAV_TOKEN: ${{ secrets.DEPOT_TOKEN }} | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Download shared WebUI assets | |
| uses: actions/download-artifact@v8 | |
| with: | |
| name: webui-generated | |
| path: ${{ github.workspace }}/llama/webui-generated/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - name: Install sccache (shared compiler cache) | |
| if: env.USE_CACHE == 'true' && env.SCCACHE_WEBDAV_TOKEN != '' | |
| continue-on-error: true | |
| run: brew install sccache | |
| - name: Build libraries | |
| shell: bash | |
| run: | | |
| mvn --no-transfer-progress -f llama/pom.xml compile | |
| .github/build.sh -DLLAMA_METAL_EMBED_LIBRARY=ON -DGGML_NATIVE=OFF -DBUILD_TESTING=ON | |
| - name: Run C++ unit tests | |
| run: ctest --test-dir llama/build --output-on-failure | |
| - name: Upload artifacts | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: natives-metal-macos-aarch64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| # --------------------------------------------------------------------------- | |
| # Java test jobs — download release artifact, run mvn test | |
| # --------------------------------------------------------------------------- | |
| test-java-linux-x86_64: | |
| name: Java Tests Ubuntu Latest x86_64 | |
| needs: [crosscompile-linux-x86_64, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Display CPU Info | |
| shell: bash | |
| run: bash .github/print-host-info.sh | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: natives-cpu-linux-x86-64 | |
| path: ${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/ | |
| - uses: ./.github/actions/restore-models | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Memory before tests | |
| run: free -h | |
| - name: Enable core dumps | |
| run: | | |
| ulimit -c unlimited | |
| echo "${{ github.workspace }}/core.%e.%p" | sudo tee /proc/sys/kernel/core_pattern | |
| - name: Run tests | |
| run: | | |
| mvn -e --no-transfer-progress -f llama/pom.xml -P jcstress test | |
| # Green tests are not the same as tests that ran. A class-level @BeforeAll assumption that | |
| # fails makes Surefire record tests="0" for that class -- no entries at all, so it is | |
| # invisible to a skip check. That is exactly how every model-gated class stayed silently | |
| # muted on every test-java-* job for months. The floor is deliberately slack; the per-class | |
| # zero check is the sensitive half. | |
| - name: Verify tests actually ran | |
| run: .github/verify-test-counts.sh llama/target/surefire-reports --min-total 1500 | |
| - uses: actions/upload-artifact@v7 | |
| if: success() | |
| with: | |
| name: jacoco-report | |
| path: llama/target/site/jacoco/jacoco.xml | |
| if-no-files-found: ignore | |
| - name: Run PIT mutation tests | |
| run: mvn --batch-mode --no-transfer-progress -f llama/pom.xml test-compile org.pitest:pitest-maven:mutationCoverage | |
| - name: Extract PIT survivors | |
| if: always() | |
| run: | | |
| echo "=== PIT Survived Mutations ===" | |
| for html_file in $(find llama/target/pit-reports -name "*.html" -type f 2>/dev/null | sort); do | |
| if grep -q "SURVIVED" "$html_file"; then | |
| echo "Found survivors in $html_file:" | |
| grep -B 2 -A 3 "SURVIVED" "$html_file" | |
| echo "" | |
| fi | |
| done | |
| - uses: actions/upload-artifact@v7 | |
| if: always() | |
| with: { name: pit-reports, path: llama/target/pit-reports/ } | |
| - name: Memory after tests | |
| if: always() | |
| run: free -h | |
| # Echo the crash logs into the job log (workspace/policies/ci-test-diagnostics.md 3.1). | |
| - name: Print crash logs (on failure) | |
| if: failure() | |
| shell: bash | |
| run: bash .github/print-crash-logs.sh llama | |
| - if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: error-log-linux-x86_64 | |
| path: | | |
| ${{ github.workspace }}/llama/hs_err_pid*.log | |
| ${{ github.workspace }}/core.* | |
| ${{ github.workspace }}/llama/*.hprof | |
| ${{ github.workspace }}/llama/target/surefire-reports/*.dump | |
| ${{ github.workspace }}/llama/target/surefire-reports/*.dumpstream | |
| ${{ github.workspace }}/llama/target/surefire-reports/*.txt | |
| ${{ github.workspace }}/llama/target/surefire-reports/TEST-*.xml | |
| if-no-files-found: warn | |
| # --------------------------------------------------------------------------- | |
| # vmlens interleaving analysis — pure-Java, needs no native library or models. | |
| # Staged to a single smoke test for now (see the `vmlens` profile in pom.xml). | |
| # --------------------------------------------------------------------------- | |
| vmlens: | |
| name: Test (vmlens interleavings) | |
| needs: startgate | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| cache: maven | |
| - name: Test under vmlens (interleaving analysis) | |
| # Add each new test in the `vmlens` package to this -Dtest list (surefire | |
| # -Dtest matches simple class names, not package globs; the default suite is | |
| # excluded from the vmlens package via pom.xml managed surefire <excludes>). | |
| run: >- | |
| mvn --batch-mode --no-transfer-progress -f llama/pom.xml -Pvmlens test | |
| -Dtest=VmlensInterleavingSmokeTest,SessionStateInterleavingTest -DfailIfNoTests=false | |
| - uses: actions/upload-artifact@v7 | |
| if: always() | |
| with: | |
| name: vmlens-report | |
| path: llama/target/vmlens-report/ | |
| if-no-files-found: ignore | |
| # The macOS and Windows Java test jobs: one reusable workflow (.github/workflows/java-tests.yml), | |
| # called with the native build under test. The ids, names and needs stay here, so `package`, | |
| # the release gate and every reader of this file see the same jobs as before. | |
| test-java-macos-arm64-metal: | |
| name: Java Tests macOS 14 arm64 (Metal) | |
| needs: [build-macos-arm64-metal, verify-model-cache] | |
| uses: ./.github/workflows/java-tests.yml | |
| with: | |
| runs-on: macos-14 | |
| natives-artifact: macos-14-metal | |
| maven-args: -Dnet.ladenthin.llama.test.ngl=0 | |
| error-artifact: error-log-macos-14-metal | |
| test-java-macos-arm64-no-metal: | |
| name: Java Tests macOS 15 arm64 (no Metal) | |
| needs: [build-macos-arm64-no-metal, verify-model-cache] | |
| uses: ./.github/workflows/java-tests.yml | |
| with: | |
| runs-on: macos-15 | |
| natives-artifact: macos-15-no-metal | |
| error-artifact: error-log-macos-15-no-metal | |
| test-java-macos-arm64-metal-15: | |
| name: Java Tests macOS 15 arm64 (Metal) | |
| needs: [build-macos-arm64-metal-15, verify-model-cache] | |
| uses: ./.github/workflows/java-tests.yml | |
| with: | |
| runs-on: macos-15 | |
| natives-artifact: natives-metal-macos-aarch64 | |
| error-artifact: error-log-macos-15-metal | |
| test-java-windows-x86_64: | |
| name: Java Tests Windows 2025 x86_64 (default / Ninja) | |
| needs: [build-windows-x86_64, verify-model-cache] | |
| uses: ./.github/workflows/java-tests.yml | |
| with: | |
| runs-on: windows-2025-vs2026 | |
| natives-artifact: natives-cpu-windows-x86-64 | |
| error-artifact: windows-output | |
| # Java/inference validation of the MSVC-built x86_64 DLL (the analogue of | |
| # test-java-windows-x86_64 for the default Ninja build). Loads the MSVC jllama.dll | |
| # via JNI and runs the full model-backed suite, so both Windows generators are | |
| # validated end-to-end before the `msvc-windows` natives jar ships. | |
| test-java-windows-x86_64-msvc: | |
| name: Java Tests Windows 2025 x86_64 (MSVC classifier) | |
| needs: [build-windows-x86_64-msvc, verify-model-cache] | |
| uses: ./.github/workflows/java-tests.yml | |
| with: | |
| runs-on: windows-2025-vs2026 | |
| natives-artifact: natives-msvc-windows-x86-64 | |
| error-artifact: windows-output-msvc | |
| # --------------------------------------------------------------------------- | |
| # Package and publish | |
| # --------------------------------------------------------------------------- | |
| package: | |
| name: Package JARs | |
| needs: | |
| - crosscompile-linux-x86_64-cuda | |
| - crosscompile-linux-aarch64 | |
| - build-linux-s390x | |
| - build-linux-x86_64-vulkan | |
| - build-linux-aarch64-vulkan | |
| - crosscompile-android-aarch64 | |
| - crosscompile-android-x86_64 | |
| - crosscompile-android-aarch64-opencl | |
| - build-windows-x86_64 | |
| - build-windows-x86 | |
| - build-windows-arm64 | |
| - build-windows-x86_64-msvc | |
| - build-windows-x86-msvc | |
| - build-windows-x86_64-cuda | |
| - build-windows-x86_64-vulkan | |
| - build-windows-x86_64-opencl | |
| - build-linux-x86_64-rocm | |
| - build-windows-x86_64-rocm | |
| - build-linux-x86_64-sycl-fp16 | |
| - build-linux-x86_64-sycl-fp32 | |
| - build-windows-x86_64-sycl | |
| - build-windows-arm64-opencl | |
| - build-linux-x86_64-openvino | |
| - build-windows-x86_64-openvino | |
| - test-cpp-linux-x86_64 | |
| - build-macos-arm64-metal-15 | |
| - test-java-linux-x86_64 | |
| - test-java-macos-arm64-metal | |
| - test-java-macos-arm64-no-metal | |
| - test-java-macos-arm64-metal-15 | |
| - test-java-windows-x86_64 | |
| - test-java-windows-x86_64-msvc | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| # Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one | |
| # subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact | |
| # holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is | |
| # claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can | |
| # byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS | |
| # dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named | |
| # outside the glob for that reason. | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| pattern: "natives-*" | |
| path: ${{ github.workspace }}/native-artifacts/ | |
| - name: Merge native libraries into the natives tree (checked per artifact) | |
| run: | | |
| bash .github/merge-native-artifacts.sh \ | |
| "${{ github.workspace }}/native-artifacts" \ | |
| "${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/" | |
| # Runtime dependencies of every shipped native library, read from the file itself (ELF | |
| # DT_NEEDED, PE imports incl. Windows arm64, Mach-O load commands). The CPU builds must | |
| # match an exact per-<OS>/<ARCH>/<backend> allowlist -- a new dependency is a load failure | |
| # on every machine that lacks it. The GPU backends legitimately need their vendor runtime, | |
| # so there only the denylist applies. The case this was written for: ggml-rpc's RDMA | |
| # transport, which upstream enables whenever the build host has libibverbs/librdma | |
| # (llama/CMakeLists.txt forces it off). | |
| - name: Verify native runtime dependencies | |
| run: | | |
| python3 .github/verify-native-deps.py llama/src/main/natives | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Build JARs | |
| # `assembly` additionally produces the fat jar-with-dependencies uber JAR | |
| # (llama-<version>-jar-with-dependencies.jar: library classes + Java runtime deps + | |
| # default-platform native libs in one drop-on-classpath JAR, runnable via its | |
| # ServerLauncher Main-Class). It lands in target/ and is uploaded in the `llama-jars` | |
| # artifact below - never a Maven Central asset; the `package-fatjars` job downstream | |
| # combines it with the natives jars into the GitHub-Release fat-jar assets. | |
| # `natives` builds one jar per <backend>-<os>-<arch> from the tree merged above; its | |
| # enforcer rule fails the build when any of them is missing. | |
| run: mvn --batch-mode --no-transfer-progress -P release,natives,assembly -Dmaven.test.skip=true -Dgpg.skip=true package | |
| # Class-file floor, checked on every jar this job just built (the classes jar, the | |
| # natives jars and the default fat jar). Production code targets Java 8, so anything a | |
| # consumer's JVM can load must be major 52 or lower: a single Java 11 class kills | |
| # the process with UnsupportedClassVersionError before any of our code runs, which | |
| # is exactly what shipped when logback's LogbackServiceProvider was the binding. | |
| # module-info.class and META-INF/versions/** are skipped unconditionally because a | |
| # classpath JVM never loads them. Kept byte-identical across all four sibling repos. | |
| - name: Verify Java 8 bytecode (no class newer than major 52) | |
| run: .github/verify-bytecode-version.sh --max-major 52 llama/target | |
| # The jars as a consumer uses them: the classes jar, its runtime dependencies and all | |
| # natives jars at once, on the classpath and on the module path. Proves the directories | |
| # never collide and each natives jar is its own automatic module; on this GPU-less runner the | |
| # loader normally falls through the GPU backends to the CPU library. | |
| - name: Load the natives jars together (classpath and module path) | |
| run: | | |
| mvn -B -q --no-transfer-progress -f llama/pom.xml dependency:build-classpath \ | |
| -DincludeScope=runtime -Dmdep.outputFile=runtime-classpath.txt | |
| bash .github/smoke-natives-jars.sh llama/target "$(cat llama/runtime-classpath.txt)" | |
| - name: Upload JARs | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-jars | |
| path: llama/target/*.jar | |
| package-fatjars: | |
| name: Package all-backends fat jars | |
| needs: [package] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-jars | |
| path: jars/ | |
| # One self-contained multi-backend server fat jar per OS/arch | |
| # (llama-<version>-all-<os>-<arch>-jar-with-dependencies.jar): the default | |
| # jar-with-dependencies reduced to that OS/arch plus every natives jar of it; each | |
| # backend keeps its own directory, which LlamaLoader tries in its fixed priority | |
| # order (first loadable backend wins, CPU fallback). | |
| # These are GitHub-Release download assets ONLY - never deployed to Maven | |
| # Central (the deploy jobs run without the `assembly` profile and are untouched). | |
| # The script takes the natives jars from .github/natives.csv and fails loud on any | |
| # mismatch with the built jars or on a natives jar holding anything but its own | |
| # directory, so a new natives jar cannot be silently skipped. | |
| - name: Assemble all-backends fat jars | |
| run: bash .github/package-fatjars.sh jars fatjars | |
| - name: Upload fat jars | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-fatjars | |
| path: fatjars/ | |
| compression-level: 0 # jars are already deflated | |
| retention-days: 7 # multi-GB artifact; release jobs consume it within the same run | |
| if-no-files-found: error | |
| # Small single-jar artifacts, llama-fatjar-smoke-<target>, one per all-backends fat jar, so the | |
| # smoke-fatjar matrix does not download the multi-GB set. check-natives.py holds these, the | |
| # matrix rows and the targets derived from natives.csv to the same list. | |
| - name: Upload linux-x86-64 smoke jar | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-fatjar-smoke-linux-x86-64 | |
| path: fatjars/llama-*-all-linux-x86-64-jar-with-dependencies.jar | |
| compression-level: 0 | |
| retention-days: 7 | |
| if-no-files-found: error | |
| - name: Upload linux-aarch64 smoke jar | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-fatjar-smoke-linux-aarch64 | |
| path: fatjars/llama-*-all-linux-aarch64-jar-with-dependencies.jar | |
| compression-level: 0 | |
| retention-days: 7 | |
| if-no-files-found: error | |
| - name: Upload windows-x86-64 smoke jar | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-fatjar-smoke-windows-x86-64 | |
| path: fatjars/llama-*-all-windows-x86-64-jar-with-dependencies.jar | |
| compression-level: 0 | |
| retention-days: 7 | |
| if-no-files-found: error | |
| - name: Upload windows-aarch64 smoke jar | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: llama-fatjar-smoke-windows-aarch64 | |
| path: fatjars/llama-*-all-windows-aarch64-jar-with-dependencies.jar | |
| compression-level: 0 | |
| retention-days: 7 | |
| if-no-files-found: error | |
| # Every all-backends fat jar, launched on a GPU-less runner of its OS/arch: every GPU backend | |
| # must fail its load cleanly (missing vendor runtimes) and the server must come up on the CPU | |
| # backend -- backend probing, per-backend extraction and the fallback chain end-to-end through a | |
| # real `java -jar`, then /health + /v1/chat/completions. One row per fat-jar target; | |
| # check-natives.py fails when the rows differ from the targets natives.csv derives, so a new | |
| # all-<os>-<arch> jar cannot ship unlaunched -- which is how the two aarch64 jars once did: | |
| # built, signed and attached to every release while nothing ran them. That breaks the rule of | |
| # workspace/policies/fat-jar-release-assets.md ("No release asset is attached that CI has not | |
| # run"), which exists because a corrupt macOS dylib shipped in three releases under a green | |
| # pipeline. fail-fast is off: one platform's failure must not hide another's. | |
| smoke-fatjar: | |
| name: Smoke test all-backends fat jar (${{ matrix.target }}) | |
| needs: [package-fatjars, verify-model-cache] | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| include: | |
| - { target: linux-x86-64, runner: ubuntu-latest } | |
| - { target: linux-aarch64, runner: ubuntu-24.04-arm } | |
| - { target: windows-x86-64, runner: windows-2025-vs2026 } | |
| - { target: windows-aarch64, runner: windows-11-arm } | |
| runs-on: ${{ matrix.runner }} | |
| env: | |
| JAR_GLOB: llama-*-all-${{ matrix.target }}-jar-with-dependencies.jar | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| persist-credentials: false | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-fatjar-smoke-${{ matrix.target }} | |
| path: fatjars/ | |
| - uses: ./.github/actions/restore-models | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| # The Java 8 floor, re-checked on the ASSEMBLED release asset rather than the module jars the | |
| # `package` job verified: package-fatjars rewrites the zip per OS/arch (drops the other | |
| # platforms, adds the backend trees), and this is the artifact users download. The script is | |
| # bash-only, so the Linux rows check; the Windows jars are assembled from the same classes. | |
| - name: Verify Java 8 bytecode (no class newer than major 52) | |
| if: runner.os == 'Linux' | |
| run: .github/verify-bytecode-version.sh --max-major 52 fatjars | |
| - name: Run fat-jar server smoke test | |
| if: runner.os != 'Windows' | |
| run: .github/smoke-test-fatjar.sh fatjars "$JAR_GLOB" "models/${DRAFT_MODEL_NAME}" | |
| - name: Run fat-jar server smoke test | |
| if: runner.os == 'Windows' | |
| shell: pwsh | |
| run: .github/smoke-test-fatjar.ps1 -JarDir fatjars -JarGlob $env:JAR_GLOB -Model "models/$env:DRAFT_MODEL_NAME" | |
| # RPC over two JVMs from the same release asset: RpcServer in one, the default NativeServer | |
| # with --rpc in the other. Requires a chat completion, the model buffer on the RPC endpoint in | |
| # the load log (the layers really went over RPC), an accepted client on the server, and a clean | |
| # non-SIGABRT failure naming the endpoint when --rpc names a server nobody runs. | |
| - name: Run fat-jar RPC smoke test (two JVMs) | |
| if: matrix.target == 'linux-x86-64' | |
| run: .github/smoke-rpc-fatjar.sh fatjars "$JAR_GLOB" "models/${DRAFT_MODEL_NAME}" | |
| - name: Upload server logs | |
| if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: fatjar-smoke-${{ matrix.target }}-logs | |
| path: | | |
| server-out.log | |
| server-err.log | |
| rpc-server.log | |
| rpc-client-out.log | |
| rpc-client-err.log | |
| rpc-unreachable.log | |
| if-no-files-found: warn | |
| # The agent release asset, launched the way the README tells a user to: `java -jar` on the agent | |
| # jar lying next to the real all-backends Linux fat jar, so the core is found only through the | |
| # agent manifest's Class-Path. Proves the two assets fit together (matching version in the file | |
| # names, nothing missing on either side), that the agent jar carries no core, and that a | |
| # one-shot answer and a read_file tool round work through the shipped jars. | |
| smoke-agent-linux: | |
| name: Smoke test the agent release jar (Linux) | |
| needs: [test-java-llama-atmosphere-agent, package-fatjars, verify-model-cache] | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-atmosphere-agent-jar | |
| path: agent-assets/ | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-fatjar-smoke-linux-x86-64 | |
| path: agent-assets/ | |
| - uses: ./.github/actions/restore-models | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| # The agent is Java 21 (Atmosphere's floor), unlike the Java 8 core: its ceiling is 65. | |
| # One waiver: JLine ships its FFM terminal provider (org/jline/terminal/impl/ffm, 25 classes) | |
| # as Java 22 bytecode. It is discovered through META-INF/jline/providers/ffm, not loaded | |
| # eagerly: on Java 21 JLine picks its JNI provider, and a forced FFM provider falls back to a | |
| # dumb terminal instead of failing (verified on 21.0.10). Only that package is waived, so | |
| # any other class above 65 still fails the gate. | |
| - name: Verify Java 21 bytecode (no class newer than major 65) | |
| run: > | |
| .github/verify-bytecode-version.sh --max-major 65 | |
| --allow 'llama-atmosphere-agent-*:org/jline/terminal/impl/ffm/*' | |
| agent-assets/llama-atmosphere-agent-*-jar-with-dependencies.jar | |
| - name: Run the agent release-jar smoke test | |
| run: .github/smoke-agent-jar.sh agent-assets "models/${TOOL_MODEL_NAME}" | |
| - name: Upload agent logs | |
| if: failure() | |
| uses: actions/upload-artifact@v7 | |
| with: | |
| name: agent-smoke-linux-logs | |
| path: agent-*.log | |
| if-no-files-found: warn | |
| # --------------------------------------------------------------------------- | |
| # macOS member of the cross-repo "no release asset is attached that CI has not run" convention | |
| # (workspace/policies/fat-jar-release-assets.md; BitcoinAddressFinder and srcmorph run the shared | |
| # .github/smoke-fatjar-cli.sh in the same job shape). It closes the gap that let a corrupt dylib | |
| # ship: the three macOS Java test jobs each test the dylib THEIR OWN build job produced, so until | |
| # now nothing ever loaded the one that goes into the published jar — see CLAUDE.md, "macOS arm64: | |
| # three build jobs, one shipped dylib". | |
| # | |
| # macOS-specific in two ways. It targets the DEFAULT fat jar from `llama-jars`, because there is | |
| # no `all-macos-*` fat jar to target: macOS has no GPU classifier (Metal ships in the default | |
| # jar), so package-fatjars builds no macOS variant. And it asserts native loadability rather than | |
| # a CLI exit code — this jar's Main-Class is a server that never returns, and `codesign --strict` | |
| # is what actually detects a dylib assembled from two builds. Deliberately model-free (no GGUF, | |
| # no cache restore, ~1 min): a full model-backed macOS server smoke would be strictly more, but | |
| # this catches the failure class that shipped and is cheap enough to always run. | |
| # --------------------------------------------------------------------------- | |
| smoke-fatjar-macos: | |
| name: Smoke test packaged natives (macOS) | |
| needs: [package] | |
| runs-on: macos-15 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-jars | |
| path: jars/ | |
| - uses: actions/setup-java@v6 | |
| with: | |
| distribution: 'temurin' | |
| java-version-file: .java-version | |
| - name: Run packaged-native smoke test | |
| run: .github/smoke-native-macos.sh jars 'llama-*-jar-with-dependencies.jar' | |
| report: | |
| name: Report | |
| needs: [package] | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: write | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-java@v6 | |
| with: { java-version-file: .java-version, distribution: temurin } | |
| - uses: actions/download-artifact@v8 | |
| with: { name: jacoco-report, path: target/site/jacoco/ } | |
| continue-on-error: true | |
| # Submits the dependency graph to GitHub. Informational: it says nothing about whether the | |
| # artifacts are correct, but it sits in the `report` job, which the release path needs -- so | |
| # without this flag a third-party action having a bad day can block a publish. | |
| - uses: advanced-security/maven-dependency-submission-action@v6 | |
| continue-on-error: true | |
| - name: Coveralls | |
| uses: coverallsapp/github-action@v2 | |
| with: | |
| github-token: ${{ secrets.GITHUB_TOKEN }} | |
| file: target/site/jacoco/jacoco.xml | |
| format: jacoco | |
| continue-on-error: true | |
| - name: Codecov | |
| uses: codecov/codecov-action@v7 | |
| with: | |
| token: ${{ secrets.CODECOV_TOKEN }} | |
| files: target/site/jacoco/jacoco.xml | |
| continue-on-error: true | |
| check-snapshot: | |
| name: "Check: main branch / SNAPSHOT" | |
| needs: [report] | |
| runs-on: ubuntu-latest | |
| if: >- | |
| (github.event_name == 'push' && github.ref == 'refs/heads/main') || | |
| (github.event_name == 'workflow_dispatch' && !startsWith(github.ref, 'refs/tags/v')) | |
| steps: | |
| - name: Confirm snapshot ref | |
| run: echo "Confirmed on snapshot ref ${{ github.ref }}" | |
| check-tag: | |
| name: "Check: v* tag" | |
| needs: [report] | |
| runs-on: ubuntu-latest | |
| if: startsWith(github.ref, 'refs/tags/v') | |
| steps: | |
| - name: Confirm tag ref | |
| run: echo "Confirmed on tag ${{ github.ref }}" | |
| publish-snapshot: | |
| name: Publish Snapshot to Central | |
| needs: [check-snapshot, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar, smoke-fatjar-macos, smoke-agent-linux, vmlens, shared-files] | |
| if: needs.check-snapshot.result == 'success' && inputs.publish_to_central | |
| runs-on: ubuntu-latest | |
| environment: maven-central | |
| permissions: | |
| contents: write | |
| steps: | |
| - uses: actions/checkout@v7 | |
| # Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one | |
| # subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact | |
| # holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is | |
| # claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can | |
| # byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS | |
| # dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named | |
| # outside the glob for that reason. | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| pattern: "natives-*" | |
| path: ${{ github.workspace }}/native-artifacts/ | |
| - name: Merge native libraries into the natives tree (checked per artifact) | |
| run: | | |
| bash .github/merge-native-artifacts.sh \ | |
| "${{ github.workspace }}/native-artifacts" \ | |
| "${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/" | |
| - name: Set up Maven Central Repository | |
| uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: 'temurin' | |
| server-id: central | |
| server-username: MAVEN_USERNAME | |
| server-password: MAVEN_PASSWORD | |
| gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }} | |
| gpg-passphrase: MAVEN_GPG_PASSPHRASE | |
| - name: Guard - require a -SNAPSHOT version | |
| shell: bash | |
| run: | | |
| VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1) | |
| echo "Resolved project version: $VERSION" | |
| case "$VERSION" in | |
| *-SNAPSHOT) echo "OK: -SNAPSHOT version, continuing snapshot deploy." ;; | |
| *) echo "::error::Refusing to publish non-SNAPSHOT version '$VERSION' from the snapshot job. Snapshot publishing requires a -SNAPSHOT version; releases go through the v* tag path."; exit 1 ;; | |
| esac | |
| # One reactor deploy publishes all five Maven artifacts at the same version: | |
| # net.ladenthin:llama-parent (the pom), :llama (the classes jar + natives jars), | |
| # :llama-langchain4j, :llama-kotlin and :llama-platform (a pom). The `release` profile (GPG + Central | |
| # Publishing) is inherited from the parent, so every module — including the | |
| # parent pom — is signed. The Android AARs are published by the separate | |
| # Gradle step below (Maven cannot deploy <packaging>aar</packaging>). | |
| # Informational only (nothing depends on it): logs the effective POM with the same | |
| # profile set as the deploy below, so the resolved central-publishing configuration | |
| # (waitUntil/waitMaxTime etc.) is visible for debugging. | |
| - name: Show effective POM (debug) | |
| run: mvn --batch-mode --no-transfer-progress -P release,natives help:effective-pom | |
| - name: Publish snapshot (reactor - parent + llama + llama-langchain4j + llama-kotlin + llama-platform) | |
| run: mvn --batch-mode --no-transfer-progress -P release,natives -Dmaven.test.skip=true deploy | |
| env: | |
| MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| # The agent's thin jar + pom (not a reactor module, so a deploy of its own). It resolves the | |
| # core and the natives jars llama-platform names from the local repository, where the reactor | |
| # deploy above has just installed them; check-natives.py holds its version to the reactor's. | |
| - name: Publish snapshot (llama-atmosphere-agent) | |
| run: mvn --batch-mode --no-transfer-progress -f llama-atmosphere-agent/pom.xml -P release -Dmaven.test.skip=true deploy | |
| env: | |
| MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| # Android AARs (llama-android, llama-android-opencl): assembled and published by | |
| # the plain-Gradle build in llama-android/ — Maven cannot deploy | |
| # <packaging>aar</packaging>. Natives are already on disk from the artifact | |
| # downloads above; the core jar was just built by the reactor deploy. | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Publish Android AAR snapshots (llama-android + llama-android-opencl) | |
| run: | | |
| mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/ | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/ | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/ | |
| gradle -p llama-android publishAllPublicationsToCentralSnapshotsRepository | |
| env: | |
| CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }} | |
| # Runs even when the deploy step failed: a Central publish-poll timeout reds the | |
| # job *after* the bundle was uploaded (and typically published server-side), while | |
| # the signed jars + .asc files already exist in target/ (signing happens at | |
| # verify). Collecting on failure lets the github-snapshot job still attach them. | |
| - name: Collect signed artifacts | |
| if: ${{ !cancelled() }} | |
| run: | | |
| mkdir -p signed-snapshot-assets | |
| cp llama/target/*.jar signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama-langchain4j/target/*.jar signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama-langchain4j/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama-kotlin/target/*.jar signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama-kotlin/target/*.jar.asc signed-snapshot-assets/ 2>/dev/null || true | |
| cp llama-android/build/aar/*.aar signed-snapshot-assets/ 2>/dev/null || true | |
| - uses: actions/upload-artifact@v7 | |
| if: ${{ !cancelled() }} | |
| with: | |
| name: signed-snapshot-assets | |
| path: signed-snapshot-assets/ | |
| github-snapshot: | |
| name: Update Snapshot Pre-release on GitHub | |
| needs: [publish-snapshot, package-fatjars, test-java-llama-atmosphere-agent] | |
| # Also runs when publish-snapshot FAILED (not when skipped/cancelled): a Central | |
| # publish-poll timeout reds that job after the artifacts were already uploaded — | |
| # the GitHub pre-release assets must not be lost in that case. | |
| if: ${{ !cancelled() && (needs.publish-snapshot.result == 'success' || needs.publish-snapshot.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }} | |
| runs-on: ubuntu-latest | |
| # maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is | |
| # scoped to this environment) for signing the fat jars below. This environment has | |
| # no approval gate (the standalone verify-signing-key jobs use it on every run), so | |
| # declaring it here does not block the release. | |
| environment: maven-central | |
| permissions: | |
| contents: write | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: signed-snapshot-assets | |
| path: snapshot-assets/ | |
| # All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub | |
| # download assets only, deliberately NOT deployed to Maven Central. Downloaded | |
| # into the same directory so the one upload glob below picks everything up | |
| # (fat-jar names are disjoint from the signed thin-jar names). | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-fatjars | |
| path: snapshot-assets/ | |
| # The agent jar (+ sha256) — built without the core, run next to one of the fat jars above. | |
| # Same directory, so sign-fatjars.sh signs it and the upload glob attaches it. | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-atmosphere-agent-jar | |
| path: snapshot-assets/ | |
| # GPG-sign the fat jars so each carries a detached .asc signature alongside its | |
| # .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at | |
| # deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity) | |
| # are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because | |
| # only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md. | |
| - name: GPG-sign the fat jars (.asc) | |
| env: | |
| GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| run: bash .github/sign-fatjars.sh snapshot-assets | |
| - name: Report unsigned assets (does not block the upload) | |
| # Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even | |
| # when their publish job failed, because a Central publish-poll timeout must not cost the | |
| # GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at | |
| # all. Refusing to attach on a signing failure would defeat exactly that. So annotate here, | |
| # upload regardless, and fail the job afterwards. Assets always land; an unsigned release is | |
| # still loudly red rather than quietly wrong. | |
| # See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red". | |
| id: signatures | |
| run: | | |
| set -uo pipefail | |
| dir="snapshot-assets" | |
| jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort) | |
| if [ -z "$jars" ]; then | |
| echo "::error::no jars in $dir -- the collection step produced nothing" | |
| echo "missing=-1" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| missing=0 | |
| for jar in $jars; do | |
| if [ ! -e "$jar.asc" ]; then | |
| echo "::error::unsigned: $(basename "$jar") has no detached .asc" | |
| missing=$((missing + 1)) | |
| fi | |
| done | |
| echo "missing=$missing" >> "$GITHUB_OUTPUT" | |
| [ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed" | |
| exit 0 | |
| - name: Update snapshot pre-release | |
| env: | |
| GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} | |
| run: | | |
| gh release view snapshot --repo ${{ github.repository }} 2>/dev/null \ | |
| || gh release create snapshot \ | |
| --repo ${{ github.repository }} \ | |
| --prerelease \ | |
| --title "Snapshot (latest)" \ | |
| --notes "Latest snapshot build from the main branch." | |
| gh release upload snapshot snapshot-assets/* \ | |
| --repo ${{ github.repository }} \ | |
| --clobber | |
| - name: Fail if anything was attached unsigned | |
| # After the upload on purpose: the assets must exist even when the signature does not. | |
| if: ${{ always() && steps.signatures.outputs.missing != '0' }} | |
| run: | | |
| echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)" | |
| exit 1 | |
| publish-release: | |
| name: Publish Release to Central | |
| if: needs.check-tag.result == 'success' && inputs.publish_to_central | |
| needs: [check-tag, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar, smoke-fatjar-macos, smoke-agent-linux, vmlens, shared-files] | |
| runs-on: ubuntu-latest | |
| environment: maven-central | |
| permissions: | |
| contents: write | |
| steps: | |
| - uses: actions/checkout@v7 | |
| # Every shipped native build uploads `natives-<classifier>`. Downloaded UNMERGED (one | |
| # subdirectory per artifact) on purpose: merge-native-artifacts.sh checks that each artifact | |
| # holds exactly the <OS>/<ARCH>/<backend>/ directory its name promises and that no path is | |
| # claimed twice, then merges. `merge-multiple: true` would silently overwrite -- and can | |
| # byte-level interleave -- two artifacts with the same path, which is how a corrupt macOS | |
| # dylib shipped; the test-only macOS builds (macos-14-metal, macos-15-no-metal) are named | |
| # outside the glob for that reason. | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| pattern: "natives-*" | |
| path: ${{ github.workspace }}/native-artifacts/ | |
| - name: Merge native libraries into the natives tree (checked per artifact) | |
| run: | | |
| bash .github/merge-native-artifacts.sh \ | |
| "${{ github.workspace }}/native-artifacts" \ | |
| "${{ github.workspace }}/llama/src/main/natives/net/ladenthin/llama/" | |
| - name: Set up Maven Central Repository | |
| uses: actions/setup-java@v6 | |
| with: | |
| java-version-file: .java-version | |
| distribution: 'temurin' | |
| server-id: central | |
| server-username: MAVEN_USERNAME | |
| server-password: MAVEN_PASSWORD | |
| gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }} | |
| gpg-passphrase: MAVEN_GPG_PASSPHRASE | |
| # One reactor deploy publishes all five Maven artifacts at the same version: | |
| # net.ladenthin:llama-parent (the pom), :llama (the classes jar + natives jars), | |
| # :llama-langchain4j, :llama-kotlin and :llama-platform (a pom). The `release` profile (GPG + Central Publishing) is inherited | |
| # from the parent, so every module — including the parent pom — is signed. | |
| # Informational only (nothing depends on it): logs the effective POM with the same | |
| # profile set as the deploy below, so the resolved central-publishing configuration | |
| # (waitUntil/waitMaxTime etc.) is visible for debugging. | |
| - name: Show effective POM (debug) | |
| run: mvn --batch-mode --no-transfer-progress -P release,natives help:effective-pom | |
| - name: Publish release (reactor - parent + llama + llama-langchain4j + llama-kotlin + llama-platform) | |
| run: mvn --batch-mode --no-transfer-progress -P release,natives -Dmaven.test.skip=true deploy | |
| env: | |
| MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| # The agent's thin jar + pom (not a reactor module, so a deploy of its own). It resolves the | |
| # core and the natives jars llama-platform names from the local repository, where the reactor | |
| # deploy above has just installed them; check-natives.py holds its version to the reactor's. | |
| - name: Publish release (llama-atmosphere-agent) | |
| run: mvn --batch-mode --no-transfer-progress -f llama-atmosphere-agent/pom.xml -P release -Dmaven.test.skip=true deploy | |
| env: | |
| MAVEN_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| MAVEN_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| # Android AARs (llama-android, llama-android-opencl): Maven cannot deploy | |
| # <packaging>aar</packaging>, so the plain-Gradle build signs + publishes them | |
| # into a local Maven-layout staging repo, which is zipped into a Central | |
| # Portal bundle and uploaded via the Publisher API (publishingType=AUTOMATIC: | |
| # the deployment publishes as soon as portal validation passes). Natives are | |
| # already on disk from the artifact downloads above; the core jar was just | |
| # built by the reactor deploy. | |
| - uses: gradle/actions/setup-gradle@v6 | |
| with: | |
| gradle-version: ${{ env.GRADLE_VERSION }} | |
| - name: Publish Android AAR release bundle to Central Portal | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| mkdir -p llama-android/natives/cpu/arm64-v8a llama-android/natives/cpu/x86_64 llama-android/natives/opencl/arm64-v8a | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/cpu/libjllama.so llama-android/natives/cpu/arm64-v8a/ | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/x86_64/cpu/libjllama.so llama-android/natives/cpu/x86_64/ | |
| cp llama/src/main/natives/net/ladenthin/llama/Linux-Android/aarch64/opencl/libjllama.so llama-android/natives/opencl/arm64-v8a/ | |
| gradle -p llama-android publishAllPublicationsToStagingRepository | |
| VERSION=$(mvn -q -DforceStdout help:evaluate -Dexpression=project.version | tail -n1) | |
| ( cd llama-android/build/staging-repo && zip -r -q ../central-bundle.zip net -x "*maven-metadata*" ) | |
| TOKEN=$(printf "%s:%s" "$CENTRAL_USERNAME" "$CENTRAL_PASSWORD" | base64 -w0) | |
| curl --fail-with-body -X POST \ | |
| -H "Authorization: Bearer $TOKEN" \ | |
| -F "bundle=@llama-android/build/central-bundle.zip" \ | |
| "https://central.sonatype.com/api/v1/publisher/upload?publishingType=AUTOMATIC&name=llama-android-$VERSION" | |
| env: | |
| CENTRAL_USERNAME: ${{ secrets.CENTRAL_USERNAME }} | |
| CENTRAL_PASSWORD: ${{ secrets.CENTRAL_TOKEN }} | |
| MAVEN_GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| MAVEN_GPG_KEY_ID: ${{ secrets.GPG_KEY_ID }} | |
| # Runs even when the deploy step failed: a Central publish-poll timeout reds the | |
| # job *after* the bundle was uploaded (and typically published server-side), while | |
| # the signed jars + .asc files already exist in target/ (signing happens at | |
| # verify). Collecting on failure lets the github-release-signed job still attach them. | |
| - name: Collect signed artifacts | |
| if: ${{ !cancelled() }} | |
| run: | | |
| mkdir -p signed-release-assets | |
| cp llama/target/*.jar signed-release-assets/ 2>/dev/null || true | |
| cp llama/target/*.jar.asc signed-release-assets/ 2>/dev/null || true | |
| cp llama-langchain4j/target/*.jar signed-release-assets/ 2>/dev/null || true | |
| cp llama-langchain4j/target/*.jar.asc signed-release-assets/ 2>/dev/null || true | |
| cp llama-kotlin/target/*.jar signed-release-assets/ 2>/dev/null || true | |
| cp llama-kotlin/target/*.jar.asc signed-release-assets/ 2>/dev/null || true | |
| cp llama-android/build/aar/*.aar signed-release-assets/ 2>/dev/null || true | |
| - uses: actions/upload-artifact@v7 | |
| if: ${{ !cancelled() }} | |
| with: | |
| name: signed-release-assets | |
| path: signed-release-assets/ | |
| github-release-signed: | |
| name: Attach Signed Binaries to GitHub Release | |
| needs: [publish-release, package-fatjars, test-java-llama-atmosphere-agent] | |
| # Also runs when publish-release FAILED (not when skipped/cancelled): a Central | |
| # publish-poll timeout reds that job after the artifacts were already uploaded — | |
| # the GitHub release assets must not be lost in that case. | |
| if: ${{ !cancelled() && (needs.publish-release.result == 'success' || needs.publish-release.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }} | |
| runs-on: ubuntu-latest | |
| # maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is | |
| # scoped to this environment) for signing the fat jars below. This environment has | |
| # no approval gate (the standalone verify-signing-key jobs use it on every run), so | |
| # declaring it here does not block the release. | |
| environment: maven-central | |
| permissions: | |
| contents: write | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: signed-release-assets | |
| path: release-assets/ | |
| # All-backends server fat jars (+ default CPU fat jar + sha256 files) — GitHub | |
| # download assets only, deliberately NOT deployed to Maven Central. Downloaded | |
| # into the same directory so the one upload glob below picks everything up | |
| # (fat-jar names are disjoint from the signed thin-jar names). | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-fatjars | |
| path: release-assets/ | |
| # The agent jar (+ sha256) — built without the core, run next to one of the fat jars above. | |
| # Same directory, so sign-fatjars.sh signs it and the upload glob attaches it. | |
| - uses: actions/download-artifact@v8 | |
| with: | |
| name: llama-atmosphere-agent-jar | |
| path: release-assets/ | |
| # GPG-sign the fat jars so each carries a detached .asc signature alongside its | |
| # .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at | |
| # deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity) | |
| # are kept; the .asc adds authenticity. Signed here (not in package-fatjars) because | |
| # only this dispatch-gated path has the key. See workspace/policies/fat-jar-release-assets.md. | |
| - name: GPG-sign the fat jars (.asc) | |
| env: | |
| GPG_PRIVATE_KEY: ${{ secrets.GPG_PRIVATE_KEY }} | |
| GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} | |
| run: bash .github/sign-fatjars.sh release-assets | |
| - name: Report unsigned assets (does not block the upload) | |
| # Deliberately NON-blocking, and deliberately BEFORE the upload. Both attach jobs run even | |
| # when their publish job failed, because a Central publish-poll timeout must not cost the | |
| # GitHub assets: if Central is unreachable these are the ONLY way to get the artifacts at | |
| # all. Refusing to attach on a signing failure would defeat exactly that. So annotate here, | |
| # upload regardless, and fail the job afterwards. Assets always land; an unsigned release is | |
| # still loudly red rather than quietly wrong. | |
| # See workspace/policies/fat-jar-release-assets.md, "Attach first, then go red". | |
| id: signatures | |
| run: | | |
| set -uo pipefail | |
| dir="release-assets" | |
| jars=$(find "$dir" -maxdepth 1 -name '*.jar' | sort) | |
| if [ -z "$jars" ]; then | |
| echo "::error::no jars in $dir -- the collection step produced nothing" | |
| echo "missing=-1" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| missing=0 | |
| for jar in $jars; do | |
| if [ ! -e "$jar.asc" ]; then | |
| echo "::error::unsigned: $(basename "$jar") has no detached .asc" | |
| missing=$((missing + 1)) | |
| fi | |
| done | |
| echo "missing=$missing" >> "$GITHUB_OUTPUT" | |
| [ "$missing" -eq 0 ] && echo "all $(echo "$jars" | wc -l) jar(s) signed" | |
| exit 0 | |
| - name: Upload release assets | |
| uses: softprops/action-gh-release@v3 | |
| with: | |
| files: release-assets/* | |
| - name: Fail if anything was attached unsigned | |
| # After the upload on purpose: the assets must exist even when the signature does not. | |
| if: ${{ always() && steps.signatures.outputs.missing != '0' }} | |
| run: | | |
| echo "::error::${{ steps.signatures.outputs.missing }} asset(s) attached without a signature (-1 means none were collected at all)" | |
| exit 1 |