Skip to content

Commit 80ddeea

Browse files
Merge pull request #461 from bernardladenthin/claude/busy-archimedes-lo9uyy
feat: llama.cpp RPC — --rpc client and an in-JVM RpcServer
2 parents f266ad4 + 9c6caef commit 80ddeea

33 files changed

Lines changed: 3905 additions & 23 deletions

‎.github/smoke-rpc-fatjar.sh‎

Lines changed: 119 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,119 @@
1+
#!/usr/bin/env bash
2+
3+
# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
4+
#
5+
# SPDX-License-Identifier: MIT
6+
7+
# RPC smoke test over two JVMs, on the real release asset:
8+
#
9+
# JVM A java -cp <fatjar> net.ladenthin.llama.RpcServer serves this runner's devices
10+
# JVM B java -jar <fatjar> -m <model> --rpc 127.0.0.1:<A> the default NativeServer, which
11+
# offloads its layers to A
12+
#
13+
# and checks that B answers a chat completion, that its load log shows a model buffer on A's
14+
# endpoint (the layers really went over RPC, not silently to the CPU), and that A accepted a
15+
# client. A third launch names a server nobody runs and must fail with a message naming it and a
16+
# normal exit -- not a SIGABRT, which is what ggml-rpc did before patches/0015.
17+
#
18+
# Usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>
19+
# Output lands in rpc-server.log, rpc-client-out.log, rpc-client-err.log, rpc-unreachable.log
20+
# (uploaded by the CI job on failure).
21+
set -euo pipefail
22+
23+
JAR_DIR="${1:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
24+
JAR_GLOB="${2:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
25+
MODEL="${3:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
26+
RPC_PORT="${RPC_PORT:-50152}"
27+
HTTP_PORT="${HTTP_PORT:-18181}"
28+
UNUSED_PORT="${UNUSED_PORT:-50153}"
29+
30+
fail() {
31+
echo "::error::$*" >&2
32+
for f in rpc-server.log rpc-client-out.log rpc-client-err.log rpc-unreachable.log; do
33+
[ -f "$f" ] && { echo "--- $f (tail) ---"; tail -60 "$f"; }
34+
done
35+
exit 1
36+
}
37+
38+
mapfile -t JARS < <(find "$JAR_DIR" -maxdepth 1 -name "$JAR_GLOB" | sort)
39+
[ "${#JARS[@]}" -eq 1 ] || fail "expected exactly 1 jar matching $JAR_GLOB in $JAR_DIR, got ${#JARS[@]}: ${JARS[*]:-none}"
40+
JAR="${JARS[0]}"
41+
[ -f "$MODEL" ] || fail "model file missing: $MODEL"
42+
43+
PIDS=()
44+
cleanup() {
45+
for pid in "${PIDS[@]}"; do
46+
kill "$pid" 2> /dev/null || true
47+
done
48+
}
49+
trap cleanup EXIT
50+
51+
# --- JVM A: the RPC server ----------------------------------------------------------------------
52+
java -cp "$JAR" net.ladenthin.llama.RpcServer --port "$RPC_PORT" --threads 2 --device CPU > rpc-server.log 2>&1 &
53+
PIDS+=($!)
54+
SERVER_PID=$!
55+
# Up to 300 s, like the NativeServer smoke: an all-backends jar extracts every GPU backend's
56+
# library (CUDA, ROCm and SYCL are hundreds of MB) and fails to load each before it reaches one
57+
# this GPU-less runner can load. 60 s was too short for that on the first CI run.
58+
for _ in $(seq 1 100); do
59+
kill -0 "$SERVER_PID" 2> /dev/null || fail "RpcServer exited before listening"
60+
grep -q "RpcServer listening on 127.0.0.1:$RPC_PORT" rpc-server.log && break
61+
sleep 3
62+
done
63+
grep -q "RpcServer listening on 127.0.0.1:$RPC_PORT" rpc-server.log || fail "RpcServer never reported listening"
64+
grep -q "serving \[CPU\]" rpc-server.log || fail "RpcServer --device CPU did not serve exactly the CPU"
65+
# The backend is chosen once. RpcServer is the one entry point that reaches the loader before
66+
# LlamaModel, and JNI_OnLoad initializes LlamaModel, whose static block re-entered the loader and
67+
# ran a second complete load over the library being loaded (LlamaLoader.runOnceOnThisThread).
68+
# A manifest-less jar prints the line zero times.
69+
[ "$(grep -c '\[jllama\] using native backend' rpc-server.log)" -le 1 ] \
70+
|| fail "the native library was loaded more than once: $(grep -c '\[jllama\] using native backend' rpc-server.log) backend selections"
71+
echo "RPC server up: $(grep 'RpcServer listening' rpc-server.log)"
72+
73+
# --- JVM B: the model, offloaded over RPC --------------------------------------------------------
74+
java -jar "$JAR" -m "$MODEL" --host 127.0.0.1 --port "$HTTP_PORT" --chat-template chatml \
75+
--rpc "127.0.0.1:$RPC_PORT" -ngl 99 -lv 4 > rpc-client-out.log 2> rpc-client-err.log &
76+
PIDS+=($!)
77+
CLIENT_PID=$!
78+
CODE=""
79+
for _ in $(seq 1 100); do
80+
kill -0 "$CLIENT_PID" 2> /dev/null || fail "the RPC client server exited before becoming healthy"
81+
CODE="$(curl -s -o /dev/null -w '%{http_code}' "http://127.0.0.1:$HTTP_PORT/health" || true)"
82+
[ "$CODE" = "200" ] && break
83+
sleep 3
84+
done
85+
[ "$CODE" = "200" ] || fail "/health never returned 200 (last code: ${CODE:-none})"
86+
87+
RESPONSE="$(curl -sS --fail -X POST "http://127.0.0.1:$HTTP_PORT/v1/chat/completions" \
88+
-H 'Content-Type: application/json' \
89+
-d '{"messages":[{"role":"user","content":"Say hello."}],"max_tokens":8,"temperature":0}')" \
90+
|| fail "chat completion over RPC failed"
91+
echo "$RESPONSE" | python3 -c '
92+
import json, sys
93+
message = json.load(sys.stdin)["choices"][0]["message"]
94+
assert message is not None, "choices[0].message missing"
95+
print("chat completion over RPC OK:", json.dumps(message)[:200])
96+
' || fail "malformed chat completion response: $RESPONSE"
97+
98+
grep -h "model buffer size" rpc-client-out.log rpc-client-err.log | grep -q "127.0.0.1:$RPC_PORT" \
99+
|| fail "no model buffer on the RPC server in the load log -- the layers did not go over RPC"
100+
grep -q "Accepted client connection" rpc-server.log || fail "the RPC server never accepted a client"
101+
echo "layers offloaded over RPC: $(grep -h 'model buffer size' rpc-client-out.log rpc-client-err.log | grep "127.0.0.1:$RPC_PORT" | head -1)"
102+
103+
kill "$CLIENT_PID" 2> /dev/null || true
104+
wait "$CLIENT_PID" 2> /dev/null || true
105+
106+
# --- an unreachable server fails the start cleanly -----------------------------------------------
107+
set +e
108+
timeout 120 java -jar "$JAR" -m "$MODEL" --host 127.0.0.1 --port "$((HTTP_PORT + 1))" \
109+
--rpc "127.0.0.1:$UNUSED_PORT" > rpc-unreachable.log 2>&1
110+
status=$?
111+
set -e
112+
[ "$status" -ne 0 ] || fail "a server naming an unreachable RPC endpoint started anyway"
113+
[ "$status" -ne 124 ] || fail "a server naming an unreachable RPC endpoint hung instead of failing"
114+
# 134 = SIGABRT: the GGML_ABORT patches/0015 removed from the registration path
115+
[ "$status" -ne 134 ] || fail "an unreachable RPC endpoint aborted the JVM (exit 134)"
116+
grep -q "127.0.0.1:$UNUSED_PORT" rpc-unreachable.log || fail "the failure does not name the unreachable endpoint"
117+
echo "unreachable endpoint rejected (exit $status): $(grep -m1 "127.0.0.1:$UNUSED_PORT" rpc-unreachable.log)"
118+
119+
echo "RPC smoke test PASSED"

‎.github/verify-native-deps.py‎

Lines changed: 211 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,211 @@
1+
#!/usr/bin/env python3
2+
# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
3+
#
4+
# SPDX-License-Identifier: MIT
5+
"""Fail when a shipped native library needs a runtime library it did not need before.
6+
7+
Every jllama library is ONE file with llama.cpp and ggml linked in statically, so its dynamic
8+
dependencies are exactly what a consumer's machine must provide. A new one is a silent break on
9+
every machine that lacks it -- the case this guards against is ggml-rpc's RDMA transport, which
10+
upstream switches on whenever the build host has libibverbs/librdma and which would make the
11+
library unloadable without rdma-core. It reads the dependency list straight from the file (ELF
12+
DT_NEEDED, PE import table, Mach-O LC_LOAD_DYLIB) with the standard library only, so it runs on
13+
any runner and checks every architecture, including the ones binutils cannot read (Windows arm64,
14+
Mach-O).
15+
16+
Usage:
17+
verify-native-deps.py --default <resources-root> # exact allowlist per <OS>/<ARCH>
18+
verify-native-deps.py --deny <dir>... # classifier trees: only the denylist
19+
20+
--default checks net/ladenthin/llama/<OS>/<ARCH>/ under the root against ALLOWED below: a
21+
dependency outside the list fails, and so does an <OS>/<ARCH> without a list (a new platform
22+
must be listed consciously). --deny scans every native library under the directories for DENIED
23+
names only, because GPU classifiers legitimately need their vendor runtime.
24+
25+
Exit codes: 0 clean, 1 violation, 2 nothing found to check.
26+
"""
27+
28+
import os
29+
import struct
30+
import sys
31+
32+
# What each default-JAR library needed when this check was introduced (5.1.0 plus the RPC backend,
33+
# which adds nothing: its sockets are libc/libSystem/WS2_32, all already present).
34+
ALLOWED = {
35+
"Linux/x86_64": {"libdl.so.2", "libgomp.so.1", "libpthread.so.0", "librt.so.1", "libstdc++.so.6",
36+
"libm.so.6", "libgcc_s.so.1", "libc.so.6", "ld-linux-x86-64.so.2"},
37+
"Linux/aarch64": {"libgomp.so.1", "libstdc++.so.6", "libm.so.6", "libgcc_s.so.1", "libc.so.6",
38+
"ld-linux-aarch64.so.1"},
39+
"Linux/s390x": {"libstdc++.so.6", "libm.so.6", "libgcc_s.so.1", "libc.so.6", "ld64.so.1"},
40+
"Linux-Android/aarch64": {"liblog.so", "libm.so", "libdl.so", "libc.so", "libandroid.so"},
41+
"Linux-Android/x86_64": {"liblog.so", "libm.so", "libdl.so", "libc.so", "libandroid.so"},
42+
"Windows/x86_64": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll", "vcomp140.dll"},
43+
"Windows/x86": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll", "vcomp140.dll"},
44+
"Windows/aarch64": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll"},
45+
"Mac/aarch64": {"/usr/lib/libc++.1.dylib", "/usr/lib/libSystem.B.dylib",
46+
"/System/Library/Frameworks/Foundation.framework/Versions/C/Foundation",
47+
"/System/Library/Frameworks/Metal.framework/Versions/A/Metal",
48+
"/System/Library/Frameworks/MetalKit.framework/Versions/A/MetalKit",
49+
"/System/Library/Frameworks/Accelerate.framework/Versions/A/Accelerate",
50+
"/usr/lib/libobjc.A.dylib",
51+
"/System/Library/Frameworks/CoreFoundation.framework/Versions/A/CoreFoundation",
52+
"/System/Library/Frameworks/Security.framework/Versions/A/Security",
53+
# KNOWN DEFECT, allowed only so this check reports NEW dependencies: the macOS
54+
# build picks up the runner's Homebrew OpenSSL, so the shipped dylib does not load
55+
# on a Mac without `brew install openssl@3`. See TODO.md ("macOS dylib links
56+
# Homebrew OpenSSL"); remove these two lines with the fix.
57+
"/opt/homebrew/opt/openssl@3/lib/libssl.3.dylib",
58+
"/opt/homebrew/opt/openssl@3/lib/libcrypto.3.dylib"},
59+
}
60+
61+
# Never acceptable in any artifact: libraries a consumer cannot be expected to have.
62+
DENIED = ("libibverbs", "librdma", "rdma.dylib", "libmlx")
63+
64+
LIB_NAMES = ("libjllama.so", "jllama.dll", "libjllama.dylib")
65+
66+
67+
def elf_needed(data):
68+
if data[:4] != b"\x7fELF":
69+
raise ValueError("not an ELF file")
70+
is64 = data[4] == 2
71+
end = "<" if data[5] == 1 else ">"
72+
if is64:
73+
shoff = struct.unpack_from(end + "Q", data, 0x28)[0]
74+
shentsize, shnum = struct.unpack_from(end + "HH", data, 0x3A)
75+
else:
76+
shoff = struct.unpack_from(end + "I", data, 0x20)[0]
77+
shentsize, shnum = struct.unpack_from(end + "HH", data, 0x2E)
78+
sections = []
79+
for i in range(shnum):
80+
off = shoff + i * shentsize
81+
if is64:
82+
_, sh_type, _, _, sh_offset, sh_size, sh_link = struct.unpack_from(end + "IIQQQQI", data, off)
83+
else:
84+
_, sh_type, _, _, sh_offset, sh_size, sh_link = struct.unpack_from(end + "IIIIIII", data, off)
85+
sections.append((sh_type, sh_offset, sh_size, sh_link))
86+
out = []
87+
for sh_type, sh_offset, sh_size, sh_link in sections:
88+
if sh_type != 6: # SHT_DYNAMIC
89+
continue
90+
strtab = sections[sh_link]
91+
entry = 16 if is64 else 8
92+
for off in range(sh_offset, sh_offset + sh_size, entry):
93+
tag, val = struct.unpack_from(end + ("qQ" if is64 else "iI"), data, off)
94+
if tag == 0:
95+
break
96+
if tag == 1: # DT_NEEDED
97+
start = strtab[1] + val
98+
out.append(data[start:data.index(b"\0", start)].decode())
99+
return out
100+
101+
102+
def pe_imports(data):
103+
if data[:2] != b"MZ":
104+
raise ValueError("not a PE file")
105+
pe = struct.unpack_from("<I", data, 0x3C)[0]
106+
nsections = struct.unpack_from("<H", data, pe + 6)[0]
107+
opt_size = struct.unpack_from("<H", data, pe + 20)[0]
108+
opt = pe + 24
109+
magic = struct.unpack_from("<H", data, opt)[0]
110+
dd = opt + (112 if magic == 0x20B else 96)
111+
import_rva = struct.unpack_from("<I", data, dd + 8)[0]
112+
sec = opt + opt_size
113+
table = []
114+
for i in range(nsections):
115+
vsize, vaddr, rsize, raddr = struct.unpack_from("<IIII", data, sec + i * 40 + 8)
116+
table.append((vaddr, max(vsize, rsize), raddr))
117+
118+
def to_offset(rva):
119+
for vaddr, size, raddr in table:
120+
if vaddr <= rva < vaddr + size:
121+
return rva - vaddr + raddr
122+
raise ValueError("RVA outside every section")
123+
124+
out = []
125+
if import_rva == 0:
126+
return out
127+
off = to_offset(import_rva)
128+
while True:
129+
name_rva = struct.unpack_from("<I", data, off + 12)[0]
130+
if name_rva == 0:
131+
break
132+
start = to_offset(name_rva)
133+
out.append(data[start:data.index(b"\0", start)].decode())
134+
off += 20
135+
return out
136+
137+
138+
def macho_dylibs(data):
139+
magic = struct.unpack_from("<I", data, 0)[0]
140+
if magic != 0xFEEDFACF:
141+
raise ValueError("not a 64-bit Mach-O file")
142+
ncmds = struct.unpack_from("<I", data, 16)[0]
143+
off = 32
144+
out = []
145+
for _ in range(ncmds):
146+
cmd, size = struct.unpack_from("<II", data, off)
147+
if cmd in (0xC, 0x80000018, 0x8000001F, 0x80000023): # LOAD_DYLIB, WEAK, REEXPORT, UPWARD
148+
name_off = struct.unpack_from("<I", data, off + 8)[0]
149+
start = off + name_off
150+
out.append(data[start:data.index(b"\0", start)].decode())
151+
off += size
152+
return out
153+
154+
155+
def dependencies(path):
156+
with open(path, "rb") as f:
157+
data = f.read()
158+
if path.endswith(".so"):
159+
return elf_needed(data)
160+
if path.endswith(".dll"):
161+
return pe_imports(data)
162+
return macho_dylibs(data)
163+
164+
165+
def find_libraries(root):
166+
for dirpath, _, files in os.walk(root):
167+
for name in files:
168+
if name in LIB_NAMES:
169+
yield os.path.join(dirpath, name)
170+
171+
172+
def denied(deps):
173+
return [d for d in deps if any(bad in d.lower() for bad in DENIED)]
174+
175+
176+
def main(argv):
177+
if len(argv) < 3 or argv[1] not in ("--default", "--deny"):
178+
print(__doc__, file=sys.stderr)
179+
return 2
180+
mode, roots = argv[1], argv[2:]
181+
checked = 0
182+
failures = []
183+
for root in roots:
184+
for path in sorted(find_libraries(root)):
185+
deps = dependencies(path)
186+
checked += 1
187+
rel = os.path.relpath(path, root).replace(os.sep, "/")
188+
print(f"{rel}: {' '.join(deps)}")
189+
for d in denied(deps):
190+
failures.append(f"{rel} needs {d}, which no consumer can be expected to have")
191+
if mode == "--default":
192+
parts = rel.split("/")
193+
key = "/".join(parts[-3:-1]) if len(parts) >= 3 else ""
194+
allowed = ALLOWED.get(key)
195+
if allowed is None:
196+
failures.append(f"{rel}: no dependency allowlist for '{key}' -- add one to ALLOWED")
197+
continue
198+
for d in deps:
199+
if d.lower() not in {a.lower() for a in allowed}:
200+
failures.append(f"{rel} needs {d}, which it did not need before (allowed: {sorted(allowed)})")
201+
if checked == 0:
202+
print(f"no native library found under {roots}", file=sys.stderr)
203+
return 2
204+
for f in failures:
205+
print(f"::error::{f}", file=sys.stderr)
206+
print(f"{checked} native libraries checked, {len(failures)} violations")
207+
return 1 if failures else 0
208+
209+
210+
if __name__ == "__main__":
211+
sys.exit(main(sys.argv))

‎.github/workflows/publish.yml‎

Lines changed: 21 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3544,6 +3544,17 @@ jobs:
35443544
with:
35453545
name: Windows-x86_64-openvino
35463546
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
3547+
# Runtime dependencies of every shipped native library, read from the file itself (ELF
3548+
# DT_NEEDED, PE imports incl. Windows arm64, Mach-O load commands). The default tree must
3549+
# match an exact per-{OS}/{ARCH} allowlist -- a new dependency is a load failure on every
3550+
# machine that lacks it. The classifier trees legitimately need their vendor runtime, so
3551+
# there only the denylist applies. The case this was written for: ggml-rpc's RDMA
3552+
# transport, which upstream enables whenever the build host has libibverbs/librdma
3553+
# (llama/CMakeLists.txt forces it off).
3554+
- name: Verify native runtime dependencies
3555+
run: |
3556+
python3 .github/verify-native-deps.py --default llama/src/main/resources
3557+
python3 .github/verify-native-deps.py --deny llama/src/main/resources_*
35473558
- uses: actions/setup-java@v6
35483559
with:
35493560
distribution: 'temurin'
@@ -3679,6 +3690,12 @@ jobs:
36793690
run: .github/verify-bytecode-version.sh --max-major 52 fatjars
36803691
- name: Run fat-jar server smoke test
36813692
run: .github/smoke-test-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
3693+
# RPC over two JVMs from the same release asset: RpcServer in one, the default NativeServer
3694+
# with --rpc in the other. Requires a chat completion, the model buffer on the RPC endpoint in
3695+
# the load log (the layers really went over RPC), an accepted client on the server, and a clean
3696+
# non-SIGABRT failure naming the endpoint when --rpc names a server nobody runs.
3697+
- name: Run fat-jar RPC smoke test (two JVMs)
3698+
run: .github/smoke-rpc-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
36823699
- name: Upload server logs
36833700
if: failure()
36843701
uses: actions/upload-artifact@v7
@@ -3687,6 +3704,10 @@ jobs:
36873704
path: |
36883705
server-out.log
36893706
server-err.log
3707+
rpc-server.log
3708+
rpc-client-out.log
3709+
rpc-client-err.log
3710+
rpc-unreachable.log
36903711
if-no-files-found: warn
36913712

36923713
# The agent release asset, launched the way the README tells a user to: `java -jar` on the agent

0 commit comments

Comments
 (0)