Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
119 changes: 119 additions & 0 deletions .github/smoke-rpc-fatjar.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,119 @@
#!/usr/bin/env bash

# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
#
# SPDX-License-Identifier: MIT

# RPC smoke test over two JVMs, on the real release asset:
#
# JVM A java -cp <fatjar> net.ladenthin.llama.RpcServer serves this runner's devices
# JVM B java -jar <fatjar> -m <model> --rpc 127.0.0.1:<A> the default NativeServer, which
# offloads its layers to A
#
# and checks that B answers a chat completion, that its load log shows a model buffer on A's
# endpoint (the layers really went over RPC, not silently to the CPU), and that A accepted a
# client. A third launch names a server nobody runs and must fail with a message naming it and a
# normal exit -- not a SIGABRT, which is what ggml-rpc did before patches/0015.
#
# Usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>
# Output lands in rpc-server.log, rpc-client-out.log, rpc-client-err.log, rpc-unreachable.log
# (uploaded by the CI job on failure).
set -euo pipefail

JAR_DIR="${1:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
JAR_GLOB="${2:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
MODEL="${3:?usage: smoke-rpc-fatjar.sh <jar-dir> <jar-glob> <model-path>}"
RPC_PORT="${RPC_PORT:-50152}"
HTTP_PORT="${HTTP_PORT:-18181}"
UNUSED_PORT="${UNUSED_PORT:-50153}"

fail() {
echo "::error::$*" >&2
for f in rpc-server.log rpc-client-out.log rpc-client-err.log rpc-unreachable.log; do
[ -f "$f" ] && { echo "--- $f (tail) ---"; tail -60 "$f"; }
done
exit 1
}

mapfile -t JARS < <(find "$JAR_DIR" -maxdepth 1 -name "$JAR_GLOB" | sort)
[ "${#JARS[@]}" -eq 1 ] || fail "expected exactly 1 jar matching $JAR_GLOB in $JAR_DIR, got ${#JARS[@]}: ${JARS[*]:-none}"
JAR="${JARS[0]}"
[ -f "$MODEL" ] || fail "model file missing: $MODEL"

PIDS=()
cleanup() {
for pid in "${PIDS[@]}"; do
kill "$pid" 2> /dev/null || true
done
}
trap cleanup EXIT

# --- JVM A: the RPC server ----------------------------------------------------------------------
java -cp "$JAR" net.ladenthin.llama.RpcServer --port "$RPC_PORT" --threads 2 --device CPU > rpc-server.log 2>&1 &
PIDS+=($!)
SERVER_PID=$!
# Up to 300 s, like the NativeServer smoke: an all-backends jar extracts every GPU backend's
# library (CUDA, ROCm and SYCL are hundreds of MB) and fails to load each before it reaches one
# this GPU-less runner can load. 60 s was too short for that on the first CI run.
for _ in $(seq 1 100); do
kill -0 "$SERVER_PID" 2> /dev/null || fail "RpcServer exited before listening"
grep -q "RpcServer listening on 127.0.0.1:$RPC_PORT" rpc-server.log && break
sleep 3
done
grep -q "RpcServer listening on 127.0.0.1:$RPC_PORT" rpc-server.log || fail "RpcServer never reported listening"
grep -q "serving \[CPU\]" rpc-server.log || fail "RpcServer --device CPU did not serve exactly the CPU"
# The backend is chosen once. RpcServer is the one entry point that reaches the loader before
# LlamaModel, and JNI_OnLoad initializes LlamaModel, whose static block re-entered the loader and
# ran a second complete load over the library being loaded (LlamaLoader.runOnceOnThisThread).
# A manifest-less jar prints the line zero times.
[ "$(grep -c '\[jllama\] using native backend' rpc-server.log)" -le 1 ] \
|| fail "the native library was loaded more than once: $(grep -c '\[jllama\] using native backend' rpc-server.log) backend selections"
echo "RPC server up: $(grep 'RpcServer listening' rpc-server.log)"

# --- JVM B: the model, offloaded over RPC --------------------------------------------------------
java -jar "$JAR" -m "$MODEL" --host 127.0.0.1 --port "$HTTP_PORT" --chat-template chatml \
--rpc "127.0.0.1:$RPC_PORT" -ngl 99 -lv 4 > rpc-client-out.log 2> rpc-client-err.log &
PIDS+=($!)
CLIENT_PID=$!
CODE=""
for _ in $(seq 1 100); do
kill -0 "$CLIENT_PID" 2> /dev/null || fail "the RPC client server exited before becoming healthy"
CODE="$(curl -s -o /dev/null -w '%{http_code}' "http://127.0.0.1:$HTTP_PORT/health" || true)"
[ "$CODE" = "200" ] && break
sleep 3
done
[ "$CODE" = "200" ] || fail "/health never returned 200 (last code: ${CODE:-none})"

RESPONSE="$(curl -sS --fail -X POST "http://127.0.0.1:$HTTP_PORT/v1/chat/completions" \
-H 'Content-Type: application/json' \
-d '{"messages":[{"role":"user","content":"Say hello."}],"max_tokens":8,"temperature":0}')" \
|| fail "chat completion over RPC failed"
echo "$RESPONSE" | python3 -c '
import json, sys
message = json.load(sys.stdin)["choices"][0]["message"]
assert message is not None, "choices[0].message missing"
print("chat completion over RPC OK:", json.dumps(message)[:200])
' || fail "malformed chat completion response: $RESPONSE"

grep -h "model buffer size" rpc-client-out.log rpc-client-err.log | grep -q "127.0.0.1:$RPC_PORT" \
|| fail "no model buffer on the RPC server in the load log -- the layers did not go over RPC"
grep -q "Accepted client connection" rpc-server.log || fail "the RPC server never accepted a client"
echo "layers offloaded over RPC: $(grep -h 'model buffer size' rpc-client-out.log rpc-client-err.log | grep "127.0.0.1:$RPC_PORT" | head -1)"

kill "$CLIENT_PID" 2> /dev/null || true
wait "$CLIENT_PID" 2> /dev/null || true

# --- an unreachable server fails the start cleanly -----------------------------------------------
set +e
timeout 120 java -jar "$JAR" -m "$MODEL" --host 127.0.0.1 --port "$((HTTP_PORT + 1))" \
--rpc "127.0.0.1:$UNUSED_PORT" > rpc-unreachable.log 2>&1
status=$?
set -e
[ "$status" -ne 0 ] || fail "a server naming an unreachable RPC endpoint started anyway"
[ "$status" -ne 124 ] || fail "a server naming an unreachable RPC endpoint hung instead of failing"
# 134 = SIGABRT: the GGML_ABORT patches/0015 removed from the registration path
[ "$status" -ne 134 ] || fail "an unreachable RPC endpoint aborted the JVM (exit 134)"
grep -q "127.0.0.1:$UNUSED_PORT" rpc-unreachable.log || fail "the failure does not name the unreachable endpoint"
echo "unreachable endpoint rejected (exit $status): $(grep -m1 "127.0.0.1:$UNUSED_PORT" rpc-unreachable.log)"

echo "RPC smoke test PASSED"
211 changes: 211 additions & 0 deletions .github/verify-native-deps.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,211 @@
#!/usr/bin/env python3
# SPDX-FileCopyrightText: 2026 Bernard Ladenthin <bernard.ladenthin@gmail.com>
#
# SPDX-License-Identifier: MIT
"""Fail when a shipped native library needs a runtime library it did not need before.

Every jllama library is ONE file with llama.cpp and ggml linked in statically, so its dynamic
dependencies are exactly what a consumer's machine must provide. A new one is a silent break on
every machine that lacks it -- the case this guards against is ggml-rpc's RDMA transport, which
upstream switches on whenever the build host has libibverbs/librdma and which would make the
library unloadable without rdma-core. It reads the dependency list straight from the file (ELF
DT_NEEDED, PE import table, Mach-O LC_LOAD_DYLIB) with the standard library only, so it runs on
any runner and checks every architecture, including the ones binutils cannot read (Windows arm64,
Mach-O).

Usage:
verify-native-deps.py --default <resources-root> # exact allowlist per <OS>/<ARCH>
verify-native-deps.py --deny <dir>... # classifier trees: only the denylist

--default checks net/ladenthin/llama/<OS>/<ARCH>/ under the root against ALLOWED below: a
dependency outside the list fails, and so does an <OS>/<ARCH> without a list (a new platform
must be listed consciously). --deny scans every native library under the directories for DENIED
names only, because GPU classifiers legitimately need their vendor runtime.

Exit codes: 0 clean, 1 violation, 2 nothing found to check.
"""

import os
import struct
import sys

# What each default-JAR library needed when this check was introduced (5.1.0 plus the RPC backend,
# which adds nothing: its sockets are libc/libSystem/WS2_32, all already present).
ALLOWED = {
"Linux/x86_64": {"libdl.so.2", "libgomp.so.1", "libpthread.so.0", "librt.so.1", "libstdc++.so.6",
"libm.so.6", "libgcc_s.so.1", "libc.so.6", "ld-linux-x86-64.so.2"},
"Linux/aarch64": {"libgomp.so.1", "libstdc++.so.6", "libm.so.6", "libgcc_s.so.1", "libc.so.6",
"ld-linux-aarch64.so.1"},
"Linux/s390x": {"libstdc++.so.6", "libm.so.6", "libgcc_s.so.1", "libc.so.6", "ld64.so.1"},
"Linux-Android/aarch64": {"liblog.so", "libm.so", "libdl.so", "libc.so", "libandroid.so"},
"Linux-Android/x86_64": {"liblog.so", "libm.so", "libdl.so", "libc.so", "libandroid.so"},
"Windows/x86_64": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll", "vcomp140.dll"},
"Windows/x86": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll", "vcomp140.dll"},
"Windows/aarch64": {"ws2_32.dll", "kernel32.dll", "shell32.dll", "advapi32.dll"},
"Mac/aarch64": {"/usr/lib/libc++.1.dylib", "/usr/lib/libSystem.B.dylib",
"/System/Library/Frameworks/Foundation.framework/Versions/C/Foundation",
"/System/Library/Frameworks/Metal.framework/Versions/A/Metal",
"/System/Library/Frameworks/MetalKit.framework/Versions/A/MetalKit",
"/System/Library/Frameworks/Accelerate.framework/Versions/A/Accelerate",
"/usr/lib/libobjc.A.dylib",
"/System/Library/Frameworks/CoreFoundation.framework/Versions/A/CoreFoundation",
"/System/Library/Frameworks/Security.framework/Versions/A/Security",
# KNOWN DEFECT, allowed only so this check reports NEW dependencies: the macOS
# build picks up the runner's Homebrew OpenSSL, so the shipped dylib does not load
# on a Mac without `brew install openssl@3`. See TODO.md ("macOS dylib links
# Homebrew OpenSSL"); remove these two lines with the fix.
"/opt/homebrew/opt/openssl@3/lib/libssl.3.dylib",
"/opt/homebrew/opt/openssl@3/lib/libcrypto.3.dylib"},
}

# Never acceptable in any artifact: libraries a consumer cannot be expected to have.
DENIED = ("libibverbs", "librdma", "rdma.dylib", "libmlx")

LIB_NAMES = ("libjllama.so", "jllama.dll", "libjllama.dylib")


def elf_needed(data):
if data[:4] != b"\x7fELF":
raise ValueError("not an ELF file")
is64 = data[4] == 2
end = "<" if data[5] == 1 else ">"
if is64:
shoff = struct.unpack_from(end + "Q", data, 0x28)[0]
shentsize, shnum = struct.unpack_from(end + "HH", data, 0x3A)
else:
shoff = struct.unpack_from(end + "I", data, 0x20)[0]
shentsize, shnum = struct.unpack_from(end + "HH", data, 0x2E)
sections = []
for i in range(shnum):
off = shoff + i * shentsize
if is64:
_, sh_type, _, _, sh_offset, sh_size, sh_link = struct.unpack_from(end + "IIQQQQI", data, off)
else:
_, sh_type, _, _, sh_offset, sh_size, sh_link = struct.unpack_from(end + "IIIIIII", data, off)
sections.append((sh_type, sh_offset, sh_size, sh_link))
out = []
for sh_type, sh_offset, sh_size, sh_link in sections:
if sh_type != 6: # SHT_DYNAMIC
continue
strtab = sections[sh_link]
entry = 16 if is64 else 8
for off in range(sh_offset, sh_offset + sh_size, entry):
tag, val = struct.unpack_from(end + ("qQ" if is64 else "iI"), data, off)
if tag == 0:
break
if tag == 1: # DT_NEEDED
start = strtab[1] + val
out.append(data[start:data.index(b"\0", start)].decode())
return out


def pe_imports(data):
if data[:2] != b"MZ":
raise ValueError("not a PE file")
pe = struct.unpack_from("<I", data, 0x3C)[0]
nsections = struct.unpack_from("<H", data, pe + 6)[0]
opt_size = struct.unpack_from("<H", data, pe + 20)[0]
opt = pe + 24
magic = struct.unpack_from("<H", data, opt)[0]
dd = opt + (112 if magic == 0x20B else 96)
import_rva = struct.unpack_from("<I", data, dd + 8)[0]
sec = opt + opt_size
table = []
for i in range(nsections):
vsize, vaddr, rsize, raddr = struct.unpack_from("<IIII", data, sec + i * 40 + 8)
table.append((vaddr, max(vsize, rsize), raddr))

def to_offset(rva):
for vaddr, size, raddr in table:
if vaddr <= rva < vaddr + size:
return rva - vaddr + raddr
raise ValueError("RVA outside every section")

out = []
if import_rva == 0:
return out
off = to_offset(import_rva)
while True:
name_rva = struct.unpack_from("<I", data, off + 12)[0]
if name_rva == 0:
break
start = to_offset(name_rva)
out.append(data[start:data.index(b"\0", start)].decode())
off += 20
return out


def macho_dylibs(data):
magic = struct.unpack_from("<I", data, 0)[0]
if magic != 0xFEEDFACF:
raise ValueError("not a 64-bit Mach-O file")
ncmds = struct.unpack_from("<I", data, 16)[0]
off = 32
out = []
for _ in range(ncmds):
cmd, size = struct.unpack_from("<II", data, off)
if cmd in (0xC, 0x80000018, 0x8000001F, 0x80000023): # LOAD_DYLIB, WEAK, REEXPORT, UPWARD
name_off = struct.unpack_from("<I", data, off + 8)[0]
start = off + name_off
out.append(data[start:data.index(b"\0", start)].decode())
off += size
return out


def dependencies(path):
with open(path, "rb") as f:
data = f.read()
if path.endswith(".so"):
return elf_needed(data)
if path.endswith(".dll"):
return pe_imports(data)
return macho_dylibs(data)


def find_libraries(root):
for dirpath, _, files in os.walk(root):
for name in files:
if name in LIB_NAMES:
yield os.path.join(dirpath, name)


def denied(deps):
return [d for d in deps if any(bad in d.lower() for bad in DENIED)]


def main(argv):
if len(argv) < 3 or argv[1] not in ("--default", "--deny"):
print(__doc__, file=sys.stderr)
return 2
mode, roots = argv[1], argv[2:]
checked = 0
failures = []
for root in roots:
for path in sorted(find_libraries(root)):
deps = dependencies(path)
checked += 1
rel = os.path.relpath(path, root).replace(os.sep, "/")
print(f"{rel}: {' '.join(deps)}")
for d in denied(deps):
failures.append(f"{rel} needs {d}, which no consumer can be expected to have")
if mode == "--default":
parts = rel.split("/")
key = "/".join(parts[-3:-1]) if len(parts) >= 3 else ""
allowed = ALLOWED.get(key)
if allowed is None:
failures.append(f"{rel}: no dependency allowlist for '{key}' -- add one to ALLOWED")
continue
for d in deps:
if d.lower() not in {a.lower() for a in allowed}:
failures.append(f"{rel} needs {d}, which it did not need before (allowed: {sorted(allowed)})")
if checked == 0:
print(f"no native library found under {roots}", file=sys.stderr)
return 2
for f in failures:
print(f"::error::{f}", file=sys.stderr)
print(f"{checked} native libraries checked, {len(failures)} violations")
return 1 if failures else 0


if __name__ == "__main__":
sys.exit(main(sys.argv))
21 changes: 21 additions & 0 deletions .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -3544,6 +3544,17 @@ jobs:
with:
name: Windows-x86_64-openvino
path: ${{ github.workspace }}/llama/src/main/resources_windows_openvino/net/ladenthin/llama/
# Runtime dependencies of every shipped native library, read from the file itself (ELF
# DT_NEEDED, PE imports incl. Windows arm64, Mach-O load commands). The default tree must
# match an exact per-{OS}/{ARCH} allowlist -- a new dependency is a load failure on every
# machine that lacks it. The classifier trees legitimately need their vendor runtime, so
# there only the denylist applies. The case this was written for: ggml-rpc's RDMA
# transport, which upstream enables whenever the build host has libibverbs/librdma
# (llama/CMakeLists.txt forces it off).
- name: Verify native runtime dependencies
run: |
python3 .github/verify-native-deps.py --default llama/src/main/resources
python3 .github/verify-native-deps.py --deny llama/src/main/resources_*
- uses: actions/setup-java@v6
with:
distribution: 'temurin'
Expand Down Expand Up @@ -3679,6 +3690,12 @@ jobs:
run: .github/verify-bytecode-version.sh --max-major 52 fatjars
- name: Run fat-jar server smoke test
run: .github/smoke-test-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
# RPC over two JVMs from the same release asset: RpcServer in one, the default NativeServer
# with --rpc in the other. Requires a chat completion, the model buffer on the RPC endpoint in
# the load log (the layers really went over RPC), an accepted client on the server, and a clean
# non-SIGABRT failure naming the endpoint when --rpc names a server nobody runs.
- name: Run fat-jar RPC smoke test (two JVMs)
run: .github/smoke-rpc-fatjar.sh fatjars 'llama-*-all-linux-x86-64-jar-with-dependencies.jar' "models/${DRAFT_MODEL_NAME}"
- name: Upload server logs
if: failure()
uses: actions/upload-artifact@v7
Expand All @@ -3687,6 +3704,10 @@ jobs:
path: |
server-out.log
server-err.log
rpc-server.log
rpc-client-out.log
rpc-client-err.log
rpc-unreachable.log
if-no-files-found: warn

# The agent release asset, launched the way the README tells a user to: `java -jar` on the agent
Expand Down
Loading
Loading