From 0bae46887ef92e94a5a340ae4ef03bd4014e6357 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 12:10:19 +0000 Subject: [PATCH] Ship the agent as a core-less release jar, gate publishing on it The llama-atmosphere-agent is now a GitHub Release asset (never Maven Central): llama-atmosphere-agent--jar-with-dependencies.jar, built by a new `assembly` profile that excludes the core and its whole runtime graph plus jspecify/slf4j-simple, all of which every core fat jar already carries. ~7 MB instead of hundreds, no natives twice, no second SLF4J provider. Its manifest Class-Path names the core fat jars, so `java -jar` works next to any of them. CI: the model-free agent job builds it (+ .sha256) and uploads it; the snapshot and release attach jobs put it next to the core fat jars, where sign-fatjars.sh signs it. New job smoke-agent-linux runs the real pair: bytecode <= 65, the jar alone must fail for the missing core, --help, a one-shot answer and a read_file round on the cached tool model. The model-free job, the model-backed integration job and the smoke now gate both publish jobs. ToolCallingIntegrationTest: the streaming variant failed on both Windows x86-64 jobs after b11211 (512 tokens, no tool call), while the prompt never asked for the tool. Ask for it outright and put the streamed content, finish_reason and chunk count into both assertion messages. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01QSXAwbXqx9u5xsMArMvgT8 --- .github/smoke-agent-jar.sh | 115 ++++++++++++++++++ .github/workflows/publish.yml | 104 ++++++++++++++-- CHANGELOG.md | 17 +++ CLAUDE.md | 27 +++- README.md | 29 ++++- llama-atmosphere-agent/README.md | 31 ++++- llama-atmosphere-agent/pom.xml | 60 ++++++++- .../src/assembly/agent-jar.xml | 45 +++++++ .../llama/ToolCallingIntegrationTest.java | 32 ++++- 9 files changed, 427 insertions(+), 33 deletions(-) create mode 100755 .github/smoke-agent-jar.sh create mode 100644 llama-atmosphere-agent/src/assembly/agent-jar.xml diff --git a/.github/smoke-agent-jar.sh b/.github/smoke-agent-jar.sh new file mode 100755 index 000000000..07a09a460 --- /dev/null +++ b/.github/smoke-agent-jar.sh @@ -0,0 +1,115 @@ +#!/usr/bin/env bash + +# SPDX-FileCopyrightText: 2026 Bernard Ladenthin +# +# SPDX-License-Identifier: MIT + +# Smoke test for the llama-atmosphere-agent release asset, run exactly the way the README +# tells a user to run it: the agent jar lies next to a core fat jar and is started with +# `java -jar`, so the core is found only through the agent manifest's Class-Path. That +# makes this the check that the two release assets actually fit together (same version in +# the file names, nothing missing on either side), which no test run from the source tree +# can see. +# +# 1. the agent jar carries no core: started alone, loading a model fails with +# NoClassDefFoundError for net.ladenthin.llama.LlamaModel; +# 2. --help exits 0 and prints the usage; +# 3. a one-shot prompt with the model loaded in-process answers "2+2" with a 4; +# 4. a one-shot prompt that needs a reading tool (read_file) reads a marker file from +# --workspace, so the whole tool loop runs through the shipped jars. +# +# Usage: smoke-agent-jar.sh +# must hold exactly one llama-atmosphere-agent-*-jar-with-dependencies.jar and at +# least one core fat jar its Class-Path names. Output of each run is kept in agent-*.log in +# the working directory (uploaded by the CI job on failure). +set -euo pipefail + +JAR_DIR="${1:?usage: smoke-agent-jar.sh }" +MODEL="${2:?usage: smoke-agent-jar.sh }" +TIMEOUT="${AGENT_SMOKE_TIMEOUT:-600}" +# The checks grep plain text; never let a CI runner that forces colour put escapes in it. +export NO_COLOR=1 +unset CLICOLOR_FORCE + +fail() { + echo "::error::$*" >&2 + exit 1 +} + +[ -f "$MODEL" ] || fail "model not found: $MODEL" +MODEL="$(cd "$(dirname "$MODEL")" && pwd)/$(basename "$MODEL")" +JAR_DIR="$(cd "$JAR_DIR" && pwd)" + +mapfile -t AGENTS < <(find "$JAR_DIR" -maxdepth 1 -name 'llama-atmosphere-agent-*-jar-with-dependencies.jar' | sort) +[ "${#AGENTS[@]}" -eq 1 ] || fail "expected exactly 1 agent jar in $JAR_DIR, got ${#AGENTS[@]}: ${AGENTS[*]:-none}" +AGENT="${AGENTS[0]}" +echo "Agent jar: $(basename "$AGENT") ($(du -h "$AGENT" | cut -f1))" + +# The manifest names the core jars by file name; at least one of them must be here, or +# `java -jar` would start without a core. unzip wraps manifest lines at 72 bytes with a +# leading space, so the continuation lines are joined first. +CLASS_PATH="$(unzip -p "$AGENT" META-INF/MANIFEST.MF | tr -d '\r' | sed -e ':a' -e 'N' -e '$!ba' -e 's/\n //g' \ + | sed -n 's/^Class-Path: //p')" +[ -n "$CLASS_PATH" ] || fail "agent manifest has no Class-Path" +found="" +for entry in $CLASS_PATH; do + if [ -f "$JAR_DIR/$entry" ]; then + found="$entry" + break + fi +done +[ -n "$found" ] || fail "none of the core jars the agent manifest names is in $JAR_DIR: $CLASS_PATH (present: $(ls "$JAR_DIR"))" +echo "Core jar picked up via Class-Path: $found" + +# 1. Without a core next to it the agent must not work: that is what makes it small. +ALONE="$(mktemp -d)" +cp "$AGENT" "$ALONE/" +set +e +timeout "$TIMEOUT" java -jar "$ALONE/$(basename "$AGENT")" --model "$MODEL" --plain --prompt hi \ + > agent-alone.log 2>&1 < /dev/null +rc=$? +set -e +rm -rf "$ALONE" +[ "$rc" -ne 0 ] || fail "the agent jar ran without a core jar next to it - does it bundle the core?" +grep -q 'NoClassDefFoundError: net/ladenthin/llama/LlamaModel' agent-alone.log \ + || { cat agent-alone.log; fail "agent without core failed, but not for the missing core (see above)"; } +echo "OK: the agent jar carries no core" + +# 2. --help +java -jar "$AGENT" --help > agent-help.log 2>&1 < /dev/null || { cat agent-help.log; fail "--help exited non-zero"; } +grep -q 'Usage: LocalAgent' agent-help.log || { cat agent-help.log; fail "--help printed no usage"; } +echo "OK: --help" + +run_agent() { + local log="$1" + shift + set +e + timeout "$TIMEOUT" java -jar "$AGENT" --model "$MODEL" --ngl 0 --plain --temperature 0 "$@" \ + > "$log" 2>"${log%.log}.err.log" < /dev/null + local status=$? + set -e + if [ "$status" -ne 0 ]; then + echo "===== $log =====" && cat "$log" + echo "===== ${log%.log}.err.log (last 80 lines) =====" && tail -n 80 "${log%.log}.err.log" + fail "agent exited with $status ($log)" + fi +} + +# 3. A plain answer through the in-process server. +run_agent agent-answer.log --prompt 'What is 2 + 2? Answer with one short sentence.' +grep -q '4' agent-answer.log || { cat agent-answer.log; fail "the answer does not contain 4"; } +echo "OK: plain answer" + +# 4. A tool round: the marker exists only in the file, so it reaches the output only +# through read_file (whose result the console prints) or an answer built from it. +WORKSPACE="$(mktemp -d)" +MARKER="AGENT_SMOKE_$(date +%s)_$RANDOM" +printf '%s\n' "$MARKER" > "$WORKSPACE/marker.txt" +run_agent agent-tool.log --workspace "$WORKSPACE" \ + --prompt 'Read the file marker.txt with the read_file tool and tell me its exact content.' +rm -rf "$WORKSPACE" +grep -q 'read_file' agent-tool.log || { cat agent-tool.log; fail "the model did not call read_file"; } +grep -q "$MARKER" agent-tool.log || { cat agent-tool.log; fail "the marker never reached the output"; } +echo "OK: tool round (read_file)" + +echo "Agent release asset smoke test passed." diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 42b46f623..8cefbbe5b 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -533,17 +533,19 @@ jobs: # job — never in the dockcross cross-compilers (which have no node) or per-platform. # --------------------------------------------------------------------------- # --------------------------------------------------------------------------- - # llama-atmosphere-agent: the standalone (non-reactor, unpublished) local coding-agent - # project that wires Atmosphere's built-in OpenAI-compatible agent runtime to this - # project's OpenAiCompatServer. Two jobs, mirroring the langchain4j pair: + # llama-atmosphere-agent: the standalone (non-reactor, never on Maven Central) local + # coding-agent project that wires Atmosphere's built-in OpenAI-compatible agent runtime to + # this project's OpenAiCompatServer. Three jobs, all publish gates: # - model-free: unit tests + the wire-contract tests, which drive the REAL # OpenAiCompatServer over a loopback socket with a scripted backend (no native lib, # no GGUF) and pin the streamed tool_calls / role=tool / multi-round shape — seconds, - # on every PR. + # on every PR. It also builds the GitHub Release asset (the agent jar WITHOUT the core). # - model-backed: the same loop against the cached Qwen2.5-1.5B tool model through the # downloaded Linux native library (chat, streaming, tool call + result, read/write/read - # loop). Validation-only, not a publish gate: a small model's wording is not a release - # signal, the deterministic contract is the model-free job. + # loop). A gate since its assertions stopped pinning wording (content checks are limited + # to facts no instruct model gets wrong and to tool results) and it ran green throughout. + # - smoke-agent-linux (further down, after package-fatjars): the release asset itself, + # started next to the real core fat jar. # The project is built with -Dllama.version= against the core that was just # installed to the local repo, so it always tests the code of this checkout. # --------------------------------------------------------------------------- @@ -569,6 +571,28 @@ jobs: run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" spotless:check - name: Build and test (unit + model-free wire contract against the real OpenAiCompatServer) run: mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" verify + # The GitHub Release asset llama-atmosphere-agent--jar-with-dependencies.jar: + # the agent plus Atmosphere/JLine, WITHOUT the core (src/assembly/agent-jar.xml), so it is a + # few MB and the natives are not in the release twice. Never deployed to Maven Central. + # smoke-agent-linux launches it next to the real core fat jar; the attach jobs sign it. + - name: Build the agent release jar (without the core) + run: > + mvn -B --no-transfer-progress -f llama-atmosphere-agent/pom.xml "-Dllama.version=${VERSION}" + -P assembly -DskipTests package + - name: Collect the agent release jar + sha256 + run: | + mkdir -p agent-jar + cp "llama-atmosphere-agent/target/llama-atmosphere-agent-${VERSION}-jar-with-dependencies.jar" agent-jar/ + (cd agent-jar && for f in *.jar; do sha256sum "$f" > "$f.sha256"; done) + ls -la agent-jar + - name: Upload the agent release jar + uses: actions/upload-artifact@v7 + with: + name: llama-atmosphere-agent-jar + path: agent-jar/ + compression-level: 0 # jars are already deflated + retention-days: 7 + if-no-files-found: error test-java-llama-atmosphere-agent-integration: name: Integration Test llama-atmosphere-agent (model-backed) @@ -3665,6 +3689,50 @@ jobs: server-err.log if-no-files-found: warn + # The agent release asset, launched the way the README tells a user to: `java -jar` on the agent + # jar lying next to the real all-backends Linux fat jar, so the core is found only through the + # agent manifest's Class-Path. Proves the two assets fit together (matching version in the file + # names, nothing missing on either side), that the agent jar carries no core, and that a + # one-shot answer and a read_file tool round work through the shipped jars. + smoke-agent-linux: + name: Smoke test the agent release jar (Linux) + needs: [test-java-llama-atmosphere-agent, package-fatjars, verify-model-cache] + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/download-artifact@v8 + with: + name: llama-atmosphere-agent-jar + path: agent-assets/ + - uses: actions/download-artifact@v8 + with: + name: llama-fatjar-smoke-linux + path: agent-assets/ + - name: Restore shared GGUF model cache (populated by download-models; no re-download) + uses: actions/cache/restore@v6 + with: + path: models/ + key: gguf-models-${{ hashFiles('.github/models.csv') }} + enableCrossOsArchive: true + - name: Validate model files + run: .github/validate-models.sh + - uses: actions/setup-java@v6 + with: + distribution: 'temurin' + java-version: ${{ env.JAVA_VERSION }} + # The agent is Java 21 (Atmosphere's floor), unlike the Java 8 core: its ceiling is 65. + - name: Verify Java 21 bytecode (no class newer than major 65) + run: .github/verify-bytecode-version.sh --max-major 65 agent-assets/llama-atmosphere-agent-*-jar-with-dependencies.jar + - name: Run the agent release-jar smoke test + run: .github/smoke-agent-jar.sh agent-assets "models/${TOOL_MODEL_NAME}" + - name: Upload agent logs + if: failure() + uses: actions/upload-artifact@v7 + with: + name: agent-smoke-linux-logs + path: agent-*.log + if-no-files-found: warn + smoke-fatjar-windows: name: Smoke test all-backends fat jar (Windows) needs: [package-fatjars, verify-model-cache] @@ -3880,7 +3948,7 @@ jobs: publish-snapshot: name: Publish Snapshot to Central - needs: [check-snapshot, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos] + needs: [check-snapshot, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos, smoke-agent-linux] if: needs.check-snapshot.result == 'success' && inputs.publish_to_central runs-on: ubuntu-latest environment: maven-central @@ -4059,11 +4127,11 @@ jobs: github-snapshot: name: Update Snapshot Pre-release on GitHub - needs: [publish-snapshot, package-fatjars] + needs: [publish-snapshot, package-fatjars, test-java-llama-atmosphere-agent] # Also runs when publish-snapshot FAILED (not when skipped/cancelled): a Central # publish-poll timeout reds that job after the artifacts were already uploaded — # the GitHub pre-release assets must not be lost in that case. - if: ${{ !cancelled() && (needs.publish-snapshot.result == 'success' || needs.publish-snapshot.result == 'failure') && needs.package-fatjars.result == 'success' }} + if: ${{ !cancelled() && (needs.publish-snapshot.result == 'success' || needs.publish-snapshot.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }} runs-on: ubuntu-latest # maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is # scoped to this environment) for signing the fat jars below. This environment has @@ -4086,6 +4154,12 @@ jobs: with: name: llama-fatjars path: snapshot-assets/ + # The agent jar (+ sha256) — built without the core, run next to one of the fat jars above. + # Same directory, so sign-fatjars.sh signs it and the upload glob attaches it. + - uses: actions/download-artifact@v8 + with: + name: llama-atmosphere-agent-jar + path: snapshot-assets/ # GPG-sign the fat jars so each carries a detached .asc signature alongside its # .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at # deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity) @@ -4147,7 +4221,7 @@ jobs: publish-release: name: Publish Release to Central if: needs.check-tag.result == 'success' && inputs.publish_to_central - needs: [check-tag, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos] + needs: [check-tag, crosscompile-linux-x86_64-cuda, crosscompile-android-aarch64-opencl, package-android-aar, test-android-emulator, code-style, test-java-llama-langchain4j, test-java-llama-kotlin, test-java-llama-atmosphere-agent, test-java-llama-atmosphere-agent-integration, package-fatjars, smoke-fatjar-linux, smoke-fatjar-windows, smoke-fatjar-linux-aarch64, smoke-fatjar-windows-arm64, smoke-fatjar-macos, smoke-agent-linux] runs-on: ubuntu-latest environment: maven-central permissions: @@ -4326,11 +4400,11 @@ jobs: github-release-signed: name: Attach Signed Binaries to GitHub Release - needs: [publish-release, package-fatjars] + needs: [publish-release, package-fatjars, test-java-llama-atmosphere-agent] # Also runs when publish-release FAILED (not when skipped/cancelled): a Central # publish-poll timeout reds that job after the artifacts were already uploaded — # the GitHub release assets must not be lost in that case. - if: ${{ !cancelled() && (needs.publish-release.result == 'success' || needs.publish-release.result == 'failure') && needs.package-fatjars.result == 'success' }} + if: ${{ !cancelled() && (needs.publish-release.result == 'success' || needs.publish-release.result == 'failure') && needs.package-fatjars.result == 'success' && needs.test-java-llama-atmosphere-agent.result == 'success' }} runs-on: ubuntu-latest # maven-central so the GPG_PRIVATE_KEY / GPG_PASSPHRASE secret is delivered (it is # scoped to this environment) for signing the fat jars below. This environment has @@ -4353,6 +4427,12 @@ jobs: with: name: llama-fatjars path: release-assets/ + # The agent jar (+ sha256) — built without the core, run next to one of the fat jars above. + # Same directory, so sign-fatjars.sh signs it and the upload glob attaches it. + - uses: actions/download-artifact@v8 + with: + name: llama-atmosphere-agent-jar + path: release-assets/ # GPG-sign the fat jars so each carries a detached .asc signature alongside its # .sha256 checksum — signature parity with the thin jars (which maven-gpg signs at # deploy) and with the BAF / srcmorph sibling fat jars. The .sha256 files (integrity) diff --git a/CHANGELOG.md b/CHANGELOG.md index 975b57b94..317782d6f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,15 @@ from version 5.0.0 onward. Pre-fork releases (`1.x`–`4.2.0`) were authored by ## [Unreleased] ### Fixed +- **`ToolCallingIntegrationTest#requiredToolCallIsParsedFromStreamingResponse` failed on both Windows + x86-64 jobs after the b11211 bump** (the Ubuntu run and the blocking twin stayed green). The streamed + request generated its full 512 tokens without a tool call. The prompt ("Write an example") never asked + for the tool, so the grammar-constrained answer was a fragile ~90-token call even when it worked, and a + numerically different CPU path on those runners (most likely upstream's new tiled k-quant matmul, + which picks its microkernel by ISA) tipped greedy decoding over. The test pins how a tool call is + parsed and streamed, not whether a 1.5B model infers one, so the user message now asks for the call + outright; both assertions carry the streamed content, `finish_reason` and chunk count, so a future + failure says what the model did instead of `but: was ""`. - **`LlamaModel.setLogger` was silently overridden by every model load, and never saw the server's own log lines.** llama.cpp's `common_init()` — run on each load — re-points `llama_log_set()` at its own default callback, so a logger set *before* `new LlamaModel(…)` (the natural order) stopped receiving @@ -30,6 +39,14 @@ from version 5.0.0 onward. Pre-fork releases (`1.x`–`4.2.0`) were authored by documented as the no-ops they are (`common_init()` forces both on), `setLogFile` as additive. ### Added +- **The agent is a release asset: `llama-atmosphere-agent--jar-with-dependencies.jar`**, with + `.sha256` and a GPG `.asc`, on every GitHub release and the rolling `snapshot` pre-release — never on + Maven Central. It carries **no core** (~7 MB instead of hundreds, natives not in the release twice): + put it next to a core fat jar of the same version and `java -jar` finds the core through its manifest + `Class-Path`, or name both with `java -cp`. A new CI job, `smoke-agent-linux`, launches exactly that + pair (bytecode ≤ Java 21, the jar alone must fail for the missing core, `--help`, a one-shot answer and + a `read_file` round on the cached tool model), and it, the model-free agent job and the model-backed + agent integration test now gate both publish jobs. - **`llama-atmosphere-agent`: `--log-verbosity ` (default `2`) and `--verbose`** for the in-process `--model` mode. llama.cpp's per-request INFO lines go to stderr, the console the streamed answer is printed to, and interleaved with it; the agent now loads the model with warnings-and-errors only. A diff --git a/CLAUDE.md b/CLAUDE.md index 30ea41fb5..0ad7a342d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -2199,13 +2199,36 @@ releases as a signed Central Portal bundle upload (staging repo → zip → Publ A **copy-and-run general-purpose terminal agent** (Claude Code / OpenCode reduced to the essentials, offline) that pairs [Atmosphere](https://github.com/Atmosphere/atmosphere)'s built-in OpenAI-compatible agent runtime with this project's `OpenAiCompatServer`. Like `android-llmservice/` it is a -**standalone Maven project, NOT a reactor module and NOT published** — it is an application, and it +**standalone Maven project, NOT a reactor module and NEVER on Maven Central** — it is an application, and it needs Java 21 (Atmosphere's floor) while the core stays Java 8. CI builds it against the core it just installed (`-Dllama.version=`); a user copies the folder and runs `mvn compile exec:java -Dexec.args="…"` with no `-D` at all — the pom's `llama.version` names the **released** core the READMEs describe (currently `5.2.0`, written as if released so the docs are right the moment the release lands). +**Release asset: the agent jar WITHOUT the core.** `mvn -P assembly package` (the pom's `assembly` +profile, descriptor `src/assembly/agent-jar.xml`) builds +`llama-atmosphere-agent--jar-with-dependencies.jar` — named after the **core** version +it was built against, not the agent's own `1.0.0-SNAPSHOT`, because it only runs next to that core. +It excludes `net.ladenthin:llama` **with its whole runtime graph** (`useTransitiveFiltering`: Jackson 2, +slf4j-api, and Jackson 3's `jackson-annotations`, which resolves through the core's trail) plus +`jspecify` and `slf4j-simple`, all of which every core fat jar already bundles — so the asset is ~7 MB +instead of hundreds, the natives are not in the release twice, and there are never two SLF4J providers. +The manifest's `Class-Path` names the four `llama--all---…` fat jars and then the default +`llama--jar-with-dependencies.jar`, so `java -jar` works when the agent lies next to any of them +(missing entries are ignored; note that a manifest `Class-Path` is honoured under `java -cp` too). +**Rename a core fat jar and this list must follow** — `smoke-agent-linux` is what notices. +CI wiring (`publish.yml`): the model-free job builds it, writes the `.sha256` and uploads artifact +`llama-atmosphere-agent-jar`; `github-snapshot` / `github-release-signed` download it into the asset +directory next to `llama-fatjars`, so `sign-fatjars.sh` signs it (`*-jar-with-dependencies*.jar`) and +the one upload attaches it. **`smoke-agent-linux`** (`.github/smoke-agent-jar.sh`) runs the asset the +way the README tells a user to — `java -jar` next to the real `all-linux-x86-64` fat jar — and checks: +bytecode ≤ 65 (Java 21, unlike the core's 52), that the jar started **alone** fails with +`NoClassDefFoundError: net/ladenthin/llama/LlamaModel` (i.e. it really carries no core), `--help`, a +one-shot `2 + 2` answer and a `read_file` round that must surface a marker from `--workspace`, all on +the cached `TOOL_MODEL_NAME` with `--ngl 0`. **All three agent jobs gate both publish jobs** +(model-free, model-backed integration, smoke). + **What Atmosphere is, for this purpose.** `org.atmosphere:atmosphere-ai` (4.0.71) ships `BuiltInAgentRuntime` + `OpenAiCompatibleClient`: a zero-framework OpenAI client that *always* streams (`stream:true`), accumulates `delta.tool_calls` by `index`, executes `ToolDefinition` @@ -2234,7 +2257,7 @@ Spring Boot starter are a deployment layer on top of the same runtime. text so far instead of erroring. That is a SHOULD for Atmosphere's `OpenAiCompatibleClient`, not for this project. - `AtmosphereToolLoopIntegrationTest` — **model-backed, CI only** (`test-java-llama-atmosphere-agent-integration`, - validation-only, not a publish gate): the same loop against the cached Qwen2.5-1.5B tool model + a publish gate): the same loop against the cached Qwen2.5-1.5B tool model through the downloaded Linux natives — plain chat, streaming (≥ 2 chunks), a tool call whose result is answered, a read→write→read loop that changes a temp file. Self-skips without the GGUF. diff --git a/README.md b/README.md index d443db084..08150937e 100644 --- a/README.md +++ b/README.md @@ -1026,8 +1026,9 @@ essentials, fully offline; it edits files and, with `--allow-shell`, runs any co (`docker`, `git`, build tools) — built from [Atmosphere](https://github.com/Atmosphere/atmosphere)'s built-in OpenAI-compatible agent runtime (streaming, tool loop, workspace file tools) driven **headless** against this project's OpenAI-compatible server. It is a standalone Maven project (not a -reactor module, not published); you clone the repository and run it from that folder. It needs only -JDK 21+ and Maven — the core jar from Maven Central ships the natives: +reactor module, never on Maven Central). Either download it from a release (below, JDK 21+ only), or +clone the repository and run it from that folder, which needs only JDK 21+ and Maven — the core jar +from Maven Central ships the natives: ```bash # get the folder and a tool-capable model (Qwen3-4B-Instruct-2507, 2.3 GB) @@ -1041,6 +1042,27 @@ mvn -q compile exec:java \ -Dexec.args="--model models/Qwen3-4B-Instruct-2507-Q4_K_M.gguf --ctx-size 16384 --workspace /path/to/project --allow-shell" ``` +**Download instead of cloning.** Every [GitHub release](https://github.com/bernardladenthin/java-llama.cpp/releases) +carries `llama-atmosphere-agent--jar-with-dependencies.jar` (with `.sha256` and a GPG `.asc`, +like the other fat jars). It is a few MB because it holds **no core**: it runs next to one of the core +fat jars of the same release, so the natives are downloaded once. Put both in one directory and +`java -jar` finds the core through the agent's manifest (the all-backends jars are tried before the +CPU-only default jar): + +```bash +# e.g. Linux x86-64: the agent + the all-backends core fat jar of the same version +java -jar llama-atmosphere-agent-5.2.0-jar-with-dependencies.jar \ + --model Qwen3-4B-Instruct-2507-Q4_K_M.gguf --ctx-size 16384 --workspace /path/to/project --allow-shell + +# or with the classpath spelled out (any directory layout; `;` instead of `:` on Windows) +java -cp llama-atmosphere-agent-5.2.0-jar-with-dependencies.jar:llama-5.2.0-all-linux-x86-64-jar-with-dependencies.jar \ + net.ladenthin.llama.atmosphere.LocalAgent --model Qwen3-4B-Instruct-2507-Q4_K_M.gguf --workspace /path/to/project +``` + +Started without a core jar next to it, the agent stops with `NoClassDefFoundError: +net/ladenthin/llama/LlamaModel`. CI launches exactly this pair on every run (`smoke-agent-linux`) +before anything is published. + > [!WARNING] > `--allow-shell` lets the model run any command with your user's rights. By default every write and > every command is confirmed on the console (`[y]es / [n]o / [a]uto`); `--auto` turns that off. Use a @@ -1081,7 +1103,8 @@ mvn -q compile exec:java \ The full streaming tool-calling loop (tools → `delta.tool_calls` → Java tool → `role:"tool"` result → next turn, over several rounds) is verified on every PR against the real `OpenAiCompatServer` with -no model, and in CI against the Qwen2.5-1.5B tool model. See +no model, and in CI against the Qwen2.5-1.5B tool model — both gate every publish, as does the +release-jar smoke above. See [`llama-atmosphere-agent/README.md`](llama-atmosphere-agent/) for the options and the verified compatibility matrix. diff --git a/llama-atmosphere-agent/README.md b/llama-atmosphere-agent/README.md index 30c5c8ef1..ea38f6dee 100644 --- a/llama-atmosphere-agent/README.md +++ b/llama-atmosphere-agent/README.md @@ -22,9 +22,9 @@ offline: - **Shell:** an opt-in `run_command` tool (`--allow-shell`) that runs any command line through the system shell (`cmd.exe` on Windows, `sh` elsewhere). -This folder is a **standalone Maven project**, deliberately *not* a reactor module and *not* -published: CI builds and tests it against the core of the same checkout; you copy the folder and -run it. Its `pom.xml` pins `llama.version` to the release these instructions describe (**5.2.0**); +This folder is a **standalone Maven project**, deliberately *not* a reactor module and *never* on +Maven Central: CI builds and tests it against the core of the same checkout; you copy the folder and +run it — or download the ready-built jar from a release (see below). Its `pom.xml` pins `llama.version` to the release these instructions describe (**5.2.0**); pass `-Dllama.version=…` to run against another core, e.g. a `-SNAPSHOT` before a release. > [!WARNING] @@ -39,7 +39,25 @@ You need **JDK 21+** and **Maven** (`mvn -v` must report Java 21 or newer). No C CMake and no separate llama.cpp install: the core jar from Maven Central ships the native libraries for Windows, Linux and macOS. -**1. Get this folder.** It is not published as an artifact; clone the repository (or download it +**Or skip Maven entirely.** Every [GitHub release](https://github.com/bernardladenthin/java-llama.cpp/releases) +carries `llama-atmosphere-agent--jar-with-dependencies.jar` (+ `.sha256`, GPG `.asc`). It holds +**no core** — that is why it is a few MB — so download it together with a core fat jar of the same +release (`llama--all---jar-with-dependencies.jar`, or the CPU-only +`llama--jar-with-dependencies.jar`) into one directory, and `java -jar` finds the core through +its manifest: + +```bash +java -jar llama-atmosphere-agent-5.2.0-jar-with-dependencies.jar \ + --model Qwen3-4B-Instruct-2507-Q4_K_M.gguf --workspace /path/to/project +# any other layout: name both on the classpath (`;` instead of `:` on Windows) +java -cp llama-atmosphere-agent-5.2.0-jar-with-dependencies.jar:/downloads/llama-5.2.0-all-linux-x86-64-jar-with-dependencies.jar \ + net.ladenthin.llama.atmosphere.LocalAgent --model Qwen3-4B-Instruct-2507-Q4_K_M.gguf --workspace /path/to/project +``` + +Every option below works the same way; only `mvn -q compile exec:java -Dexec.args="…"` becomes +`java -jar llama-atmosphere-agent-….jar …`. + +**1. Get this folder.** To build it yourself instead, clone the repository (or download it as a ZIP from GitHub) and work in `llama-atmosphere-agent/`. The folder is self-contained — you can copy it anywhere, `.mvn/jvm.config` included: @@ -626,7 +644,10 @@ bearer auth, `/v1/models`, SSE framing) over a loopback socket with a scripted e llama.cpp-shaped chunks; no native library, no model, seconds, on every PR. *model* = `AtmosphereToolLoopIntegrationTest`: the same loop against the Qwen2.5-1.5B-Instruct tool model in CI (plain chat, streaming, a tool call whose result is answered, a read→write→read loop that -changes a temp file). It self-skips without the GGUF: +changes a temp file). It self-skips without the GGUF. Both gate every publish, together with +`smoke-agent-linux`, which starts the **release jar** next to the real Linux fat jar (`java -jar`, +the core found only through the manifest `Class-Path`) and runs a one-shot answer and a `read_file` +round on the same model: ```bash mvn -f llama-atmosphere-agent/pom.xml test -Dtest=AtmosphereToolLoopIntegrationTest \ diff --git a/llama-atmosphere-agent/pom.xml b/llama-atmosphere-agent/pom.xml index ed489f759..e87bc04a7 100644 --- a/llama-atmosphere-agent/pom.xml +++ b/llama-atmosphere-agent/pom.xml @@ -10,11 +10,12 @@ SPDX-License-Identifier: MIT 4.0.0 net.ladenthin @@ -56,6 +57,7 @@ SPDX-License-Identifier: MIT 3.16.0 3.6.0 3.6.4 + 3.8.0 3.10.3 2.99.0 net.ladenthin.llama.atmosphere.LocalAgent @@ -171,4 +173,52 @@ SPDX-License-Identifier: MIT + + + + assembly + + + + org.apache.maven.plugins + maven-assembly-plugin + ${assembly.plugin.version} + + llama-atmosphere-agent-${llama.version} + + src/assembly/agent-jar.xml + + + + ${agent.main} + + + llama-${llama.version}-all-linux-x86-64-jar-with-dependencies.jar llama-${llama.version}-all-linux-aarch64-jar-with-dependencies.jar llama-${llama.version}-all-windows-x86-64-jar-with-dependencies.jar llama-${llama.version}-all-windows-aarch64-jar-with-dependencies.jar llama-${llama.version}-jar-with-dependencies.jar + + + + + + build-agent-jar + package + + single + + + + + + + + diff --git a/llama-atmosphere-agent/src/assembly/agent-jar.xml b/llama-atmosphere-agent/src/assembly/agent-jar.xml new file mode 100644 index 000000000..d954ea045 --- /dev/null +++ b/llama-atmosphere-agent/src/assembly/agent-jar.xml @@ -0,0 +1,45 @@ + + + + + jar-with-dependencies + + jar + + false + + + / + true + true + runtime + true + + net.ladenthin:llama + org.jspecify:jspecify + org.slf4j:slf4j-simple + + + + diff --git a/llama/src/test/java/net/ladenthin/llama/ToolCallingIntegrationTest.java b/llama/src/test/java/net/ladenthin/llama/ToolCallingIntegrationTest.java index bd5a1a78a..a0755c2df 100644 --- a/llama/src/test/java/net/ladenthin/llama/ToolCallingIntegrationTest.java +++ b/llama/src/test/java/net/ladenthin/llama/ToolCallingIntegrationTest.java @@ -67,7 +67,7 @@ public void requiredToolCallIsParsedFromBlockingResponse() throws IOException { ChatResponse response = model.chat(toolRequest()); List calls = response.getFirstMessage().orElseThrow().getToolCalls(); - assertThat(calls, hasSize(1)); + assertThat("tool calls of " + response, calls, hasSize(1)); assertThat(calls.get(0).getName(), is("test")); assertThat( MAPPER.readTree(calls.get(0).getArgumentsJson()).path("success").asBoolean(), is(true)); @@ -87,9 +87,17 @@ public void requiredToolCallIsParsedFromStreamingResponse() throws IOException { StringBuilder name = new StringBuilder(); StringBuilder arguments = new StringBuilder(); + StringBuilder content = new StringBuilder(); + String finishReason = null; for (String chunk : chunks) { - JsonNode toolCalls = - MAPPER.readTree(chunk).path("choices").path(0).path("delta").path("tool_calls"); + JsonNode choice = MAPPER.readTree(chunk).path("choices").path(0); + if (choice.path("delta").path("content").isTextual()) { + content.append(choice.path("delta").path("content").asText()); + } + if (choice.path("finish_reason").isTextual()) { + finishReason = choice.path("finish_reason").asText(); + } + JsonNode toolCalls = choice.path("delta").path("tool_calls"); if (!toolCalls.isArray()) { continue; } @@ -104,14 +112,26 @@ public void requiredToolCallIsParsedFromStreamingResponse() throws IOException { } } - assertThat(name.toString(), is("test")); - assertThat(MAPPER.readTree(arguments.toString()).path("success").asBoolean(), is(true)); + // Everything the stream carried, so a failure says what the model did instead of only that + // no tool name arrived (the first Windows failure after the b11211 bump reported just ""). + String diagnostics = chunks.size() + " chunks, finish_reason=" + finishReason + ", arguments=" + arguments + + ", content=" + content; + assertThat(diagnostics, name.toString(), is("test")); + assertThat( + diagnostics, + MAPPER.readTree(arguments.toString()).path("success").asBoolean(), + is(true)); } private static ChatRequest toolRequest() { return ChatRequest.empty() .appendMessage("system", "You are a coding assistant.") - .appendMessage("user", "Write an example") + // Ask for the call outright. This test pins how a tool call is parsed and streamed, + // not whether a 1.5B model infers one from a vague request: with "Write an example" + // the constrained output was a whitespace-heavy ~90-token call that, under greedy + // decoding, a numerically different CPU path (b11211 on the Windows runners) turned + // into 512 tokens with no tool call at all. + .appendMessage("user", "Call the test tool with success set to true.") .appendTool(TEST_TOOL) .withToolChoice("required") .withParallelToolCalls(Boolean.FALSE)