diff --git a/CHANGELOG.md b/CHANGELOG.md index 1aa5bbec..bdc79c41 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval- ## Unreleased +## [0.182.0] — 2026-09-15 + +### Breaking + +- Retire the legacy loops supervisor-run reader and its compatibility exports. + Runtime supervisor runs now use the `.agent/supervisor` layout and the runtime journal dialect only. +- Remove legacy `.loops`, `state.status`, and `result.sup_status` fallback behavior. + Missing runtime evidence remains explicitly unavailable. + +### Changed + +- Reduce paired-score gate work and use the exact shortcut for equal-magnitude Wilcoxon differences. +- Refresh the analyst benchmark implementation identity after the evaluation changes. + ## [0.181.0] — 2026-09-14 ### Changed diff --git a/clients/python/pyproject.toml b/clients/python/pyproject.toml index 36c9f2c3..78c87bfc 100644 --- a/clients/python/pyproject.toml +++ b/clients/python/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "agent-eval-rpc" -version = "0.181.0" +version = "0.182.0" description = "Python RPC client, official optimizer bridge, and DSPy metric adapter for @tangle-network/agent-eval." readme = "README.md" requires-python = ">=3.10" diff --git a/clients/python/src/agent_eval_rpc/__init__.py b/clients/python/src/agent_eval_rpc/__init__.py index 07456580..87f47405 100644 --- a/clients/python/src/agent_eval_rpc/__init__.py +++ b/clients/python/src/agent_eval_rpc/__init__.py @@ -53,7 +53,7 @@ try: __version__ = version("agent-eval-rpc") except PackageNotFoundError: - __version__ = "0.181.0" + __version__ = "0.182.0" __all__ = [ "Client", diff --git a/clients/python/uv.lock b/clients/python/uv.lock index d116e2a4..73666706 100644 --- a/clients/python/uv.lock +++ b/clients/python/uv.lock @@ -34,7 +34,7 @@ conflicts = [[ [[package]] name = "agent-eval-rpc" -version = "0.181.0" +version = "0.182.0" source = { editable = "." } dependencies = [ { name = "filelock" }, diff --git a/package.json b/package.json index 5d4157a1..3654ad04 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@tangle-network/agent-eval", - "version": "0.181.0", + "version": "0.182.0", "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.", "homepage": "https://github.com/tangle-network/agent-eval#readme", "repository": { diff --git a/src/analyst/benchmark-implementation.ts b/src/analyst/benchmark-implementation.ts index 880c4a63..b06de696 100644 --- a/src/analyst/benchmark-implementation.ts +++ b/src/analyst/benchmark-implementation.ts @@ -10,7 +10,7 @@ export const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([ ]) export const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = - '39632afc0f02da9333079d4c6323bab101d1bd0eb47a0ea06fa910ae181b4b40' + 'dfee549f65d9965471f1925b1cb8ab2860c74b09f62ea60e49478c0bee16883a' /** The published benchmark evidence was produced at this package version, by * the retired one-shot direct runner, before trace analysts moved to the