From b1340b9ad11ab1d72efdfe76f8a5170881e7a950 Mon Sep 17 00:00:00 2001 From: roylin Date: Tue, 8 Sep 2026 20:24:29 +0800 Subject: [PATCH] docs: publish v8.5.1 site line and clarify grep vs zvec planes Archive v8.4.0, make v8.5.1 the current docs default, and update README plus guide/tools so exact grep, trigram pruning, and bm25 stay separate. Co-authored-by: Cursor --- README.md | 31 +- README.zh-CN.md | 29 +- website/README.md | 18 +- website/docs/v8.5.1/en/_meta.json | 12 + website/docs/v8.5.1/en/_nav.json | 42 + website/docs/v8.5.1/en/api/index.mdx | 337 +++++++ website/docs/v8.5.1/en/guide/_meta.json | 72 ++ website/docs/v8.5.1/en/guide/agent-dir.mdx | 196 ++++ website/docs/v8.5.1/en/guide/agents-md.mdx | 81 ++ website/docs/v8.5.1/en/guide/api-contract.mdx | 873 ++++++++++++++++++ website/docs/v8.5.1/en/guide/architecture.mdx | 240 +++++ .../en/guide/cluster-extension-points.mdx | 601 ++++++++++++ website/docs/v8.5.1/en/guide/commands.mdx | 70 ++ website/docs/v8.5.1/en/guide/context.mdx | 295 ++++++ .../guide/convention-over-configuration.mdx | 233 +++++ .../docs/v8.5.1/en/guide/examples/_meta.json | 22 + .../v8.5.1/en/guide/examples/auto-compact.mdx | 173 ++++ .../docs/v8.5.1/en/guide/examples/batch.mdx | 194 ++++ .../v8.5.1/en/guide/examples/direct-tools.mdx | 306 ++++++ .../en/guide/examples/external-tasks.mdx | 250 +++++ .../v8.5.1/en/guide/examples/git-worktree.mdx | 232 +++++ .../docs/v8.5.1/en/guide/examples/hooks.mdx | 209 +++++ .../docs/v8.5.1/en/guide/examples/index.mdx | 24 + .../v8.5.1/en/guide/examples/lane-queue.mdx | 214 +++++ .../docs/v8.5.1/en/guide/examples/memory.mdx | 196 ++++ .../en/guide/examples/model-switching.mdx | 367 ++++++++ .../en/guide/examples/orchestration.mdx | 797 ++++++++++++++++ .../v8.5.1/en/guide/examples/planning.mdx | 140 +++ .../v8.5.1/en/guide/examples/prompt-slots.mdx | 366 ++++++++ .../v8.5.1/en/guide/examples/quick-start.mdx | 126 +++ .../en/guide/examples/ripgrep-context.mdx | 160 ++++ .../v8.5.1/en/guide/examples/security.mdx | 441 +++++++++ .../v8.5.1/en/guide/examples/skill-tool.mdx | 200 ++++ .../docs/v8.5.1/en/guide/examples/skills.mdx | 188 ++++ .../v8.5.1/en/guide/examples/streaming.mdx | 209 +++++ .../en/guide/examples/structured-output.mdx | 599 ++++++++++++ .../v8.5.1/en/guide/filesystem-agents.mdx | 75 ++ .../v8.5.1/en/guide/filesystem-config.mdx | 93 ++ .../docs/v8.5.1/en/guide/filesystem-first.mdx | 68 ++ .../en/guide/filesystem-instructions.mdx | 50 + .../v8.5.1/en/guide/filesystem-schedules.mdx | 74 ++ .../v8.5.1/en/guide/filesystem-skills.mdx | 67 ++ .../docs/v8.5.1/en/guide/filesystem-tools.mdx | 67 ++ website/docs/v8.5.1/en/guide/hooks.mdx | 144 +++ website/docs/v8.5.1/en/guide/index.mdx | 391 ++++++++ website/docs/v8.5.1/en/guide/isolation.mdx | 100 ++ website/docs/v8.5.1/en/guide/lane-queue.mdx | 58 ++ website/docs/v8.5.1/en/guide/limits.mdx | 189 ++++ website/docs/v8.5.1/en/guide/mcp.mdx | 260 ++++++ website/docs/v8.5.1/en/guide/memory.mdx | 122 +++ .../docs/v8.5.1/en/guide/multi-machine.mdx | 256 +++++ .../docs/v8.5.1/en/guide/orchestration.mdx | 470 ++++++++++ website/docs/v8.5.1/en/guide/persistence.mdx | 297 ++++++ website/docs/v8.5.1/en/guide/providers.mdx | 131 +++ website/docs/v8.5.1/en/guide/rfcs/_meta.json | 1 + .../en/guide/rfcs/workspace-remote-git.mdx | 537 +++++++++++ website/docs/v8.5.1/en/guide/security.mdx | 239 +++++ website/docs/v8.5.1/en/guide/sessions.mdx | 593 ++++++++++++ website/docs/v8.5.1/en/guide/skills.mdx | 82 ++ website/docs/v8.5.1/en/guide/tasks.mdx | 352 +++++++ website/docs/v8.5.1/en/guide/teams.mdx | 128 +++ website/docs/v8.5.1/en/guide/telemetry.mdx | 136 +++ website/docs/v8.5.1/en/guide/tools.mdx | 652 +++++++++++++ website/docs/v8.5.1/en/guide/tui.mdx | 719 +++++++++++++++ website/docs/v8.5.1/en/guide/verification.mdx | 390 ++++++++ .../v8.5.1/en/guide/workspace-backends.mdx | 541 +++++++++++ website/docs/v8.5.1/en/index.mdx | 7 + website/docs/v8.5.1/zh/_meta.json | 12 + website/docs/v8.5.1/zh/_nav.json | 42 + website/docs/v8.5.1/zh/api/index.mdx | 306 ++++++ website/docs/v8.5.1/zh/guide/_meta.json | 72 ++ website/docs/v8.5.1/zh/guide/agent-dir.mdx | 194 ++++ website/docs/v8.5.1/zh/guide/agents-md.mdx | 73 ++ website/docs/v8.5.1/zh/guide/api-contract.mdx | 807 ++++++++++++++++ website/docs/v8.5.1/zh/guide/architecture.mdx | 182 ++++ .../zh/guide/cluster-extension-points.mdx | 598 ++++++++++++ website/docs/v8.5.1/zh/guide/commands.mdx | 59 ++ website/docs/v8.5.1/zh/guide/context.mdx | 221 +++++ .../guide/convention-over-configuration.mdx | 230 +++++ .../docs/v8.5.1/zh/guide/examples/_meta.json | 22 + .../v8.5.1/zh/guide/examples/auto-compact.mdx | 168 ++++ .../docs/v8.5.1/zh/guide/examples/batch.mdx | 189 ++++ .../v8.5.1/zh/guide/examples/direct-tools.mdx | 296 ++++++ .../zh/guide/examples/external-tasks.mdx | 247 +++++ .../v8.5.1/zh/guide/examples/git-worktree.mdx | 231 +++++ .../docs/v8.5.1/zh/guide/examples/hooks.mdx | 199 ++++ .../docs/v8.5.1/zh/guide/examples/index.mdx | 22 + .../v8.5.1/zh/guide/examples/lane-queue.mdx | 215 +++++ .../docs/v8.5.1/zh/guide/examples/memory.mdx | 189 ++++ .../zh/guide/examples/model-switching.mdx | 362 ++++++++ .../zh/guide/examples/orchestration.mdx | 785 ++++++++++++++++ .../v8.5.1/zh/guide/examples/planning.mdx | 134 +++ .../v8.5.1/zh/guide/examples/prompt-slots.mdx | 358 +++++++ .../v8.5.1/zh/guide/examples/quick-start.mdx | 122 +++ .../zh/guide/examples/ripgrep-context.mdx | 155 ++++ .../v8.5.1/zh/guide/examples/security.mdx | 435 +++++++++ .../v8.5.1/zh/guide/examples/skill-tool.mdx | 192 ++++ .../docs/v8.5.1/zh/guide/examples/skills.mdx | 180 ++++ .../v8.5.1/zh/guide/examples/streaming.mdx | 205 ++++ .../zh/guide/examples/structured-output.mdx | 588 ++++++++++++ .../v8.5.1/zh/guide/filesystem-agents.mdx | 74 ++ .../v8.5.1/zh/guide/filesystem-config.mdx | 92 ++ .../docs/v8.5.1/zh/guide/filesystem-first.mdx | 68 ++ .../zh/guide/filesystem-instructions.mdx | 50 + .../v8.5.1/zh/guide/filesystem-schedules.mdx | 74 ++ .../v8.5.1/zh/guide/filesystem-skills.mdx | 74 ++ .../docs/v8.5.1/zh/guide/filesystem-tools.mdx | 69 ++ website/docs/v8.5.1/zh/guide/hooks.mdx | 128 +++ website/docs/v8.5.1/zh/guide/index.mdx | 320 +++++++ website/docs/v8.5.1/zh/guide/isolation.mdx | 86 ++ website/docs/v8.5.1/zh/guide/lane-queue.mdx | 46 + website/docs/v8.5.1/zh/guide/limits.mdx | 183 ++++ website/docs/v8.5.1/zh/guide/mcp.mdx | 241 +++++ website/docs/v8.5.1/zh/guide/memory.mdx | 84 ++ .../docs/v8.5.1/zh/guide/multi-machine.mdx | 242 +++++ .../docs/v8.5.1/zh/guide/orchestration.mdx | 438 +++++++++ website/docs/v8.5.1/zh/guide/persistence.mdx | 277 ++++++ website/docs/v8.5.1/zh/guide/providers.mdx | 119 +++ website/docs/v8.5.1/zh/guide/rfcs/_meta.json | 1 + .../zh/guide/rfcs/workspace-remote-git.mdx | 500 ++++++++++ website/docs/v8.5.1/zh/guide/security.mdx | 65 ++ website/docs/v8.5.1/zh/guide/sessions.mdx | 555 +++++++++++ website/docs/v8.5.1/zh/guide/skills.mdx | 47 + website/docs/v8.5.1/zh/guide/tasks.mdx | 334 +++++++ website/docs/v8.5.1/zh/guide/teams.mdx | 127 +++ website/docs/v8.5.1/zh/guide/telemetry.mdx | 132 +++ website/docs/v8.5.1/zh/guide/tools.mdx | 559 +++++++++++ website/docs/v8.5.1/zh/guide/tui.mdx | 377 ++++++++ website/docs/v8.5.1/zh/guide/verification.mdx | 356 +++++++ .../v8.5.1/zh/guide/workspace-backends.mdx | 528 +++++++++++ website/docs/v8.5.1/zh/index.mdx | 7 + website/rspress.config.ts | 3 +- website/theme/components/TuiWelcomeBanner.tsx | 2 +- website/version-snapshots.json | 9 +- 134 files changed, 30143 insertions(+), 46 deletions(-) create mode 100644 website/docs/v8.5.1/en/_meta.json create mode 100644 website/docs/v8.5.1/en/_nav.json create mode 100644 website/docs/v8.5.1/en/api/index.mdx create mode 100644 website/docs/v8.5.1/en/guide/_meta.json create mode 100644 website/docs/v8.5.1/en/guide/agent-dir.mdx create mode 100644 website/docs/v8.5.1/en/guide/agents-md.mdx create mode 100644 website/docs/v8.5.1/en/guide/api-contract.mdx create mode 100644 website/docs/v8.5.1/en/guide/architecture.mdx create mode 100644 website/docs/v8.5.1/en/guide/cluster-extension-points.mdx create mode 100644 website/docs/v8.5.1/en/guide/commands.mdx create mode 100644 website/docs/v8.5.1/en/guide/context.mdx create mode 100644 website/docs/v8.5.1/en/guide/convention-over-configuration.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/_meta.json create mode 100644 website/docs/v8.5.1/en/guide/examples/auto-compact.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/batch.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/direct-tools.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/external-tasks.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/git-worktree.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/hooks.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/index.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/lane-queue.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/memory.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/model-switching.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/orchestration.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/planning.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/prompt-slots.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/quick-start.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/ripgrep-context.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/security.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/skill-tool.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/skills.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/streaming.mdx create mode 100644 website/docs/v8.5.1/en/guide/examples/structured-output.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-agents.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-config.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-first.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-instructions.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-schedules.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-skills.mdx create mode 100644 website/docs/v8.5.1/en/guide/filesystem-tools.mdx create mode 100644 website/docs/v8.5.1/en/guide/hooks.mdx create mode 100644 website/docs/v8.5.1/en/guide/index.mdx create mode 100644 website/docs/v8.5.1/en/guide/isolation.mdx create mode 100644 website/docs/v8.5.1/en/guide/lane-queue.mdx create mode 100644 website/docs/v8.5.1/en/guide/limits.mdx create mode 100644 website/docs/v8.5.1/en/guide/mcp.mdx create mode 100644 website/docs/v8.5.1/en/guide/memory.mdx create mode 100644 website/docs/v8.5.1/en/guide/multi-machine.mdx create mode 100644 website/docs/v8.5.1/en/guide/orchestration.mdx create mode 100644 website/docs/v8.5.1/en/guide/persistence.mdx create mode 100644 website/docs/v8.5.1/en/guide/providers.mdx create mode 100644 website/docs/v8.5.1/en/guide/rfcs/_meta.json create mode 100644 website/docs/v8.5.1/en/guide/rfcs/workspace-remote-git.mdx create mode 100644 website/docs/v8.5.1/en/guide/security.mdx create mode 100644 website/docs/v8.5.1/en/guide/sessions.mdx create mode 100644 website/docs/v8.5.1/en/guide/skills.mdx create mode 100644 website/docs/v8.5.1/en/guide/tasks.mdx create mode 100644 website/docs/v8.5.1/en/guide/teams.mdx create mode 100644 website/docs/v8.5.1/en/guide/telemetry.mdx create mode 100644 website/docs/v8.5.1/en/guide/tools.mdx create mode 100644 website/docs/v8.5.1/en/guide/tui.mdx create mode 100644 website/docs/v8.5.1/en/guide/verification.mdx create mode 100644 website/docs/v8.5.1/en/guide/workspace-backends.mdx create mode 100644 website/docs/v8.5.1/en/index.mdx create mode 100644 website/docs/v8.5.1/zh/_meta.json create mode 100644 website/docs/v8.5.1/zh/_nav.json create mode 100644 website/docs/v8.5.1/zh/api/index.mdx create mode 100644 website/docs/v8.5.1/zh/guide/_meta.json create mode 100644 website/docs/v8.5.1/zh/guide/agent-dir.mdx create mode 100644 website/docs/v8.5.1/zh/guide/agents-md.mdx create mode 100644 website/docs/v8.5.1/zh/guide/api-contract.mdx create mode 100644 website/docs/v8.5.1/zh/guide/architecture.mdx create mode 100644 website/docs/v8.5.1/zh/guide/cluster-extension-points.mdx create mode 100644 website/docs/v8.5.1/zh/guide/commands.mdx create mode 100644 website/docs/v8.5.1/zh/guide/context.mdx create mode 100644 website/docs/v8.5.1/zh/guide/convention-over-configuration.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/_meta.json create mode 100644 website/docs/v8.5.1/zh/guide/examples/auto-compact.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/batch.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/direct-tools.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/external-tasks.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/git-worktree.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/hooks.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/index.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/lane-queue.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/memory.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/model-switching.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/orchestration.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/planning.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/prompt-slots.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/quick-start.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/ripgrep-context.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/security.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/skill-tool.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/skills.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/streaming.mdx create mode 100644 website/docs/v8.5.1/zh/guide/examples/structured-output.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-agents.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-config.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-first.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-instructions.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-schedules.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-skills.mdx create mode 100644 website/docs/v8.5.1/zh/guide/filesystem-tools.mdx create mode 100644 website/docs/v8.5.1/zh/guide/hooks.mdx create mode 100644 website/docs/v8.5.1/zh/guide/index.mdx create mode 100644 website/docs/v8.5.1/zh/guide/isolation.mdx create mode 100644 website/docs/v8.5.1/zh/guide/lane-queue.mdx create mode 100644 website/docs/v8.5.1/zh/guide/limits.mdx create mode 100644 website/docs/v8.5.1/zh/guide/mcp.mdx create mode 100644 website/docs/v8.5.1/zh/guide/memory.mdx create mode 100644 website/docs/v8.5.1/zh/guide/multi-machine.mdx create mode 100644 website/docs/v8.5.1/zh/guide/orchestration.mdx create mode 100644 website/docs/v8.5.1/zh/guide/persistence.mdx create mode 100644 website/docs/v8.5.1/zh/guide/providers.mdx create mode 100644 website/docs/v8.5.1/zh/guide/rfcs/_meta.json create mode 100644 website/docs/v8.5.1/zh/guide/rfcs/workspace-remote-git.mdx create mode 100644 website/docs/v8.5.1/zh/guide/security.mdx create mode 100644 website/docs/v8.5.1/zh/guide/sessions.mdx create mode 100644 website/docs/v8.5.1/zh/guide/skills.mdx create mode 100644 website/docs/v8.5.1/zh/guide/tasks.mdx create mode 100644 website/docs/v8.5.1/zh/guide/teams.mdx create mode 100644 website/docs/v8.5.1/zh/guide/telemetry.mdx create mode 100644 website/docs/v8.5.1/zh/guide/tools.mdx create mode 100644 website/docs/v8.5.1/zh/guide/tui.mdx create mode 100644 website/docs/v8.5.1/zh/guide/verification.mdx create mode 100644 website/docs/v8.5.1/zh/guide/workspace-backends.mdx create mode 100644 website/docs/v8.5.1/zh/index.mdx diff --git a/README.md b/README.md index 234d6e28..2a30ea57 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@ and recovery sit behind explicit contracts — from Rust, Node.js, Python, Go, o

Start · - v8.4 · + v8.5 · Why Code · Capabilities · Configure · @@ -34,28 +34,25 @@ and recovery sit behind explicit contracts — from Rust, Node.js, Python, Go, o Documentation

-## What's new in 8.4 +## What's new in 8.5 -Harness convergence (one baseline path, refuse dual surfaces): +Workspace search planes stay separate; session-store reopen recovers under flock: -- **Thin library defaults.** `a3s-code-core` defaults to `local-code`; SDK - crates default to bundled zvec FTS. Enable `advanced-harness`, `server`, - and/or `headless-search` explicitly for product embeds. -- **Unified `task` fan-out.** Model-visible `parallel_task` and matching SDK - helpers are removed; multi-item delegation uses `task` / `session.tasks`. -- **Active-only durable memory.** Serving is `active_recall` only; Candidate - shadow mode is gone. -- **`update_plan` + reply language.** Built-in checklist tool and host - `set_output_language` / `outputLanguage` on Rust and all SDKs. -- **SDK capabilities v2** with `tier: baseline | advanced`. Gate-mode - evaluation fail-closes on incomplete evidence. +- **`grep` candidate pruning (CODE-G1).** Default `local-code` builds an + in-tree trigram filter under `.a3s-code/grep-trigram` so literal needles open + fewer files before the exact regex scan. Non-literals and index failures fail + open. Exact matches remain owned by Code — this path never opens durable zvec + FTS (`mode: "bm25"` stays the ranked plane). +- **Session-store WAL flock (8.5.1).** Concurrent writers re-read the durable + max sequence under a cross-process flock; corrupt WALs can be quarantined so + hosts continue from durable snapshots. -Docs: [a3s-lab.github.io/Code](https://a3s-lab.github.io/Code/) (`v8.4.0`). -Wrap-up: [manual/HARNESS_CONVERGENCE.md](manual/HARNESS_CONVERGENCE.md) · -[manual/FIRST_PRINCIPLES_E2E.md](manual/FIRST_PRINCIPLES_E2E.md). +Docs: [a3s-lab.github.io/Code](https://a3s-lab.github.io/Code/) (`v8.5.1`). ### Earlier lines +- **8.4** — thin `local-code` defaults, unified `task` fan-out, Active-only + durable memory, `update_plan`, SDK capabilities v2. - **8.3** — negotiable session-store durability, typed tool-result trust, workspace source snapshots, fallible FFI init, host checkpoint hooks. - **8.0+** — run-owned spacetime, generation-exact capabilities, portable diff --git a/README.zh-CN.md b/README.zh-CN.md index 4f58c9a3..c79861eb 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -24,7 +24,7 @@

起步 · - v8.4 · + v8.5 · 为何选择 Code · 能力 · 配置 · @@ -32,28 +32,23 @@ 文档

-## 8.4 有什么新内容 +## 8.5 有什么新内容 -Harness 收敛(单一基线路径,拒绝双轨表面): +工作区搜索平面保持分离;会话存储重开在 flock 下恢复: -- **库默认变薄。** `a3s-code-core` 默认 `local-code`;SDK crate 默认捆绑 - zvec FTS。产品嵌入需显式启用 `advanced-harness`、`server` 和/或 - `headless-search`。 -- **统一 `task` 扇出。** 模型可见的 `parallel_task` 与对应 SDK 辅助 API 已移除; - 多条目委派使用 `task` / `session.tasks`。 -- **仅 Active 的 Durable Memory。** 服务路径仅为 `active_recall`;Candidate - shadow 模式已移除。 -- **`update_plan` + 回复语言。** 内置清单工具;宿主可在 Rust 与各 SDK 上使用 - `set_output_language` / `outputLanguage`。 -- **SDK capabilities v2**,带 `tier: baseline | advanced`。Gate 模式在证据 - 不完整时 fail-close。 +- **`grep` 候选裁剪(CODE-G1)。** 默认 `local-code` 在 + `.a3s-code/grep-trigram` 下构建进程内 trigram 过滤器,使字面量模式在精确 + 正则扫描前打开更少文件。非字面量与索引失败会失败开放。精确匹配仍由 Code + 拥有 — 该路径不会打开持久 zvec FTS(排序检索仍用 `mode: "bm25"`)。 +- **会话存储 WAL flock(8.5.1)。** 并发写者在跨进程 flock 下重读持久最大 + 序号;可隔离损坏的 WAL,使宿主从持久快照继续。 -文档:[a3s-lab.github.io/Code](https://a3s-lab.github.io/Code/)(`v8.4.0`)。 -收口:[manual/HARNESS_CONVERGENCE.md](manual/HARNESS_CONVERGENCE.md) · -[manual/FIRST_PRINCIPLES_E2E.md](manual/FIRST_PRINCIPLES_E2E.md)。 +文档:[a3s-lab.github.io/Code](https://a3s-lab.github.io/Code/)(`v8.5.1`)。 ### 更早的版本线 +- **8.4** — 变薄的 `local-code` 默认、统一 `task` 扇出、仅 Active 的 Durable + Memory、`update_plan`、SDK capabilities v2。 - **8.3** — 可协商会话存储耐久、类型化工具结果信任、工作区源快照、可失败 FFI init、宿主 checkpoint 钩子。 - **8.0+** — Run 拥有的时空组合、generation-exact 能力、可移植检查点、收敛工作流。 diff --git a/website/README.md b/website/README.md index 2d0c5f69..7883ead7 100644 --- a/website/README.md +++ b/website/README.md @@ -32,15 +32,15 @@ configuration, wire behavior, or SDK surface to remain reproducible. Required parameter or function-signature breaks must use the appropriate minor or major product version rather than being hidden inside a documentation patch. -The active `v8.4.0` content lives under `docs/v8.4.0`. The `v8.3.0`, `v8.2.0`, -`v8.1.0`, `v8.0.0`, `v7.0.1`, `v6.9.0`, `v6.8.0`, `v6.7.0`, `v6.6.0`, `v6.5.2`, -`v6.5.1`, and `v6.5.0` epochs remain read-only historical snapshots; this policy -does not rewrite existing archives. The v6.9 website was published after the -package tag, so its exact source is pinned by the non-release `docs/v6.9.0` tag. -For a new minor or major line, always create a new current directory and archive -the previously supported line. Keep only supported or contract-distinct -revisions in the public selector; release tags and `CHANGELOG.md` retain the -complete patch history. +The active `v8.5.1` content lives under `docs/v8.5.1`. The `v8.4.0`, `v8.3.0`, +`v8.2.0`, `v8.1.0`, `v8.0.0`, `v7.0.1`, `v6.9.0`, `v6.8.0`, `v6.7.0`, `v6.6.0`, +`v6.5.2`, `v6.5.1`, and `v6.5.0` epochs remain read-only historical snapshots; +this policy does not rewrite existing archives. The v6.9 website was published +after the package tag, so its exact source is pinned by the non-release +`docs/v6.9.0` tag. For a new minor or major line, always create a new current +directory and archive the previously supported line. Keep only supported or +contract-distinct revisions in the public selector; release tags and +`CHANGELOG.md` retain the complete patch history. When archiving a release, list the exact revision in `multiVersion.versions` in `rspress.config.ts`. Record its tag, source tree, file count, and canonical diff --git a/website/docs/v8.5.1/en/_meta.json b/website/docs/v8.5.1/en/_meta.json new file mode 100644 index 00000000..a8792bcf --- /dev/null +++ b/website/docs/v8.5.1/en/_meta.json @@ -0,0 +1,12 @@ +[ + { + "type": "dir", + "name": "guide", + "label": "Documentation" + }, + { + "type": "dir", + "name": "api", + "label": "API" + } +] diff --git a/website/docs/v8.5.1/en/_nav.json b/website/docs/v8.5.1/en/_nav.json new file mode 100644 index 00000000..4c79e292 --- /dev/null +++ b/website/docs/v8.5.1/en/_nav.json @@ -0,0 +1,42 @@ +[ + { + "text": "Docs", + "link": "/guide/", + "activeMatch": "^/guide/(?!examples/)" + }, + { + "text": "Examples", + "link": "/guide/examples/", + "activeMatch": "^/guide/examples/" + }, + { + "text": "API", + "link": "/api/", + "activeMatch": "^/api/" + }, + { + "text": "Resources", + "items": [ + { + "text": "GitHub", + "link": "https://github.com/A3S-Lab/Code" + }, + { + "text": "Changelog", + "link": "https://github.com/A3S-Lab/Code/blob/main/CHANGELOG.md" + }, + { + "text": "Rust API", + "link": "https://docs.rs/a3s-code-core" + }, + { + "text": "npm", + "link": "https://www.npmjs.com/package/@a3s-lab/code" + }, + { + "text": "PyPI", + "link": "https://pypi.org/project/a3s-code/" + } + ] + } +] diff --git a/website/docs/v8.5.1/en/api/index.mdx b/website/docs/v8.5.1/en/api/index.mdx new file mode 100644 index 00000000..d38be98c --- /dev/null +++ b/website/docs/v8.5.1/en/api/index.mdx @@ -0,0 +1,337 @@ +--- +title: SDKs and APIs +description: Install the A3S Code Rust, Node.js, Python, and Go SDKs and find their API documentation. +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# SDKs and APIs + +A3S Code provides Rust, Node.js, Python, and Go SDKs. Install the `a3s` CLI when +you want the terminal application. Treat each registry and +[GitHub Releases](https://github.com/A3S-Lab/Code/releases) as the source of +truth for package versions and release status. + +| Entry | Package or command | Documentation | Use it for | +| -------- | ----------------------------------- | -------------------------------------------------- | -------------------------------------------------- | +| Terminal | `a3s code` | [A3S CLI](https://github.com/A3S-Lab/a3s) | Run a coding agent directly in your terminal | +| Rust | `a3s-code-core` | [docs.rs](https://docs.rs/a3s-code-core) | Use the complete runtime API or extension traits | +| Node.js | `@a3s-lab/code` | [npm](https://www.npmjs.com/package/@a3s-lab/code) | Subscribe to async events in a Node.js application | +| Python | `a3s-code` | [PyPI](https://pypi.org/project/a3s-code/) | Use synchronous or asynchronous Python APIs | +| Go | `github.com/A3S-Lab/Code/sdk/go/v8` | [Setup below](#go-module-and-bridge) | Use a pure-Go API backed by the native runtime | + +## Install + +```bash +# Rust +cargo add a3s-code-core + +# Node.js +npm install @a3s-lab/code + +# Python +python -m pip install a3s-code + +# Go +go get github.com/A3S-Lab/Code/sdk/go/v8 +``` + +## Python wheel platforms (v8.5.1) + +`a3s-code` on PyPI is a small pure-Python bootstrap. On first import it +downloads the matching native wheel from the v8.5.1 GitHub Release and checks +its SHA-256 manifest. Native wheels use the CPython 3.10 stable ABI +(`cp310-abi3`), so the same asset supports CPython 3.10 through 3.14: + +| Host | Wheel platform tag | Baseline | Bundled browser | +| ------------------- | ------------------------ | ----------- | --------------- | +| Apple Silicon macOS | `macosx_11_0_arm64` | macOS 11+ | Moli arm64 | +| Intel macOS | `macosx_12_0_x86_64` | macOS 12+ | Moli x64 | +| Linux x86_64 | `manylinux_2_28_x86_64` | glibc 2.28+ | Moli x64 | +| Linux arm64 | `manylinux_2_39_aarch64` | glibc 2.39+ | Moli arm64 | +| Windows x86_64 | `win_amd64` | Windows 10+ | Moli x64 | +| Windows arm64 | `win_arm64` | Windows 10+ | Moli arm64 | + +Each wheel contains `a3s_code/moli/` and a provenance record. +The bootstrap extracts both the native extension and sidecar into the shared +per-user cache. A process lock and atomic replacement make first import safe +when several applications start together; subsequent applications reuse the +same verified Moli installation. Linux musl is intentionally not listed: the +upstream Moli release has no musl asset, so use a system Moli executable or an +explicit Chrome/Lightpanda backend there. + +For an Intel Mac on macOS 12 or later, install with the interpreter that will +run your application: + +```bash +python3.14 -m ensurepip --upgrade # only if this interpreter has no pip +python3.14 -m pip install --upgrade pip +python3.14 -m pip install a3s-code +``` + +If `python3.14 -m pip` reports `No module named pip`, the error is in the +Python environment, before A3S Code is imported. Initialize or reinstall pip +for that interpreter and retry. The Intel wheel is `x86_64` and targets +macOS 12; it does not include the optional local ONNX embedding adapter. Keep +workspace retrieval model-free or configure an explicitly authorized remote +embedding provider on Intel. + +## Go module and bridge + +The Go 1.23+ API is pure Go and does not require CGO. One long-lived +`a3s-code-go-bridge` process owns the native runtime and carries multiplexed +requests and `EventEnvelopeV1` values over a versioned JSONL protocol. + +A Go-enabled repository release publishes a path-prefixed module tag +`sdk/go/vX.Y.Z`, matching release `vX.Y.Z`, together with +`a3s-code-go-bridge-SHA256SUMS`, standalone bridge executables, and bundles +that contain the matching Moli sidecar: + +| System | Asset target | Bundle browser | +| ------- | ----------------------------- | -------------- | +| Linux | `x86_64-unknown-linux-gnu` | Moli x64 | +| Linux | `aarch64-unknown-linux-gnu` | Moli arm64 | +| macOS | `x86_64-apple-darwin` | Moli x64 | +| macOS | `aarch64-apple-darwin` | Moli arm64 | +| Windows | `x86_64-pc-windows-msvc.exe` | Moli x64 | +| Windows | `aarch64-pc-windows-msvc.exe` | Moli arm64 | + +Download the bridge from +[GitHub Releases](https://github.com/A3S-Lab/Code/releases), verify it against +the published SHA-256 file, and keep its version equal to the Go module +version. Put it on `PATH`, set `A3S_CODE_GO_BRIDGE`, or pass +`code.WithBridgePath`: + +```bash +export A3S_CODE_GO_BRIDGE=/opt/a3s/bin/a3s-code-go-bridge +``` + +```powershell +$env:A3S_CODE_GO_BRIDGE = 'C:\a3s\a3s-code-go-bridge.exe' +``` + +For an architecture without a release asset, build it from a source checkout: + +```bash +bash .github/setup-workspace.sh +cargo build --release --package a3s-code-go-bridge --bin a3s-code-go-bridge +``` + +`code.Create` performs a fail-closed handshake for the transport protocol, +event protocol, and complete operation inventory. Go failures use stable +`*code.Error` codes while context cancellation and deadlines remain available +through `errors.Is`. The bridge covers the complete serializable Agent/Session +surface plus Go-backed hooks, budget guards, slash commands, and pipeline +callbacks. Arbitrary Rust trait-object implementations remain a Rust-native +extension mechanism; the other SDKs use their equivalent value configuration, +callback, direct-tool, or MCP boundary. + +## What the SDKs share + +All four SDKs use the same session lifecycle, event format, and snapshots. A UI +can subscribe to the same `AgentEvent` / `EventEnvelopeV1` stream and resume +saved work by session ID. + +### Priority scheduler surface + +v6.9 adds the same Agent-wide scheduler controls to every SDK. Select a +session's `urgent`, `interactive`, `foreground`, `background`, or `maintenance` +priority at creation time, then read the shared occupancy snapshot from the +Agent or any sibling session: + +| SDK | Session option | Agent / Session snapshot | +| ------- | -------------------------------------------------- | ------------------------------ | +| Rust | `SessionOptions::with_task_priority(TaskPriority)` | `task_scheduler_stats().await` | +| Node.js | `taskPriority` | `taskSchedulerStats()` | +| Python | `SessionOptions.task_priority` | `task_scheduler_stats()` | +| Go | `SessionOptions.TaskPriority` | `TaskSchedulerStats(ctx)` | + +The snapshot includes global capacity, active and pending totals, per-priority +counts, and shutdown state. See [Task scheduler](/guide/tasks#agent-wide-priority-scheduler) for +ordering, aging, cancellation, configuration, and complete examples. + +### Safe-point run-control surface + +Every SDK can steer or interrupt the currently active Run without opening a +second transcript operation: + +| SDK | Steer | Interrupt | Snapshot | +| ------- | ---------------------------- | ----------------------------------- | ----------------------------------- | +| Rust | `steer(SteerRequest).await` | `interrupt(InterruptRequest).await` | `run_control_snapshot().await` | +| Node.js | `steer(input, options)` | `interrupt(options)` | `runControlSnapshot()` | +| Python | `steer` / `steer_async` | `interrupt` / `interrupt_async` | sync / async `run_control_snapshot` | +| Go | `Steer(ctx, input, options)` | `Interrupt(ctx, options)` | `RunControlSnapshot(ctx)` | + +Requests use immutable Run IDs, optional optimistic turn guards, deadlines, +and idempotency keys. Receipts distinguish `accepted`, `applied`, `settled`, +and `rejected`; the shared `run_control_applied` event records safe-point +application. See [Sessions](/guide/sessions#safe-point-run-control) for complete +examples and lifecycle semantics. + +### Tool-result projection surface + +All four SDKs pin the same versioned deterministic projection policy to a +session: + +| SDK | Session option or builder | +| ------- | ---------------------------------------------------------------- | +| Rust | `with_tool_result_transform_policy(ToolResultTransformPolicyV1)` | +| Node.js | `toolResultTransformPolicy` | +| Python | `SessionOptions.tool_result_transform_policy` | +| Go | `SessionOptions.ToolResultTransformPolicy` | + +Rust and Python expose a `context_efficient()` preset; Node.js and Go accept +the same explicit fields. The policy persists in snapshots and every Tool +result carries `a3s.code.tool-result-evidence.v1` metadata. See +[Tools](/guide/tools#deterministic-tool-result-projection) for field values, +ordering, bounds, loss modes, and SDK examples. + +The shared guides place Go beside Node.js and Python for the complete common +SDK capability surface. Start with +[quick start](/guide/examples/quick-start), then continue to +[streaming](/guide/examples/streaming), +[direct tools](/guide/examples/direct-tools), [sessions](/guide/sessions), +[verification](/guide/verification), [MCP](/guide/mcp), and +[persistence](/guide/persistence). + +All four SDKs can configure persistence, memory, local/S3 workspaces, remote +Git, permissions and confirmation, hooks, MCP, queues, deterministic replay, +and orchestration. Rust additionally accepts arbitrary in-process trait +implementations such as a custom `LlmClient` or `ContextProvider`; other +languages integrate custom services through callbacks, direct tools, or MCP. +For UI integration, start with +[sessions and event streams](/guide/sessions). + +## Product capability discovery + +The release has one product-level capability contract. `sdkCapabilities()` (or +its language equivalent) returns the same ordered tiered inventory in every +official SDK. Each record has a stable identifier, category, canonical +operation names, a description, and a `hostOwned` flag and a `tier` (`baseline` | `advanced`). `hostOwned` identifies +who supplies policy, credentials, or an external lifecycle; it does not remove +the operation from an SDK. + +The inventory covers agent/runtime lifecycle, governed tools, code +intelligence, workspace retrieval and tools, memory and cognitive packages, +A3S Use tasks, model adapters, structured output, MCP and Skills, planning and +priority scheduling, programmable workflows, persistence, state graphs, +release/protocol contracts, web search, Moli, S3, agent serving, OpenTelemetry, +conversation, run observability, and governance. Use the inventory for feature +negotiation instead of guessing from package files or versions. + + + + +```rust +use a3s_code_core::{sdk_capabilities, sdk_capabilities_schema}; + +let capabilities = sdk_capabilities(); +assert_eq!(sdk_capabilities_schema(), "a3s-code/sdk-capabilities/v2"); +assert!(capabilities.iter().any(|item| item.id == "web_search")); +``` + + + + +```ts +import { sdkCapabilities, sdkCapabilitiesSchema } from '@a3s-lab/code'; + +const capabilities = sdkCapabilities(); +console.log(sdkCapabilitiesSchema(), capabilities.length); +``` + + + + +```python +from a3s_code import sdk_capabilities, sdk_capabilities_schema + +capabilities = sdk_capabilities() +assert sdk_capabilities_schema() == "a3s-code/sdk-capabilities/v2" +``` + + + + +```go +capabilities, err := code.SDKCapabilities(ctx) +if err != nil { + return err +} +fmt.Println(code.SDKCapabilitiesSchema(), len(capabilities)) +``` + + + + +## Moli runtime and web search + +`web_search` is backed by `a3s-search` v3.1.0 and uses Moli by default for +JavaScript-rendered engines. Runtime resolution is deterministic: an explicit +`browserPath`/`A3S_CODE_MOLI_EXECUTABLE`, a packaged sidecar, the verified +shared cache, a discoverable system Moli, and finally an HTTPS download of the +pinned release. The cache is per user and version/target scoped; an exclusive +install lock and atomic receipt prevent duplicate installations when several +`a3s-code` processes start together. Set `autoDownloadMoli: false` (or the +equivalent field) for strict offline operation. + +The diagnostics call is read-only. The ensure call may download only after the +caller has opted into the default automatic provisioning and the release +manifest/hash checks pass. Linux musl has no upstream Moli asset in v8.5.1; use +a system/explicit Moli executable or select the Chrome/Lightpanda backend there. + + + + +```rust +use a3s_code_core::{ensure_moli, moli_runtime_info, HeadlessConfig}; +use std::time::Duration; + +let config = HeadlessConfig::default(); +let status = moli_runtime_info(Some(&config)); +let executable = ensure_moli(&config, Duration::from_secs(120)).await?; +println!("{} {:?} {}", status.version, status.executable, executable.display()); +``` + + + + +```ts +import { BrowserBackend, ensureMoli, moliRuntimeInfo } from '@a3s-lab/code'; + +const status = moliRuntimeInfo({ + backend: BrowserBackend.Moli, + autoDownloadMoli: true, +}); +const executable = await ensureMoli({ backend: BrowserBackend.Moli }); +console.log(status.version, executable); +``` + + + + +```python +from a3s_code import ensure_moli_async, moli_runtime_info + +status = moli_runtime_info() +executable = await ensure_moli_async() +print(status["version"], executable) +``` + + + + +```go +status, err := code.MoliRuntimeInfo(ctx, code.NewMoliHeadlessConfig()) +if err != nil { + return err +} +executable, err := code.EnsureMoli(ctx, code.NewMoliHeadlessConfig()) +if err != nil { + return err +} +fmt.Println(status.Version, executable) +``` + + + diff --git a/website/docs/v8.5.1/en/guide/_meta.json b/website/docs/v8.5.1/en/guide/_meta.json new file mode 100644 index 00000000..aa7ef900 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/_meta.json @@ -0,0 +1,72 @@ +[ + "index", + "tui", + { + "type": "dir", + "name": "examples", + "label": "Examples", + "collapsible": true, + "collapsed": true + }, + { + "type": "section-header", + "label": "Filesystem first" + }, + "filesystem-first", + "convention-over-configuration", + "agents-md", + "filesystem-instructions", + "filesystem-config", + "agent-dir", + "filesystem-agents", + "filesystem-skills", + "filesystem-tools", + "filesystem-schedules", + { + "type": "section-header", + "label": "Runtime core" + }, + "api-contract", + "sessions", + "commands", + "tools", + "verification", + "tasks", + "teams", + "orchestration", + "skills", + { + "type": "section-header", + "label": "Governance engineering" + }, + "security", + "hooks", + "limits", + "isolation", + { + "type": "section-header", + "label": "Infrastructure" + }, + "architecture", + "lane-queue", + "workspace-backends", + "multi-machine", + "cluster-extension-points", + { + "type": "section-header", + "label": "Extensions" + }, + "providers", + "mcp", + "context", + "memory", + "persistence", + "telemetry", + { + "type": "dir", + "name": "rfcs", + "label": "RFCs", + "collapsible": true, + "collapsed": true + } +] diff --git a/website/docs/v8.5.1/en/guide/agent-dir.mdx b/website/docs/v8.5.1/en/guide/agent-dir.mdx new file mode 100644 index 00000000..f1ad7fad --- /dev/null +++ b/website/docs/v8.5.1/en/guide/agent-dir.mdx @@ -0,0 +1,196 @@ +--- +title: 'Agent Directory' +description: 'AgentDir structure, loading map, serve daemon, and durability boundaries' +--- + +# Agent Directory + +import { Tab, Tabs } from '@rspress/core/theme'; + +AgentDir is a single directory that defines a durable agent by convention. One folder contains the main agent's role, runtime config, private skills, directory-scoped tools, and recurring schedules. `AgentDir::load` reads the directory and synthesizes existing A3S Code config objects; it does not introduce a new runtime or a new prompt system. + +The deliberate design choice: `instructions.md` is injected as a prompt **slot**, not as a system-prompt override. The harness keeps `BOUNDARIES`, response-format contracts, tool visibility, the safety gate, and verification authoritative. A scheduled run is always a full harness turn through `AgentSession::send`, never a raw model call. + +The core implementation is gated behind the Rust `serve` Cargo feature. +Rust, Node.js, Python, and Go all expose the same daemon lifecycle through +their native SDK surface. + +## Structure + +```text +my-agent/ +├── instructions.md (required) Role and guidelines. Injected as a prompt slot. +├── agent.acl (optional) Model, providers, queue, and CodeConfig. +├── skills/ (optional) Private *.md skills. +├── schedules/ (optional) Cron jobs: frontmatter + body prompt. +└── tools/ (optional) Tool specs: kind: mcp or kind: script. +``` + +Only `instructions.md` is required. Everything else is optional; missing directories simply contribute no capability. + +`AgentDir::load` maps paths onto existing objects: + +| Path | Becomes | Notes | +| ----------------- | ------------------------ | ------------------------------------------------------------------------------------------------------ | +| `instructions.md` | `SystemPromptSlots.role` | Main-agent role slot; see [instructions.md](/guide/filesystem-instructions). | +| `agent.acl` | `CodeConfig` | Model, provider, queue, and directory discovery; see [agent.acl](/guide/filesystem-config). | +| `skills/` | `skill_dirs` | AgentDir-private skills; see [skills/ Skill Directory](/guide/filesystem-skills). | +| `schedules/*.md` | `Vec` | One recurring turn per file; see [schedules/ Schedule Directory](/guide/filesystem-schedules). | +| `tools/*.md` | `Vec` | One `kind: mcp` or `kind: script` tool per file; see [tools/ Tool Directory](/guide/filesystem-tools). | + +## Serve Daemon + +`serve_agent_dir` loads schedules into independent sessions and runs their cron loops until a cancellation token fires. Every fire routes the schedule prompt through `AgentSession::send`. + + + + +```rust +use a3s_code_core::config::AgentDir; +use a3s_code_core::serve::serve_agent_dir; +use a3s_code_core::{Agent, SessionOptions}; +use tokio_util::sync::CancellationToken; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let agent_dir = AgentDir::load("./my-agent")?; + let agent = Agent::from_config(agent_dir.config.clone()).await?; + let cancel = CancellationToken::new(); + let shutdown = cancel.clone(); + tokio::spawn(async move { + let _ = tokio::signal::ctrl_c().await; + shutdown.cancel(); + }); + + let options = SessionOptions::new().with_file_session_store("./sessions"); + serve_agent_dir( + &agent, + &agent_dir, + "./workspace", + Some(options), + cancel, + ) + .await?; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('./my-agent/agent.acl'); +const daemon = await agent.serveAgentDir('./my-agent', './workspace', { + sessionStore: new FileSessionStore('./sessions'), +}); + +await new Promise((resolve) => process.once('SIGINT', resolve)); +await daemon.stop(); +await agent.close(); +``` + + + + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +agent = Agent.create('./my-agent/agent.acl') +options = SessionOptions() +options.session_store = FileSessionStore('./sessions') +daemon = agent.serve_agent_dir('./my-agent', './workspace', options) + +input('Press Enter to stop the AgentDir daemon...') +daemon.stop() +agent.close() +``` + + + + +```go +package main + +import ( + "context" + "log" + "os" + "os/signal" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx, stopSignal := signal.NotifyContext(context.Background(), os.Interrupt) + defer stopSignal() + + agent, err := code.NewAgent(ctx, "./my-agent/agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + daemon, err := agent.ServeAgentDir(ctx, "./my-agent", "./workspace", &code.SessionOptions{ + FileSessionStoreDir: "./sessions", + }) + if err != nil { + log.Fatal(err) + } + + <-ctx.Done() + if err := daemon.Stop(context.Background()); err != nil { + log.Fatal(err) + } +} +``` + + + + +Direct Rust consumers must enable the feature in `Cargo.toml`: + +```toml +[dependencies] +a3s-code-core = { version = "...", features = ["serve"] } +``` + +## Overriding Session Options + +The fourth argument to Rust `serve_agent_dir`, or the `options` argument to +`serveAgentDir` / `serve_agent_dir` / `ServeAgentDir`, merges into every +schedule session. Use it to pin the model, session store, permissions, or +prompt slots. The daemon always assigns each schedule a stable `session_id` of +`schedule:`; any `session_id` in the extra options is intentionally +ignored so schedules do not collide in the same store. If prompt slots are +omitted, the AgentDir `instructions.md` slot is used. + +Graceful cancellation lets in-flight turns finish the current loop iteration. An AgentDir with no enabled schedules returns immediately. + +## Durability + +By default every daemon boot starts schedule sessions fresh. Pass a file +session store in `SessionOptions`—`with_file_session_store`, +`FileSessionStore`, or `FileSessionStoreDir`, as shown above—to resume existing +`schedule:` sessions from stored conversation history. + +Recovery restores history only. The current `instructions.md`, `skills/`, and `tools/` are re-applied on every boot, so directory edits take effect after restart even for resumed sessions. + +## Status + +| Area | State | +| ----------------------------------------- | --------------------------------------------------------------------- | +| `instructions.md`, `agent.acl`, `skills/` | Loaded and used. | +| `schedules/` + serve daemon | Implemented (`serve` feature). | +| Rehydrate on boot | Implemented with `SessionStore`. | +| `tools/` (`kind: mcp`) | Implemented; declarative MCP servers register into schedule sessions. | +| `tools/` (`kind: script`) | Implemented; sandboxed QuickJS tool over the `program` path. | + +## Notes + +- `instructions.md` is a role slot, so harness boundaries, response contracts, and verification remain authoritative. +- AgentDir is the main-agent directory. It is different from `agent_dirs` / `registerAgentDir`, which scan [agents/ Role Directory](/guide/filesystem-agents) for worker definitions. +- Keep secrets in environment variables or host secret systems, not in the directory. +- `tools/` is installed by the serve daemon per schedule session. Normal interactive sessions should use direct tools, MCP, or SDK registration. diff --git a/website/docs/v8.5.1/en/guide/agents-md.mdx b/website/docs/v8.5.1/en/guide/agents-md.mdx new file mode 100644 index 00000000..83acf42c --- /dev/null +++ b/website/docs/v8.5.1/en/guide/agents-md.mdx @@ -0,0 +1,81 @@ +--- +title: 'AGENTS.md' +description: 'Workspace-level project instructions as versioned context' +--- + +# AGENTS.md + +`AGENTS.md` is the workspace-level project instruction file. It lets project rules live with the repo, so every prompt does not need to repeat build commands, code style, safety boundaries, and release flow. + +In the filesystem-first architecture, `AGENTS.md` explains how this project works. AgentDir `instructions.md` explains who one durable agent is. Both enter context composition, but neither can override harness permission gates, response contracts, or verification requirements. + +```md +# Project Instructions + +- Use `cargo test -p a3s-code-core` for core changes. +- Never commit real secrets from `.a3s/config.acl`. +- Prefer `rg` for search. +- Release checks must include package metadata, CI, and provider verification. +``` + +## Good Content + +- Build, test, lint, format, and release commands. +- Directory responsibilities, module boundaries, and code style. +- Safety rules for secrets, permissions, external side effects, and data handling. +- Verification policy for different kinds of changes. +- Project-specific terminology and common workflows. + +## Bad Content + +- Secrets, tokens, private credentials, or personal machine paths. +- Worker-agent role descriptions; put those in `.a3s/agents/`. +- Durable-agent identity and default output style; put those in `instructions.md`. +- Reusable checklists; put those in `.a3s/skills/`. + +## Nested Rules + +A3S Code builds one instruction chain when a session starts. It finds the +nearest Git root, walks from that root to the selected workspace, and includes +at most one document from every directory. The lookup order in each directory +is: + +1. `AGENTS.override.md` +2. `AGENTS.md` +3. the ordered names in `project_doc_fallback_filenames` + +Documents are joined from root to workspace. More local guidance appears later +and therefore overrides broader guidance. An `AGENTS.override.md` replaces the +ordinary `AGENTS.md` only in its own directory; it does not discard guidance +from parent directories. If no Git root exists, A3S Code checks only the +selected workspace. + +Empty files are skipped. The combined default budget is 32 KiB and can be +changed with `project_doc_max_bytes`; zero disables project instruction loading. +A3S Code accepts regular UTF-8 files inside the project root and ignores +symlink candidates and unsafe fallback names. The effective bounded chain is +mandatory session context, so the generic retrieval budget cannot silently +drop it. + +Use nested `AGENTS.md` files only when subdirectories truly differ, such as a +desktop app, API package, or SDK with a different toolchain. Do not copy the +root file just to repeat it; duplicated rules make it harder for long-running +agents to identify the current source of truth. + +```acl +project_doc_max_bytes = 65536 +project_doc_fallback_filenames = ["TEAM_GUIDE.md", ".agents.md"] +``` + +## Relationship To Other Conventions + +```text +repo/ +├── AGENTS.md # project-level durable instructions +├── agent.acl # runtime config +└── .a3s/ + ├── agents/ # worker/subagent definitions + └── skills/ # reusable skills +``` + +`AGENTS.md` gives the agent project facts and working boundaries. `agent.acl` tells the runtime how to connect models and directories. `agents/` and `skills/` provide discoverable roles and reusable process. diff --git a/website/docs/v8.5.1/en/guide/api-contract.mdx b/website/docs/v8.5.1/en/guide/api-contract.mdx new file mode 100644 index 00000000..b4554413 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/api-contract.mdx @@ -0,0 +1,873 @@ +--- +title: 'API Contract' +description: 'SDK mechanisms covered by the local integration check' +--- + +# API Contract + +This page documents only the A3S Code Node SDK behavior covered by +`scripts/docs_api_contract_smoke.mjs`. The script starts a temporary +OpenAI-compatible local test server, creates real SDK sessions, calls the native +binding, and asserts the returned values. It does not require a docs build. + +Run the contract from the repository root: + +```bash +node scripts/docs_api_contract_smoke.mjs +``` + +## Agent + +Verified entry points: + +```ts +const agent = await Agent.create(aclSource); +await agent.refreshMcpTools(); + +const session = agent.session(workspace, options); +const named = agent.sessionForAgent(workspace, 'explore', [], options); +``` + +`Agent.create()` accepts ACL source text or an `.acl` file path. JSON config is +not part of the verified contract. The integration check covers both +`apiKey`/`baseUrl` and `api_key`/`base_url` provider aliases against the local +OpenAI-compatible test server. +`sessionForAgent()` was verified with the built-in `explore` agent. + +The Node factory names remain synchronous at the JavaScript surface, but their +native implementation delegates resource resolution to the core async session +construction path. Rust embedders should use `SessionBuilder::build().await` or +the async factories; the synchronous Rust compatibility factory requires an +explicit pre-initialized memory store. + +The integration check also covers the ACL field that creates a file-backed +session store: + +```acl +storage_backend = "file" +sessions_dir = "/tmp/a3s-doc-stores/acl-storage" +``` + +This contract does not cover `storage_url`. It is not the local file-session +persistence path; use `sessions_dir` in ACL or pass `sessionStore` in SDK +options. + +## Session Options + +The integration check covers this option shape: + +```ts +const session = agent.session(workspace, { + model: 'openai/docs-alt', + builtinSkills: true, + planningMode: 'disabled', + memoryStore: new FileMemoryStore(memoryDir), + sessionStore: new FileSessionStore(sessionDir), + sessionId: 'docs-contract', + autoSave: true, + securityProvider: new DefaultSecurityProvider(), + skillDirs: [path.join(workspace, 'skills')], + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: 'Always separate blockers from nice-to-have improvements.', + }, + ], + maxToolRounds: 24, + maxParseRetries: 3, + toolTimeoutMs: 120000, + circuitBreakerThreshold: 4, + autoCompact: true, + autoCompactThreshold: 0.75, + continuationEnabled: true, + maxContinuationTurns: 3, + maxExecutionTimeMs: 300000, // 5 minutes timeout + confirmationPolicy: { + enabled: true, + defaultTimeoutMs: 60000, + timeoutAction: 'reject', + }, +}); +``` + +`model` is a per-session override. The check verifies that a session created +with `model: 'openai/docs-alt'` sends `docs-alt` to the local provider. + +Basic session accessors are verified: + +```ts +console.log(session.sessionId); +console.log(session.workspace); +console.log(session.initWarning); +console.log(session.history()); +console.log(session.cancel()); +``` + +`workspace` is returned as the SDK's canonical workspace path. + +`planningMode` accepts the explicit modes documented elsewhere: `'auto'`, +`'enabled'`, and `'disabled'`. + +The check also verifies that this `permissionPolicy` shape is accepted at +session creation: + +```ts +agent.session(workspace, { + permissionPolicy: { + deny: ['write(**/.env*)', 'bash(rm -rf*)'], + ask: ['bash(git push*)', 'bash(npm publish*)'], + allow: ['read(*)', 'search(*)', 'bash(npm run build*)'], + defaultDecision: 'ask', + enabled: true, + }, +}); +``` + +Prompt slot options are strings: + +```ts +agent.session(workspace, { + role: 'release-readiness reviewer', + guidelines: + 'Find blockers before improvements. Require command evidence for done claims.', + responseStyle: 'concise, findings first', + goalTracking: true, +}); +``` + +## Result Shape + +`session.send()` returns `AgentResult` fields on the result object itself: + +```ts +const result = await session.send('Return a short answer'); + +console.log(result.text); +console.log(result.toolCallsCount); +console.log(result.promptTokens); +console.log(result.completionTokens); +console.log(result.totalTokens); +console.log(result.verificationStatus); +console.log(result.pendingVerificationCount); +console.log(result.failedVerificationCount); +console.log(result.verificationReportCount); +console.log(result.verificationSummaryJson); +console.log(result.verificationSummaryText); +``` + +Trace events and verification reports are session APIs, not fields on +`AgentResult`. + +## Streaming + +`session.stream()` returns `EventStream`. The verified consumption contract is +`.next()`: + +```ts +const stream = await session.stream('Stream one sentence'); + +while (true) { + const { value: event, done } = await stream.next(); + if (done) break; + if (!event) continue; + if (event.text) process.stdout.write(event.text); +} +``` + +The smoke check starts another `send()` immediately after this loop. Exhaustion +therefore verifies both event delivery and release of the stream's single-flight +admission lease; no retry delay is required. + +Each SDK event is an envelope-v1 projection with `version === 1`, an open +`type` string, a complete `payload`, and optional `metadata`. Consumers must +retain a default branch for future event types. Node exposes `payloadJson` and +`metadataJson` string views; Python exposes `payload_json` and `metadata_json` +and retains `event_type` as an alias for `type`. + +Do not depend on `for await` unless your installed SDK version has separately +validated async iteration support. + +## Direct Tools + +> Full guide: [Tools](/guide/tools). + +The integration check covers these host-driven direct calls: + +```ts +await session.readFile('README.md'); +await session.glob('src/*.rs'); +await session.grep('PermissionPolicy'); +await session.bash('printf docs-bash'); +await session.tool('read', { file_path: 'README.md' }); +await session.git('status'); +await session.git('diff'); +await session.git( + 'log', + undefined, + undefined, + undefined, + undefined, + undefined, + undefined, + 5, +); +await session.tool('search_skills', { query: 'release blockers', limit: 5 }); + +session.toolNames(); +session.toolDefinitions(); +session.registerAgentDir(path.join(workspace, 'agents')); +``` + +The verified local-workspace `toolNames()` set includes `read`, `write`, `edit`, +`patch`, `download`, `search`, `ls`, `bash`, `task`, `search_skills`, `Skill`, +`program`, `git`, `batch`, `web_fetch`, and `web_search`. + +Direct host calls are privileged. Gate them in the host application before +exposing them to end users. + +`download` is invoked through the generic direct-tool API: + +```ts +const result = await session.tool('download', { + url: 'https://example.com/archive.tar.zst', + file_path: 'artifacts/archive.tar.zst', + overwrite: false, + connections: 4, + max_bytes: 536870912, + timeout: 300, + expected_sha256: + '0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef', +}); +``` + +Only `url` is required. `connections` is limited to 1–4, `max_bytes` defaults +to 512 MiB with an 8 GiB maximum, `timeout` defaults to 300 seconds with a +3600-second maximum, and `expected_sha256` must contain exactly 64 hexadecimal +characters. `file_path` is workspace-relative and may be omitted for safe +filename inference; `overwrite` defaults to `false`. + +The tool exists only for writable local workspaces. Model-driven calls remain +permission- and HITL-governed workspace mutations. The transfer preserves +signed query parameters for requests but removes them from result metadata, +validates every redirect and direct DNS target against SSRF, verifies strict +Range responses, retries only within fixed bounds, falls back sequentially +when needed, and promotes an adjacent temporary file only after completion and +optional digest verification. + +## AGENTS.md + +The script writes an `AGENTS.md` file in the workspace and asserts that its +instruction token appears in the local provider request body: + +```md +# Project Instructions + +Always mention docs-contract-agents-md-token when asked for project instructions. +``` + +Keep project instructions operational and free of secrets. + +## Programmatic Tool Calling + +`session.program()` runs bounded JavaScript in the embedded QuickJS runtime: + +```ts +const result = await session.program({ + source: ` + export default async function run(ctx, inputs) { + const text = await ctx.readFile('README.md'); + const hits = await ctx.grep(inputs.q, { glob: '*.md' }); + return { summary: 'ok', hasHits: text.includes(inputs.q) && hits.includes(inputs.q) }; + } + `, + inputs: { q: 'planningMode' }, + allowedTools: ['read', 'grep'], + limits: { timeoutMs: 30000, maxToolCalls: 12, maxOutputBytes: 65536 }, +}); + +const meta = JSON.parse(result.metadataJson); +console.log(meta.script_result); +console.log(meta.program.tool_calls); +``` + +Verified `ctx` helpers: `readFile`, `read`, `grep`, `glob`, `ls`, `bash`, +`git`, and generic `tool(name, args)`. `allowedTools` limits which registered +tools the script may call. The `program` tool is not included in its own default +tool set. + +## Verification + +> Full guide: [Verification](/guide/verification). + +Verification is session-scoped: + +```ts +const report = await session.verifyCommands('docs api check', [ + { + id: 'echo', + kind: 'command', + description: 'echo works', + command: 'printf verify', + required: true, + }, +]); + +console.log(report.subject); +console.log(session.verificationReports()); +console.log(session.verificationSummary()); +console.log(session.verificationSummaryText()); +console.log(session.verificationPresets()); +console.log(formatVerificationSummary(session.verificationSummary())); +``` + +## Memory + +> Full guide: [Memory](/guide/memory). + +Node memory was verified with `FileMemoryStore`: + +```ts +const session = agent.session(workspace, { + memoryStore: new FileMemoryStore(memoryDir), +}); + +console.log(session.hasMemory); +await session.rememberSuccess('docs memory success', ['grep'], 'remembered'); +await session.rememberFailure('docs memory failure', 'expected failure', [ + 'bash', +]); +await session.memoryRecent(10); +await session.recallSimilar('docs memory', 5); +await session.recallByTags(['grep'], 10); +``` + +The verified recent-memory method is `memoryRecent()`. `recallRecent()` is not +present on the current Node SDK surface. + +## Skills + +> Full guide: [Skills](/guide/skills). + +File-backed and inline skills are verified through `search_skills`: + +```ts +const session = agent.session(workspace, { + skillDirs: [path.join(workspace, 'skills')], + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: 'Always separate blockers from nice-to-have improvements.', + }, + ], +}); + +await session.tool('search_skills', { query: 'release blockers', limit: 5 }); +await session.tool('search_skills', { + query: 'strict release review', + limit: 5, +}); +``` + +The skill-file check uses Markdown with YAML frontmatter and the +`allowed-tools` key. + +## Side Questions + +> Full guide: [Sessions](/guide/sessions). + +The current SDK surface has no dedicated ephemeral-question helper. Use +explicit history when the call must not mutate the session transcript: + +```ts +const snapshot = session.history(); +const side = await session.send('What is this test?', snapshot); + +console.log(side.text); +console.log(session.history().length === snapshot.length); +``` + +## Runs And Cancellation + +> Full guide: [Sessions](/guide/sessions). + +Each `send()` or `stream()` records replayable run state: + +```ts +const runs = await session.runs(); +const latest = runs.at(-1); + +if (latest) { + console.log(await session.runSnapshot(latest.id)); + console.log(await session.runEvents(latest.id)); +} + +const current = await session.currentRun(); +if (current?.id && current.status === 'running') { + await session.cancelRun(current.id); +} + +console.log(session.traceEvents()); +``` + +Headless hosts can admit exact immutable run identities and receive the +authoritative snapshot without waiting for completion: + +```ts +const admitted = await session.spawnRunWithId( + 'release-42/run-7', + 'Verify the release', +); +console.log(admitted.snapshot.id, admitted.replayed); + +const recovered = await session.spawnRecoveryWithRunId( + 'checkpoint-run-6', + 'release-42/recovery-7', +); +console.log(recovered.snapshot.status, recovered.replayed); +``` + +Repeating compatible immutable input returns `replayed: true` without starting +duplicate work. Conflicting input returns `RUN_IDENTITY_CONFLICT`; closing the +session cancels detached workers. + +`currentRun()` is for the current operation. When idle, it may return +`null` or a retained snapshot depending on the preceding control flow. Use +`runs()` for completed history. + +Transcript-affecting operations are single-flight per session. An overlapping +send, stream, attachment call, slash command, or `resumeRun` fails immediately +with `SessionBusy`; it is not queued. A stream retains admission until its +producer has stopped, even when its public handle is dropped. + +## Persistence + +> Full guide: [Persistence](/guide/persistence) and [Sessions](/guide/sessions). + +File-backed session persistence was verified with stable `sessionId`, +`autoSave`, explicit `save()`, and `resumeSession()`: + +```ts +const session = agent.session(workspace, { + sessionStore: new FileSessionStore(sessionDir), + sessionId: 'docs-contract', + autoSave: true, +}); + +await session.save(); + +const resumed = agent.resumeSession('docs-contract', { + sessionStore: new FileSessionStore(sessionDir), +}); +console.log(resumed.history()); +``` + +Core persistence commits the conversation and its artifacts, traces, run +records, verification reports, and subagent task snapshots together as one +versioned `SessionSnapshotV1`. File and memory stores publish that aggregate +atomically. Legacy fragmented records remain loadable, while custom stores must +implement aggregate save explicitly. + +Use `session.close()` when a Node process should release session-scoped +background resources promptly. `close()` is the full graceful-stop entry +point: it flips `session.isClosed` to `true` (further `send` / `stream` +calls reject with a `Session closed` error), fires the session-level +`CancellationToken` so every in-flight run, delegated subagent task, and +HITL confirmation aborts. Subsequent `close()` calls are no-ops. + +For control-plane callers that only know the session ID, the same cleanup +is reachable from the agent: + +```ts +await agent.listSessions(); // ['session-a', 'session-b'] +await agent.closeSession('session-a'); // true if it was open +await agent.close(); // close every live session + disconnect global MCP +``` + +After `agent.close()`, subsequent `agent.session(...)` and +`agent.resumeSession(...)` calls reject with a `Session closed` error. +Idempotent. Use this in process-shutdown handlers to guarantee no +session-scoped workers outlive the agent. + +## Delegation + +> Full guide: [Tasks](/guide/tasks) and [Orchestration](/guide/orchestration). + +The direct helpers for the core delegation tools were verified: + +```ts +await session.task({ + agent: 'general', + description: 'docs delegated check', + prompt: 'Return a short response.', + maxSteps: 1, +}); + +await session.tasks([ + { + agent: 'general', + description: 'one', + prompt: 'Return one response.', + maxSteps: 1, + }, + { + agent: 'general', + description: 'two', + prompt: 'Return another response.', + maxSteps: 1, + }, +]); +``` + +Both helpers return `ToolResult` values from the model-visible `task` tool. + +## Hooks + +> Full guide: [Hooks](/guide/hooks). + +The verified hook management surface is: + +```ts +session.registerHook( + 'docs-observer', + 'pre_tool_use', + { tool: 'bash' }, + { priority: 1, timeoutMs: 1000 }, + () => ({ action: 'continue' }), +); + +console.log(session.hookCount()); +session.unregisterHook('docs-observer'); +``` + +Validate the specific event path you depend on before using hook behavior as a +production enforcement gate. + +## Slash Commands + +> Full guide: [Commands](/guide/commands). + +Custom slash commands are invoked through `session.send()`: + +```ts +session.registerCommand( + 'docs_status', + 'Return docs command status', + (args, ctx) => { + return `status args=${args}; session=${ctx.sessionId}; workspace=${ctx.workspace}`; + }, +); + +console.log(session.listCommands()); +const result = await session.send('/docs_status check'); +console.log(result.text); +``` + +## Lane Queue + +> Full guide: [Lane Queue](/guide/lane-queue). + +Queue infrastructure is opt-in: + +```ts +const queued = agent.session(workspace, { + queueConfig: { enableDlq: true, enableMetrics: true }, +}); + +console.log(queued.hasQueue()); +await queued.setLaneHandler('execute', { mode: 'external', timeoutMs: 1000 }); +await queued.pendingExternalTasks(); +await queued.completeExternalTask('missing', { + success: true, + result: { ok: true }, +}); +await queued.queueStats(); +await queued.queueMetrics(); +await queued.deadLetters(); +``` + +Ordinary sessions are queue-free unless `queueConfig` is provided. + +## MCP + +> Full guide: [MCP](/guide/mcp). Idle disconnect: [Cluster Extension Points](/guide/cluster-extension-points). + +The integration check covers a live stdio MCP server: + +```ts +const count = await session.addMcp({ + name: 'echo', + transport: { + type: 'stdio', + command: process.execPath, + args: ['tools/mcp_echo_server.mjs', 'example-value'], + }, + timeoutMs: 30000, +}); + +console.log(count); +console.log(await session.mcpStatus()); +console.log( + session.toolNames().filter((name) => name.startsWith('mcp__echo__')), +); + +await session.tool('mcp__echo__echo', { message: 'docs mcp ok' }); +await session.removeMcpServer('echo'); +``` + +Tools from the server are named `mcp____`. +`addMcpServer(...)` and `addMcpServerConfig(...)` remain compatibility aliases; +new examples use the compact object-shaped `addMcp(...)` API. + +Live add/remove operations target a private manager owned by this session. +Agent-global and host-supplied managers are inherited read-only capability +sources, so a session cannot mutate a sibling or global MCP configuration. + +## Cluster-grade extension points + +> Full guide: [Cluster Extension Points](/guide/cluster-extension-points) (identity labels, budget guard, cluster events, deterministic IDs/replay, loop checkpoints, retention caps). + +These contracts let a cluster control plane wire +multi-tenancy, cost governance, and crash-tolerant runs **without +forking the framework**. The framework defines decision points and +emits structured events; the host supplies the policy implementations. + +### Identity labels + +Four optional `SessionOptions` slots are propagated through hooks, +traces, and `SessionData` but never interpreted by the framework: + +```ts +const session = agent.session(workspace, { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', + sessionStore: new FileSessionStore('./sessions'), +}); +session.tenantId; // -> 'tenant-example' +session.correlationId; // -> 'trace-example' +``` + +`apply_persisted_runtime_options` restores them on resume; caller- +supplied options on resume take precedence so you can relabel. + +### Budget / cost guard + +`BudgetGuard` is consulted before every run-owned provider call (and after a +successful response for usage accounting) and before every governed tool call, +including nested and trusted host-direct calls. `Deny` returns +`CodeError::BudgetExhausted { resource, reason }`; `SoftLimit` emits +an `AgentEvent::BudgetThresholdHit { kind: "soft", .. }` and proceeds. + +Rust hosts inject the trait directly. Node.js, Python, and Go expose callback +bridges later in this section: + +```rust +let guard: Arc = /* host-supplied impl */; +let opts = SessionOptions::new().with_budget_guard(guard); +``` + +### Cluster event vocabulary + +`AgentEvent` (non-exhaustive) carries platform-level events the host +emits via `HookExecutor`: + +- `BudgetThresholdHit { resource, kind, consumed, limit, message? }` +- `PassivationRequested { reason, deadline_ms? }` +- `PeerInvocation { from_session_id, from_tenant_id?, correlation_id? }` + +In-session hooks subscribe to these to react uniformly regardless of +how the host's transport delivers them. + +### Deterministic IDs / time + +`HostEnv { id_generator, clock }` replaces the default +`uuid::Uuid::new_v4()` + wall-clock pair. Replay tooling configures +`SequentialIdGenerator` + `FixedClock` to recreate a run bit-identical +on another node. + +### Loop checkpoints + run resumption + +When a `SessionStore` is configured, the agent loop persists a +`LoopCheckpoint` after each completed tool round, keyed by `run_id`. +Any node holding the same store can rehydrate a run from its last +boundary: + +```ts +// Node — host detected node A died mid-run; on node B: +const session = agentB.session(workspace, { + sessionStore: new FileSessionStore('./sessions'), + sessionId: 'session-from-node-a', +}); +const result = await session.resumeRun('run-id-from-node-a'); +``` + +```python +# Python equivalent +opts = SessionOptions() +opts.session_store = FileSessionStore('./sessions') +opts.session_id = 'session-from-node-a' +session = agent_b.session(workspace, opts) +result = session.resume_run('run-id-from-node-a') +``` + +```go +// Go equivalent +session, err := agent.Session(ctx, workspace, &code.SessionOptions{ + FileSessionStoreDir: "./.a3s/sessions", + SessionID: "session-from-node-a", +}) +if err != nil { + return err +} +result, err := session.ResumeRun(ctx, "run-id-from-node-a") +``` + +A **new** run id is allocated for the resumed work — the framework +does not pretend the old run continues. Two distinguishable error +paths: + +- `"resume_run requires a session_store"` — host should fall back to + a fresh session. +- `"no loop checkpoint found for run 'X'"` — host can retry later + (race against checkpoint write) or treat the run as lost. + +Boundary policy: checkpoints are taken **only between tool rounds**, +never mid-tool. If a process dies mid-tool the work of that round is +lost; the LLM re-deliberates from the previous boundary. This trades +retry cost for correctness — re-executing a non-idempotent tool +across the boundary is worse than re-asking the LLM. + +### Retention caps for long-running sessions + +`SessionRetentionLimits` lets the host cap the four in-memory stores +that grow with session age: the run records, per-run event buffers, +trace events, and **terminal** subagent task snapshots. Each cap is +optional (omission keeps the finite framework default; `unbounded: true` +deliberately restores unlimited retention). Eviction is +strict FIFO; running subagent tasks are never dropped. + +```rust +use a3s_code_core::retention::SessionRetentionLimits; + +let limits = SessionRetentionLimits::new() + .with_max_runs(100) + .with_max_events_per_run(5_000) + .with_max_trace_events(10_000) + .with_max_terminal_subagent_tasks(1_000); + +let opts = SessionOptions::new().with_retention_limits(limits); +``` + +The host should pick caps from the same observability budget that caps the rest +of its in-memory state (Prometheus carries history anyway). Node exposes this +as `retentionLimits`; Python as `opts.retention_limits`; Go as +`SessionOptions.RetentionLimits`. + +### MCP idle disconnect + +`Agent::disconnect_idle_mcp(threshold_ms)` walks the connected MCP +servers and drops any whose last activity is older than +`now - threshold_ms`. The server's _registered config_ stays — a +later tool call will reconnect on demand. Returns the names of +disconnected servers. + +```ts +// Node — periodically reap quiet MCP subprocesses. +setInterval(async () => { + const dropped = await agent.disconnectIdleMcp(5 * 60 * 1000); // 5 min + if (dropped.length) { + console.log('reaped idle MCP servers:', dropped); + } +}, 60_000); +``` + +```python +# Python — same shape. +dropped = agent.disconnect_idle_mcp(5 * 60 * 1000) +``` + +```go +// Go — milliseconds, matching the other SDKs. +dropped, err := agent.DisconnectIdleMCP(ctx, 5*60*1000) +``` + +Activity is stamped on `connect` and on every successful `call_tool`. +Hosts that route tool traffic through a side channel can call +`McpManager.touch(name)` to manually keep a server warm. + +### BudgetGuard SDK bridges + +All callback-capable SDKs accept the same decision shape: + +| Return | Effect | +| -------------------------------------------------------- | -------------------------------------------------------------------- | +| `None` / `null` / `{decision:'allow'}` | proceed silently | +| `{decision:'soft', resource, consumed, limit, message?}` | emit `BudgetThresholdHit('soft')` event, proceed | +| `{decision:'deny', resource, reason}` | abort the call, throw `RuntimeError("Budget exhausted...")` (Python) | +| | / reject with `"Budget exhausted..."` (Node) | + +Missing methods on the guard object are treated as a permissive default +(Allow / no-op). Python callback errors fall back to Allow. Go callback errors, +timeouts, and malformed decisions fail closed as Deny. + +```python +# Python — attach via SessionOptions before agent.session(...) +class MyGuard: + def check_before_llm(self, session_id, estimated_tokens): + return {"decision": "deny", "resource": "llm_tokens", "reason": "cap"} + def record_after_llm(self, session_id, usage): + track(session_id, usage["total_tokens"]) + +opts = SessionOptions() +opts.budget_guard = MyGuard() +session = agent.session(workspace, opts) +``` + +```ts +// Node — attach via session.setBudgetGuard after construction. +// JsFunction values can't live inside the value-typed SessionOptions, +// so the guard is installed on the Session itself; takes effect on +// the next send/stream. +session.setBudgetGuard({ + checkBeforeLlm: (ctx) => { + if (overBudget(ctx.sessionId)) { + return { decision: 'deny', resource: 'llm_tokens', reason: 'cap' }; + } + return null; + }, + recordAfterLlm: (ctx) => { + track(ctx.sessionId, ctx.usage.totalTokens); + }, +}); +``` + +```go +err := session.SetBudgetGuard(ctx, &code.BudgetGuardHandlers{ + CheckBeforeLLM: func( + _ context.Context, + call code.BudgetLLMContext, + ) (*code.BudgetDecision, error) { + if overBudget(call.SessionID) { + return &code.BudgetDecision{ + Decision: "deny", + Resource: "llm_tokens", + Reason: "cap", + }, nil + } + return &code.BudgetDecision{Decision: "allow"}, nil + }, +}) +``` + +Node callbacks receive a single `ctx` object and must not throw; wrap handler +logic in `try/catch` and return an explicit decision. Hung or unreadable +`check*` callbacks fail closed as `deny`. Python callbacks use positional +arguments and catch exceptions as Allow. Go handlers receive typed contexts; +callback errors and timeouts fail closed. + +Pass `null` to `setBudgetGuard` (Node) or set `opts.budget_guard = +None` and re-create the session (Python) to clear. Call +`session.SetBudgetGuard(ctx, nil)` in Go. diff --git a/website/docs/v8.5.1/en/guide/architecture.mdx b/website/docs/v8.5.1/en/guide/architecture.mdx new file mode 100644 index 00000000..7ecb221f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/architecture.mdx @@ -0,0 +1,240 @@ +--- +title: 'Architecture' +description: 'Session construction, run scope, governed invocation, events, and persistence' +--- + +# Architecture + +A3S Code separates configuration resolution, per-run execution, stable wire +contracts, and persistence. The TUI and SDKs use the same runtime kernel; they do +not get separate execution paths. + +```text +CodeConfig + SessionOptions + -> validate and resolve async resources + -> ResolvedSessionConfig + -> AgentSession + -> single-flight run admission + -> safe-point run-control inbox + -> InvocationContext + ├─ LLM invoker -> provider calls + ├─ tool invoker -> model, nested, delegated, and host-direct calls + └─ events -> EventEnvelopeV1 -> Rust / Node / Python / Go consumers + -> SessionSnapshotV1 -> atomic store generation +``` + +## Async-First Session Construction + +`SessionOptions` is a public patch, not partially initialized runtime state. The +async construction path merges it with `CodeConfig`, validates conflicting +options, initializes async resources, and produces one internal +`ResolvedSessionConfig`. Session assembly consumes that resolved value instead +of resolving the same choice in multiple layers. + +For Rust hosts, prefer: + +```rust +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +`Agent::session_async`, `resume_session_async`, `session_for_agent_async`, and +`session_for_worker_async` use the same construction kernel. File-backed memory +and session stores, queues, trajectory recording, and session MCP discovery are +initialized asynchronously and return typed `SessionConfiguration` or +`SessionInitialization` errors with the failing resource. + +The synchronous `Agent::session` method exists only for compatibility with +hosts that explicitly supply a pre-initialized memory store and have already +initialized every other resource. It never starts or blocks a Tokio runtime. +Configuration that requires async work fails with +`CodeError::AsyncSessionBuildRequired`; the runtime does not silently replace +the requested resource with an easier backend. A manager passed through +`SessionOptions::with_mcp` always requires async capability discovery. The sync +path can only inherit agent-global MCP tools already cached at agent startup. + +## Single-Flight Conversation State + +Conversation history is serialized by admission, not by optimistic locking. +Only one transcript-affecting operation may be active on a session. This covers +`send`, `stream`, both attachment variants, slash commands, and `resume_run`. +An overlap fails immediately with `CodeError::SessionBusy` before history is +read or a command is dispatched. + +A stream retains its admission lease until the stream runtime finishes. Dropping +or aborting the public handle does not briefly admit a second operation while +the original producer is still writing events or history. Direct host tool +calls are control-plane operations and do not claim the conversation lease. + +## Invocation Context + +Each admitted run creates one immutable `InvocationContext` containing: + +- run ID and session ID +- the run cancellation token +- the event sender +- the governance snapshot, including the active budget guard + +That context is the source of truth for provider and tool work. It installs the +same cancellation token and session identity into `ToolContext`, so cancellation +reaches queued tools, nested `batch`/`program` calls, delegated work, planning, +structured-output repair, compaction, and other run-owned helper calls. + +## Layered Prompt And Run Continuity + +The default system prompt is assembled as a small set of non-overlapping +contracts. The shared runtime contract defines authority, autonomy, evidence, +continuity, and stop semantics; the execution-style prompt adds only the +behavior needed for that mode; the response-format fragment describes the +requested output. Filesystem instructions, Skills, and host context remain +separate inputs, so changing the default prompt does not shadow runtime policy +or remove tool capability. + +Each active Run owns a bounded control inbox. `steer` appends a new user +direction to the same transcript at the next provider or tool safe point; +`interrupt` requests cooperative cancellation and lets normal cleanup, +checkpointing, hooks, and event persistence finish. Idempotency keys, immutable +Run IDs, optional turn revisions, and deadlines prevent retries or stale UI +state from affecting a newer turn. Compaction and recovery retain the objective, +constraints, acceptance criteria, decisions, and verification evidence needed +to continue without silently changing scope. + +## Scoped Capability Composition + +A Session publishes Tool, Skill, Agent, Command, Hook, MCP, Context, Flow, +Knowledge, and UI contributions as one immutable capability-catalog generation. +Preparation follows the bounded dependency graph, validation covers the complete +projection, and one compare-and-swap makes the generation visible. A failed or +cancelled preparation rolls back completed effects without exposing a partial +catalog. + +Every admitted Run pins one projection and, when the projection came from A3S +Use, acquires the matching non-clone Use snapshot lease. Turns and Subtasks are +typed child scopes: they share the Run's structural generation and may narrow +authority, but they cannot rediscover a newer Session catalog. Run shutdown +settles supervised work and reversible effects before releasing the exact +generation lease. + +`RunCapabilityBindingV1` records the Code generation, catalog digest, complete +authority-ceiling digest, and optional Use cursor in the Run and its logical +checkpoints. Recovery therefore rejects N/N+1 drift before admitting the target +Run, including a cutover that occurs during preparation. A fresh unpublished +Session may bootstrap exactly one complete historical batch; a Session that has +already published capabilities cannot rewind to an older generation. + +## LLM Invocation Boundary + +Run-owned provider work uses one scoped LLM invoker. It checks budget and +cancellation before each provider call, records usage after each successful +response, proxies streaming completion so terminal usage is recorded, and +combines caller and run cancellation. Normal turns, planning, structured output +and repair, compaction, and memory/helper paths use this boundary instead of +maintaining independent budget logic. + +A hard budget denial is an error and is never converted into an ungoverned +fallback. Soft limits emit a `budget_threshold_hit` event and allow the call to +continue. + +## Tool Invocation Boundary + +The tool invoker is the governance kernel for model-selected, nested, +programmatic, delegated, and direct host calls. For model-owned work it applies +active-skill restrictions, permission policy, pre/post hooks, budget checks, +human confirmation, queue/timeout handling, cancellation, recursive-invocation +protection, and output sanitization. `batch` and `program` receive the scoped +invoker rather than calling a raw registry, so inner calls cannot escape those +checks. + +Direct SDK helpers have an explicit `HostDirectPolicy::TrustedControlPlane` +origin. +The host is the authority that chose the operation, so model-facing permission +and confirmation decisions are skipped. Pre-hooks can still block the call, and +budget, queue/timeout, cancellation, recursion protection, post-hooks, and +output sanitization remain active. Applications must authorize end users before +exposing this privileged path. + +Trust propagation is structural rather than ambient. Only built-in +control-plane orchestrators can convert an explicit host-direct call into the +internal trusted-nested origin for host-selected children. Public +`InvocationRuntime::invoke_tool` always creates an ordinary governed nested +call, even when the extension itself was called directly. Model dispatch also +removes inherited host-direct policy before it creates a `ToolContext`, so a +Skill, Task, or other model sub-run cannot use `batch` or `program` to amplify +its parent's authority. + +## Stable Event Protocol + +`AgentEvent` is the internal Rust runtime enum. The cross-language contract is +the lossless envelope: + +```json +{ + "version": 1, + "type": "tool_end", + "payload": {}, + "metadata": {} +} +``` + +The v1 event catalog and the exhaustive Rust mapping share one source of truth, +so adding a runtime variant without a canonical wire name is a compile-time +failure. Node and Python consume the centralized projection for convenience +fields such as `text`, `toolName`, and `tool_name`. `type` remains an open +string: unknown future types preserve their full payload and metadata rather +than collapsing to an `unknown` sentinel. + +## Atomic Session Persistence + +`SessionSnapshotV1` is one versioned persistence generation. It contains the +conversation plus artifacts, trace events, run records, verification reports, +and delegated-task snapshots. `session.save()` materializes that aggregate and +calls `SessionStore::save_snapshot` once. + +The file store writes one complete JSON envelope through a synced temporary +file and atomic replacement. The memory store replaces one aggregate entry +under one lock. Both advertise atomic snapshot capability. Historical bare +`SessionData` and fragment directories remain loadable for migration, but new +saves do not publish fragmented generations. A custom store must implement +aggregate save explicitly; the default method returns an error rather than +acknowledging a partial or no-op save. + +## MCP Ownership And Isolation + +MCP managers have explicit ownership: + +1. The agent-global manager owns servers loaded from global configuration. +2. A host-supplied manager in session options is an inherited, read-only + capability source. +3. Every session owns a new private live manager. + +Capabilities are assembled in that order, so a session-local tool can shadow an +inherited tool only inside that session. Live `add_mcp_server` and +`remove_mcp_server` calls mutate only the private manager; removing a local +shadow reveals the inherited capability again. Sibling sessions cannot mutate +one another, and delegated children inherit the ordered manager sources needed +to call the same tools without transferring ownership. + +A local stdio transport also owns the complete server process lifetime. It +starts the server as a Unix process-group leader, drains protocol output and +stderr separately, and terminates the group on close or drop. Outstanding +requests fail when either pipe closes instead of waiting for an unrelated +request deadline. + +## Programmatic Tool Calling + +The `program` tool runs JavaScript in an embedded QuickJS VM with a controlled +`ctx` object. The VM has no direct filesystem, network, subprocess, or +environment access. Its useful capabilities are tool calls routed back through +the scoped tool invoker. Use ordinary model tool calls when the next action +requires judgment; use a bounded program when the action sequence is already +known and only its result needs model interpretation. + +## Extension Points + +Use typed session options, skills, agent definitions, hooks, MCP servers, +memory/session stores, security providers, queue configuration, and workspace +services to adapt the runtime. Prefer explicit policy and replayable evidence +over alternate execution paths. diff --git a/website/docs/v8.5.1/en/guide/cluster-extension-points.mdx b/website/docs/v8.5.1/en/guide/cluster-extension-points.mdx new file mode 100644 index 00000000..479dc59e --- /dev/null +++ b/website/docs/v8.5.1/en/guide/cluster-extension-points.mdx @@ -0,0 +1,601 @@ +--- +title: 'Cluster Extension Points' +description: 'The seams a cluster host uses to run long-lived agent sessions across many nodes without forking the framework.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Cluster Extension Points + +A cluster host platform runs long-lived agent sessions across many nodes. The framework does not ship a scheduler or a placement engine. Instead it exposes a small set of seams: it defines the decision points, emits structured events, and lets the host supply the policy. Everything below is something you wire from outside the framework — you never fork it. + +This page shows the equivalent native seams in Rust, Node.js, Python, and Go. +The configuration shape follows each language's conventions, while the +underlying capability and lifecycle contract stay the same. + +## Identity labels + +Every session can carry four opaque identity labels. The framework never interprets them — it propagates them to hooks, traces, and `SessionData`, and restores them on resume. This is how a host attributes a session to a tenant, a principal, an agent template, and a wider correlation chain. + +Pair identity labels with a `sessionStore` / `session_store` so the labels survive a process restart. On resume, **caller-supplied options win**, so you can relabel a session as you move it between nodes. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let options = SessionOptions::new() + .with_tenant_id("tenant-example") + .with_principal("principal-example") + .with_agent_template_id("agent-template-example") + .with_correlation_id("trace-example") + .with_file_session_store("./.a3s/sessions"); + let session = agent + .session_builder("/path/to/project") + .options(options) + .build() + .await?; + + println!("tenant: {:?}", session.tenant_id()); + println!("principal: {:?}", session.principal()); + println!("template: {:?}", session.agent_template_id()); + println!("correlation: {:?}", session.correlation_id()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +const session = agent.session('/path/to/project', { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', +}); + +// Getters return string | null +console.log(session.tenantId); // 'tenant-example' +console.log(session.principal); // 'principal-example' +console.log(session.agentTemplateId); // 'agent-template-example' +console.log(session.correlationId); // 'trace-example' +``` + + + + +```python +opts = SessionOptions() +opts.tenant_id = 'tenant-example' +opts.principal = 'principal-example' +opts.agent_template_id = 'agent-template-example' +opts.correlation_id = 'trace-example' +session = agent.session('/path/to/project', opts) + +# Getters are properties, return str | None +print(session.tenant_id) # 'tenant-example' +print(session.principal) # 'principal-example' +print(session.agent_template_id) # 'agent-template-example' +print(session.correlation_id) # 'trace-example' +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func value(value *string) string { + if value == nil { + return "" + } + return *value +} + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, "/path/to/project", &code.SessionOptions{ + TenantID: "tenant-example", + Principal: "principal-example", + AgentTemplateID: "agent-template-example", + CorrelationID: "trace-example", + FileSessionStoreDir: "./.a3s/sessions", + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + info, err := session.Info(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("tenant:", value(info.TenantID)) + fmt.Println("principal:", value(info.Principal)) + fmt.Println("template:", value(info.AgentTemplateID)) + fmt.Println("correlation:", value(info.CorrelationID)) +} +``` + + + + +## Budget / cost guard + +A budget guard lets the host gate every LLM call against a cost or token budget. The framework calls your guard _before_ each LLM request and _after_ it returns. The guard is policy you own; the framework only enforces the decision you hand back. + + + + +```rust +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; + +use a3s_code_core::{ + budget::{BudgetDecision, BudgetGuard}, + llm::TokenUsage, + Agent, SessionOptions, +}; + +#[derive(Debug)] +struct TokenGuard { + limit: usize, + used: AtomicUsize, +} + +#[async_trait::async_trait] +impl BudgetGuard for TokenGuard { + async fn check_before_llm( + &self, + _session_id: &str, + estimated_prompt_tokens: usize, + ) -> BudgetDecision { + let used = self.used.load(Ordering::Relaxed); + if used.saturating_add(estimated_prompt_tokens) > self.limit { + BudgetDecision::Deny { + resource: "tokens".into(), + reason: "monthly cap reached".into(), + } + } else { + BudgetDecision::Allow + } + } + + async fn record_after_llm(&self, _session_id: &str, usage: &TokenUsage) { + self.used.fetch_add(usage.total_tokens, Ordering::Relaxed); + } +} + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let guard = Arc::new(TokenGuard { + limit: 50_000, + used: AtomicUsize::new(0), + }); + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_budget_guard(guard)) + .build() + .await?; + + let result = session.send("Summarize this repository.", None).await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +session.setBudgetGuard({ + checkBeforeLlm(ctx) { + if (overLimit(ctx.sessionId, ctx.estimatedTokens)) { + return { + decision: 'deny', + resource: 'tokens', + reason: 'monthly cap reached', + }; + } + return { decision: 'allow' }; + }, + recordAfterLlm(ctx) { + meter(ctx.sessionId, ctx.usage); + }, +}); + +// Clear the guard +session.setBudgetGuard(null); +``` + +Node callbacks receive a single `ctx` object and **must not throw**. Wrap logic in `try/catch` and return an explicit decision. Hung or unreadable `check*` callbacks fail closed as `deny`. + + + + +```python +class MyGuard: + def check_before_llm(self, session_id, estimated_tokens): + if over_limit(session_id, estimated_tokens): + return {'decision': 'deny', 'resource': 'tokens', 'reason': 'monthly cap reached'} + return {'decision': 'allow'} + + def record_after_llm(self, session_id, usage): + meter(session_id, usage) + +opts = SessionOptions() +opts.budget_guard = MyGuard() +session = agent.session('/path/to/project', opts) + +# To clear: set opts.budget_guard = None and re-create the session. +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "sync/atomic" + "time" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + const limit uint64 = 50_000 + var used atomic.Uint64 + err = session.SetBudgetGuard(ctx, &code.BudgetGuardHandlers{ + CheckBeforeLLM: func( + _ context.Context, + call code.BudgetLLMContext, + ) (*code.BudgetDecision, error) { + if used.Load()+uint64(call.EstimatedTokens) > limit { + return &code.BudgetDecision{ + Decision: "deny", + Resource: "tokens", + Reason: "monthly cap reached", + }, nil + } + return &code.BudgetDecision{Decision: "allow"}, nil + }, + RecordAfterLLM: func( + _ context.Context, + call code.BudgetUsageContext, + ) error { + used.Add(uint64(call.Usage.TotalTokens)) + return nil + }, + Timeout: 2 * time.Second, + }) + if err != nil { + log.Fatal(err) + } + defer session.SetBudgetGuard(context.Background(), nil) + + result, err := session.Run(ctx, "Summarize this repository.") + if err != nil { + log.Fatal(err) + } + fmt.Printf("%s\nspent %d / %d tokens\n", result.Text, used.Load(), limit) +} +``` + +Go callbacks use the same fail-closed contract: an error, timeout, or malformed +decision denies the protected LLM or tool call. + + + + +The native guard decision shape is equivalent across all four SDKs: + +| Return value | Effect | +| ----------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------- | +| `None` / `null` / `{ decision: 'allow' }` | Proceed with the LLM call. | +| `{ decision: 'soft', resource, consumed, limit, message? }` | Emits `BudgetThresholdHit` (kind `soft`) and proceeds. | +| `{ decision: 'deny', resource, reason }` | Aborts the LLM call. Python raises `RuntimeError("Budget exhausted...")`; Node rejects with `"Budget exhausted..."`. | + +Robustness is intentional but SDK-specific: a **missing guard method** is treated as the permissive default. Python callback errors fall back to Allow. Node callbacks must not throw. Go errors, timeouts, and malformed `check*` returns fail closed as `deny`. + +## Cluster event vocabulary + +The host emits cluster-level decisions as structured `AgentEvent` variants through its hook executor. In-session hooks subscribe to them uniformly — the same way they observe any other event — so policy authored at the host shows up to the agent's own hooks without special casing. + +The cluster vocabulary is: + +- **`BudgetThresholdHit { resource, kind, consumed, limit, message? }`** — a budget guard returned a `soft` decision (or the host crossed a threshold it tracks itself). `kind` distinguishes soft warnings from harder limits. +- **`PassivationRequested { reason, deadline_ms? }`** — the host is asking the session to reach a safe, persistable state so it can be evicted from this node. `deadline_ms`, when present, is the grace window before forced eviction. +- **`PeerInvocation { from_session_id, from_tenant_id?, correlation_id? }`** — another session invoked this one. The labels let the receiver attribute the call back to its origin tenant and correlation chain. + +These are observed through the same verified hook API your in-session hooks already use — `session.registerHook` in Node, `session.register_hook` in Python (see [Hooks](/guide/hooks)). Treat the three variants above as the documented contract; the host is responsible for emitting them via its hook executor. + +## Deterministic IDs and time (replay) + +A cluster that wants **bit-identical replay** of a run on a different node must remove the two sources of nondeterminism in a normal run: random IDs and the wall clock. The Rust core models both behind a `HostEnv { id_generator, clock }`. The default pairs a UUID generator with the system clock; replay tooling swaps in a `SequentialIdGenerator` and a `FixedClock` so that re-executing the same inputs produces the same IDs and timestamps, and therefore the same output, on any node. + +All four SDKs expose the same deterministic configuration. Set both the ID +prefix and fixed timestamp, and recreate the configuration from the same values +at the start of each replay. + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{ + host_env::{FixedClock, HostEnv, SequentialIdGenerator}, + Agent, SessionOptions, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let host_env = Arc::new(HostEnv::new( + Arc::new(SequentialIdGenerator::new("replay")), + Arc::new(FixedClock::new(1_700_000_000_000)), + )); + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_host_env(host_env)) + .build() + .await?; + + println!("{}", session.session_id()); + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = await agent.sessionAsync('.', { + hostEnv: { + sequentialIdPrefix: 'replay', + fixedTimeMs: 1_700_000_000_000, + }, +}); + +console.log(session.sessionId); +await session.closeAsync(); +await agent.close(); +``` + + + + +```python +from a3s_code import Agent, HostEnvConfig, SessionOptions + +agent = Agent.create('agent.acl') +options = SessionOptions() +options.host_env = HostEnvConfig( + sequential_id_prefix='replay', + fixed_time_ms=1_700_000_000_000, +) +session = agent.session('.', options) + +print(session.session_id) +session.close() +agent.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.NewAgent(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + prefix := "replay" + fixedTime := uint64(1_700_000_000_000) + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + HostEnv: &code.HostEnvConfig{ + SequentialIDPrefix: &prefix, + FixedTimeMS: &fixedTime, + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + fmt.Println(session.SessionID()) +} +``` + + + + +## Loop checkpoints and run resumption + +With a `sessionStore` / `session_store` configured, the agent loop persists a checkpoint after **each completed tool round**, keyed by run id. Any node that shares the same store can rehydrate the run and continue it. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_file_session_store("./.a3s/sessions") + .with_session_id("session-from-node-a"), + ) + .build() + .await?; + + let result = session.resume_run("run-id-from-node-a").await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session(workspace, { + sessionStore: new FileSessionStore('./.a3s/sessions'), + sessionId: 'session-from-node-a', +}); + +const result = await session.resumeRun('run-id-from-node-a'); +``` + + + + +```python +from a3s_code import FileSessionStore + +opts = SessionOptions() +opts.session_store = FileSessionStore('./.a3s/sessions') +opts.session_id = 'session-from-node-a' +session = agent.session(workspace, opts) + +result = session.resume_run('run-id-from-node-a') +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + options := &code.SessionOptions{ + SessionID: "session-from-node-a", + FileSessionStoreDir: "./.a3s/sessions", + } + session, err := agent.Session(ctx, ".", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + result, err := session.ResumeRun(ctx, "run-id-from-node-a") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +A **new run id** is allocated for the resumed work — the original run is left intact in the store. Two error paths are worth handling: + +- **`resume_run requires a session_store`** — no store was configured; fall back to a fresh session. +- **`no loop checkpoint found for run 'X'`** — the run never reached its first checkpoint, or it was pruned; retry later or treat the run as lost. + +Because checkpoints are taken only **between tool rounds, never mid-tool**, a resumed run never replays a half-executed tool. See [Persistence](/guide/persistence) for store details. + +## Retention caps for long-running sessions + +A session that runs for hours or days accumulates state in four in-memory stores: run records, per-run event buffers, trace events, and terminal subagent task snapshots. Left unbounded, these grow with session age — fine for short-lived sessions, a real leak for long-lived ones. + +`SessionRetentionLimits` caps each store. Omitted fields keep the framework's +finite defaults; set `unbounded: true` only when unlimited retention is +intentional. Eviction is strict **FIFO**, and **running subagent tasks are never +dropped** — only terminal snapshots are evicted. + +Use `retentionLimits` in Node, `opts.retention_limits` in Python, or +`SessionOptions.RetentionLimits` in Go. Rust hosts use +`SessionOptions::with_retention_limits(...)`. See [Limits](/guide/limits). + +--- + +**See also:** [Multi-machine](/guide/multi-machine) · [Persistence](/guide/persistence) · [Limits](/guide/limits) · [Hooks](/guide/hooks) diff --git a/website/docs/v8.5.1/en/guide/commands.mdx b/website/docs/v8.5.1/en/guide/commands.mdx new file mode 100644 index 00000000..0f32aedf --- /dev/null +++ b/website/docs/v8.5.1/en/guide/commands.mdx @@ -0,0 +1,70 @@ +--- +title: 'Commands' +description: 'Command surfaces and recommended control flows' +--- + +# Commands + +A3S Code is primarily SDK-driven. Product CLIs and UIs usually map commands to +session APIs rather than depending on a large public command protocol. + +This page describes the SDK command registry. For the built-in commands in the +`a3s code` terminal app, see [A3S Code TUI](/guide/tui). + +The two surfaces are intentionally different: + +| Surface | Owner | Purpose | +| ------------------------- | ---------------- | ------------------------------------------------------------------------------------------------------------------------- | +| `a3s code` slash commands | CLI/TUI | Terminal controls such as `/model`, `/flow`, `/memory`, `/kb`, `/update`, and `/exit`. | +| SDK command registry | Host application | Product-defined commands registered with `session.registerCommand(...)` and invoked through `session.send("/name args")`. | + +SDK commands execute before the LLM sees the input. A handler receives the raw +argument string plus session metadata, and returns display text. Use commands +for thin control-plane actions; use normal SDK methods for workflows, tools, +verification, and persistence. + +Inside the TUI, `/flow` manages workflow asset files and optional host +integrations. It is not the same surface as `DynamicWorkflowRuntime`, which is +the local A3S Flow-backed runtime used by `ultracode` and `?` DeepResearch +through the model-visible `dynamic_workflow` tool. + +## Recommended Mappings + +| User action | Session API | +| --------------------------------- | ------------------------------------------------------------------------------------- | +| Send a prompt | `session.send(prompt)` | +| Stream a prompt | `session.stream(prompt)` | +| Ask a side question | Create an isolated session, or call `send` / `stream` with explicit isolated history. | +| Run a deterministic tool | `session.tool(name, args)` | +| Read history | `session.history()` | +| Save state | `session.save()` | +| Resume state | `agent.resumeSession(id, options)` | +| Inspect tools | `session.toolNames()` / `session.toolDefinitions()` | +| Verify work | `session.verifyCommands(subject, commands)` | +| Replay run state | `session.runs()` / `session.runEvents(runId)` | +| Inspect or cancel the current run | `session.currentRun()` / `session.cancelRun(runId)` | + +## Building Slash Commands + +If your app exposes slash commands, keep them thin. Register the handler, list +the available commands, and invoke the command through `session.send()`: + +```ts +session.registerCommand( + 'docs_status', + 'Return docs command status', + (args, ctx) => { + return `status args=${args}; session=${ctx.sessionId}; workspace=${ctx.workspace}`; + }, +); + +console.log(session.listCommands()); + +const result = await session.send('/docs_status check'); +console.log(result.text); +``` + +Command handlers should not hide long-running work. If a command needs to run +tools, verify output, fan out to subagents, or schedule recurring automation, +have the handler return a short acknowledgement and run that work through +explicit host code. That keeps cancellation, audit, and permissions visible. diff --git a/website/docs/v8.5.1/en/guide/context.mdx b/website/docs/v8.5.1/en/guide/context.mdx new file mode 100644 index 00000000..15cbf323 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/context.mdx @@ -0,0 +1,295 @@ +--- +title: 'Context' +description: 'Budgeted context assembly for reliable coding agents' +--- + +# Context + +A3S Code treats context as a budgeted resource. The model should see the smallest useful context for the current decision, not every available file, skill, memory, and tool log. + +## Sources + +Context can come from: + +- user prompt and conversation history +- AGENTS.md project instructions +- skills and agent definitions +- memory stores +- file search and direct tool results +- MCP tools and context providers +- delegated task summaries +- trace events + +AGENTS.md injection, skills discovery, memory APIs, direct tool results, +delegated task helpers, and `traceEvents()` are part of the documented Node SDK +surface. Validate MCP context behavior against your own live integrations +before documenting it as product behavior. + +## Assembly + +```text +sources -> ContextItem -> rank -> dedupe -> budget -> render +``` + +Long grep output, logs, and child transcripts should be preserved outside the prompt and summarized into prompt-safe evidence. Use `session.traceEvents()` for compact runtime evidence. + +## Exact Cognitive Packages (Rust Host) + +An embedding Rust host can bind a session to one exact A3S Use +cognitive-package generation. A3S Code does not install packages, resolve a +Registry entry, or select `latest`; the host injects both an immutable +`CognitivePackageBindingV1` and a provider holding the matching Knowledge +lease. + +```rust +use a3s_code_core::{CognitiveContextSession, SessionOptions}; + +let cognitive_context = CognitiveContextSession::new(binding, provider)?; +let options = SessionOptions::new().with_cognitive_context(cognitive_context); +``` + +The durable `a3s.code.cognitive-package-session-binding.v1` identity includes +the package id/version, lifecycle generation, generation digest, capability +snapshot digest, exact Knowledge surface, and prompt-injection limits. Each +typed request and cited Markdown response repeats that binding and is validated +before content enters model context. + +The hard limits are four documents, 6 KiB per document, and 6 KiB total. The +host may choose smaller limits in the binding. Provider failure, malformed +citations, request mismatch, or generation drift fails closed instead of +falling back to unrelated retrieval. + +Use `with_cognitive_context`; adding a cognitive provider through the generic +context-provider list is rejected because its binding could not be persisted +correctly. An exact cognitive package cannot accompany general-purpose RAG or +graph providers, and personal-memory recall is suppressed. Code-owned +workspace instructions and skills remain available. + +The binding is stored in the session snapshot and emitted as +`cognitive_context_bound`. On resume, the host must re-inject a provider with +the same binding; a missing or different generation is rejected. This typed +boundary is currently a Rust-host integration surface rather than a Node.js, +Python, or Go session option. + + + +## Session-owned workspace retrieval + +A3S Code 8 can build a bounded retrieval index for one workspace and one +session. Construction runs asynchronously, results are verified against the +current source, and every vector is released when the session closes. The +runtime does not install or require a vector database. + +Workspace Retrieval is a host capability, not a switch the model can turn on. +Omit the typed option to keep it disabled. A disabled session constructs no +additional catalog, invokes no embedding provider, and exposes no semantic or +hybrid search mode to the model. + +Enabled sessions do not register a separate `vector_db` tool. Instead, the +single model-facing `search` schema adds `mode: "semantic"` and +`mode: "hybrid"`. Both modes query the session-owned projection through the +same governed tool path as grep, glob, and BM25; disabled sessions omit them +from the schema rather than advertising calls that cannot run. + +### Choose the smallest useful retrieval surface + +| Requirement | Use | Embedding model | +| ------------------------------------------- | --------------------------- | ---------------------------------------------------------------------- | +| Known identifier or exact text | Exact, glob, or grep search | Not required | +| Natural-language query over project text | Incremental BM25 | Not required | +| Definitions, references, or diagnostics | Code Intelligence | Not required | +| Vocabulary mismatch or multilingual meaning | Semantic retrieval | Host-supplied callback required | +| Mixed identifiers and natural language | Hybrid RRF | Optional; lexical and symbol channels remain available without vectors | +| Reduce near-duplicate evidence | Deterministic reranker | Not required; local CPU algorithm | + +Dense semantic search necessarily needs a text-to-vector function, but that +function can be an in-process CPU callback. A3S Code does not require a remote +API, GPU, bundled model, or runtime model download. Model revision, license, +artifact verification, caching, and credentials remain the host's +responsibility. + +### Asynchronous vector projection lifecycle + +The semantic serving projection is a session-owned, exact A3S Memory vector +index, not a durable or shared vector database. Product builds use a separate +session-local zvec-rust FTS collection for lexical ranking; its native handles +are short-lived and bounded. Session construction returns before corpus +embedding finishes. The background indexer reads admitted text, publishes +immutable per-file partitions atomically, and reconciles later source revisions +without re-embedding unchanged files. Reopening a session builds new +projections; closing it cancels outstanding provider work and releases all +semantic and lexical state. + +```text +session construction -> return immediately + `-> build text catalog and vector partitions in the background + +query -> fuse independent ranks -> verify current source -> render evidence + +session close -> cancel provider -> join indexer -> release all vectors +``` + +Status moves through `building`, `ready`, `degraded`, and `closed`. Queries can +use published coverage while construction continues. Hosts that need a +stronger first-query boundary can wait for readiness for up to 30 seconds; +timeout preserves the partial fallback, while cancellation or session close +interrupts the wait. + +| State | Vector projection | Query behavior | +| ---------- | ------------------------------------------------------- | ----------------------------------------------------------------- | +| `building` | Valid file partitions publish atomically as they finish | Exact/BM25 stay available; semantic coverage may be partial | +| `ready` | The observed source generation has full coverage | Semantic and hybrid queries use the complete published generation | +| `degraded` | Valid partitions remain; bounded failures are reported | Lexical paths continue and semantic results expose partial status | +| `closed` | Indexing is cancelled and vectors are released | Recreate a session before using semantic retrieval again | + +### Backend and ranking boundary + +zvec-rust owns lexical FTS/BM25 postings in product builds; a minimal build can +explicitly use the portable BM25 implementation. A lexical initialization or +query failure is reported as bounded lexical degradation and cannot change the +A3S Memory semantic authority. The SDKs expose typed lexical options and no +primitive backend-name selector. Temporary lexical collections are deleted on +normal close; process-crash residue remains subject to the host operating +system's temporary-directory policy. + +### Ranking + +One immutable chunk catalog backs incremental BM25, Memory-authoritative exact +vectors, stable source anchors, and exact-literal or Code Intelligence +candidates. Hybrid mode combines independent one-based ranks with +reciprocal-rank fusion (`k = 60`) instead of mixing incomparable raw scores. +RRF-only is the default. The optional deterministic reranker is bounded, +model-free CPU code that reduces duplicate evidence while protecting exact +identifiers. + +### Text admission and chunking + +Only manifest-admitted UTF-8 text and source files enter the catalog. +Generated files, oversized files, credentials, key material, `.a3s` control +paths, and non-text assets are excluded before chunking and embedding. PDF, +Office, image, audio, OCR, and other knowledge compilation belongs to a +separate knowledge compiler; Workspace Retrieval does not guess how to parse +those formats. + +Built-in typed strategies cover line/byte chunks, fixed UTF-8 windows, and +recursive separators. Trusted Rust hosts can supply a custom splitter whose +ranges preserve UTF-8 boundaries, cover admitted bytes, and always make +forward progress. Node.js, Python, and Go accept typed built-in strategy +objects; primitive strategy names are rejected. + +### SDK control surface + +| Host | Enable | Keep disabled | Query and status | +| ------- | ------------------------------------------------- | ------------------------------- | ------------------------------------------------------------------ | +| Rust | `SessionOptions::with_workspace_retrieval(...)` | `without_workspace_retrieval()` | `workspace_retrieval_status`, `semantic_search`, `hybrid_search` | +| Node.js | Set typed `workspaceRetrieval` in session options | Omit it | `workspaceRetrievalStatus()`, `semanticSearch()`, `hybridSearch()` | +| Python | Set `SessionOptions.workspace_retrieval` | Assign `None` | `workspace_retrieval_status()`, async semantic and hybrid search | +| Go | Set `SessionOptions.WorkspaceRetrieval` | Use `nil` | `WorkspaceRetrievalStatus`, `SemanticSearch`, `HybridSearch` | + +### CLI activation + +The `a3s` CLI keeps semantic retrieval disabled unless a trusted user ACL or a +file selected explicitly with `--config` enables it. An automatically +discovered workspace `.a3s/config.acl` may only disable an inherited retrieval +route; it cannot authorize source egress or choose an embedding backend. + +Remote embedding needs a separate provider route and an explicit source-egress +grant: + +```acl +workspace_retrieval { + enabled = true + allow_source_egress = true + model = "openai/text-embedding-3-small" + dimension = 1536 + normalization = "none" +} +``` + +Local CPU embedding is mutually exclusive with the remote fields and does not +need a source-egress grant: + +```acl +workspace_retrieval { + enabled = true + semantic_readiness_timeout_ms = 30000 + + local_cpu { + artifact_manifest = "models/multilingual-mini/model.acl" + intra_threads = 2 + } +} +``` + +The local artifact manifest is revision- and SHA-256-bound; the runtime never +downloads model files. Run `a3s config validate` and inspect the redacted +`workspaceRetrieval` section from `a3s config show` before creating a session. +The embedding route is independent from `default_model`, so selecting DeepSeek +for chat and tool calls does not implicitly turn that chat endpoint into an +embedding service. + +The CLI's `local_cpu` adapter is shipped for Linux x64/ARM64, Windows x64, and +Apple Silicon. Intel macOS 12 (`x86_64`) builds intentionally omit the optional +ONNX adapter. On Intel, leave retrieval model-free or use a separately +authorized remote embedding route. + +Provider descriptors lock identity, model, dimension, and normalization. +Runtime validation rejects partial, duplicate, unknown, dimension-mismatched, +non-finite, non-normalized, or descriptor-drifted responses. Diagnostics do +not copy input text, vectors, remote response bodies, credentials, or endpoint +values. + +### Quality and safety evidence + +Status snapshots report coverage, queue depth, failures, vector memory, +batching, request amplification, non-text provider inputs, and post-close +release. Release evaluation measures Recall@5, MRR, latency, memory, non-text +egress, and cleanup. The locked cross-SDK DeepSeek fixture completes 3/3 exact +tasks with Recall@5 1.0, MRR 0.5, 1.0x document request amplification, zero +non-text inputs, and complete vector release. This is a portability gate, not +a claim that one model or reranker is best for every repository. + +The `v7.0.1` post-release rerun at Code `5aa9642` repeated every gate on +2026-08-17: + +| SDK | Exact tasks / one-Search protocols | DeepSeek turn p50 / p95 | Tokens | +| ------- | ---------------------------------: | ----------------------: | -----: | +| Node.js | 3 / 3 | 16,033 / 16,538 ms | 14,540 | +| Python | 3 / 3 | 15,552 / 23,751 ms | 14,784 | +| Go | 3 / 3 | 16,636 / 19,009 ms | 14,171 | + +All three arms retained Recall@5 1.0, MRR 0.5, 1.0x document-request +amplification, zero non-text inputs, and complete post-close release. These +remote timings are diagnostic rather than local retrieval latency objectives. + +Before rendering a result, A3S Code rereads the authoritative file and verifies +the full-file digest and exact chunk byte range. Deleted, stale, unreadable, or +superseded candidates are not exposed. See the +[operations runbook](https://github.com/A3S-Lab/Code/blob/main/manual/WORKSPACE_RETRIEVAL_OPERATIONS.md) +and [qualification record](https://github.com/A3S-Lab/Code/blob/main/manual/WORKSPACE_RETRIEVAL_QA.md) +for production thresholds, rollback rules, and reproducible evaluation. + +## Compaction + +Enable automatic compaction for long sessions: + +```ts +const session = agent.session('/repo', { + autoCompact: true, + autoCompactThreshold: 0.75, + maxContextTokens: 128_000, +}); +``` + +When `maxContextTokens` is omitted, Core uses the selected model's declared +context window when available. Before each model request, Core accounts for the +system prompt, conversation, tool calls and results, and exposed tool schemas. +At the configured threshold it bounds oversized tool output, summarizes the +older safe prefix, keeps recent messages, and continues the same task. The +summary participates in later compactions, so long sessions can roll forward +through repeated compression; this does not enlarge the model's physical +single-request context window. A successful `context_compacted` event includes +the cumulative summary so hosts that supply external history can persist the +same compact generation across turns. + +Python exposes the same override as `max_context_tokens`. diff --git a/website/docs/v8.5.1/en/guide/convention-over-configuration.mdx b/website/docs/v8.5.1/en/guide/convention-over-configuration.mdx new file mode 100644 index 00000000..5b9a46d0 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/convention-over-configuration.mdx @@ -0,0 +1,233 @@ +--- +title: 'Convention Over Configuration' +description: 'Build durable work units with filesystem-first agents, tools, connections, subagents, schedules, persistence, and HITL' +--- + +# Convention Over Configuration + +A3S Code's convention-over-configuration capability is not a separate runner. +It is a durable-agent composition built from +[filesystem-first](/guide/filesystem-first) conventions, the session +runtime, tool permissions, multi-agent delegation, resumable orchestration, +HITL, run replay, and the serve scheduler. It turns an agent from "a +chat-capable SDK session" into a work unit with a role, tools, collaborators, +state, and takeover points. + +If you are evaluating directory-first agent frameworks, treat A3S Code as a +lower-level and more embeddable runtime. Directory conventions, tools, +connections, child agents, schedules, persistence, and observability are all +present, but they are not tied to one frontend channel or deployment platform. +A host can connect them to managed sessions, open-platform APIs, MCP, +A3S Box, or its own task system. + +## Capability Map + +| Framework concept | A3S Code capability | Notes | +| --------------------- | ------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Agent as a directory | [Agent Directory](/guide/agent-dir) | `instructions.md`, `agent.acl`, `skills/`, `tools/`, and `schedules/` synthesize a runnable agent. | +| Markdown instructions | `instructions.md`, `AGENTS.md`, prompt slots | Instructions are injected as slots, so they cannot override harness boundaries, response contracts, or safety gates. | +| Markdown skills | [Skills](/guide/skills) | File skills, inline skills, and explicit registries share discovery semantics. A3S Code ships no default embedded skills. | +| TypeScript tools | `program`, direct tools, MCP tools, AgentDir script tools | A3S Code focuses on tool registration, permission gates, typed errors, and verification evidence; script tools run in a constrained QuickJS environment. | +| Sandbox | `program` sandbox, permission policy, HITL, A3S Box | A3S Code handles tool permissions and script sandboxing; process, filesystem, and network isolation belong in A3S Box. | +| Channels | Managed sessions, open-platform WebSocket/SSE, host UI | A3S Code does not hard-code channels. Hosts connect streaming events and run replay to any frontend. | +| Connections | MCP, provider config, host-injected credentials | Connections are configured and authorized by the host. Do not put tokens in the agent directory. | +| Subagents | [Tasks](/guide/tasks), [Teams](/guide/teams), `workerAgents` | Supports focused or multi-item `task` calls, automatic delegation, and dynamic worker agents. | +| Schedules | AgentDir `schedules/` + `serve_agent_dir` | Every schedule has an independent session id and can recover context after restart when a session store is configured. | +| Durable execution | `SessionStore`, run replay, `parallelResumable` | Session history, run events, and resumable workflow checkpoints can be persisted; failed steps retry on resume. | +| Human-in-the-loop | [Security](/guide/security), confirmation inheritance, tool confirmation | High-risk tools can request confirmation. Child runs can auto-approve, fail on Ask, or inherit the parent policy. | +| Evaluations | [Verification](/guide/verification), reports, regression evidence | A3S Code productizes evaluation as verification commands, evidence summaries, run replay, and release gates rather than a separate eval suite. | + +## Minimal Shape + +An interactive team agent can start from a normal repository: + +```text +repo/ +├── agent.acl +├── AGENTS.md +├── .a3s/ +│ ├── agents/ +│ │ ├── release-reviewer.md +│ │ ├── security-reviewer.md +│ │ └── verification-runner.md +│ └── skills/ +│ └── release-readiness.md +└── src/ +``` + +`agent.acl` controls models, providers, parallelism, and automatic delegation: + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +agent_dirs = ["./.a3s/agents"] + +auto_delegation { + enabled = true + min_confidence = 0.72 + max_tasks = 4 + auto_parallel = false +} +``` + +The host starts a session and wires in skills, agent directories, persistence, and delegation policy: + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + skillDirs: ['./.a3s/skills'], + agentDirs: ['./.a3s/agents'], + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, + autoSave: true, +}); + +const result = await session.send(` +Prepare a release-readiness check: +1. Ask the exploration role to find high-risk changes. +2. Ask the security role to inspect permissions, secrets, and external side effects. +3. Ask the verification role to list required regression commands. +4. Merge everything into a blocker-first report. +`); + +console.log(result.text); +console.log(result.verificationSummaryText); +``` + +## Agent Directory Shape + +Use a filesystem-first agent when you need long-running behavior, schedules, or directory-scoped tools: + +```text +release-agent/ +├── instructions.md +├── agent.acl +├── skills/ +│ └── release-readiness.md +├── schedules/ +│ └── daily.md +└── tools/ + ├── github.md + └── search-auth.md +``` + +`instructions.md` is the role slot: + +```md +You are a release-readiness agent for this repository. +Always separate blockers from follow-up work. +Never invent versions or CI status. Read evidence from the workspace. +``` + +`schedules/daily.md` is a recurring turn: + +```md +--- +cron: '0 9 * * *' +name: daily-release-check +enabled: true +--- + +Summarize merged changes since the last run, inspect release risks, +and report only blockers plus required verification. +``` + +`serve_agent_dir` creates an independent session for every schedule. With a `SessionStore`, restart reloads the current directory config and tools while recovering prior context from the store. + +## Tools And Connections + +A3S Code exposes tools in three layers: + +| Layer | Usage | Good for | +| ------------------------- | -------------------------------------------------------- | ----------------------------------------------------------------------------------- | +| Built-in and direct tools | `session.tool(...)`, file, shell, git, `generate_object` | The host knows exactly what to run or needs deterministic calls. | +| MCP | MCP servers | GitHub, Linear, internal systems, remote host tools, or cross-process capabilities. | +| AgentDir script tools | `tools/*.md` + QuickJS `program` | Wrapping constrained scripts as model-visible tools. | + +Tools are not unlimited just because a file exists. Visibility, permission gates, HITL, allow-lists, and sandboxing are controlled by the harness or the AgentDir loader. Give high-privilege tools only to trusted directories, and inject secrets through environment variables and host connections. + +## Subagents And Teams + +A convention-first `subagents/` concept maps to three A3S Code entry points: + +| Entry point | Who decides the lanes | Use it when | +| ------------------------------------------------------------------------------------ | --------------------- | ------------------------------------------------------------------------------------- | +| `task` | Parent agent | The model decides whether to delegate one item or fan out multiple independent items. | +| `session.task(...)` / `session.tasks(...)` | Host code | The host knows the lanes but still wants agent execution. | +| `session.parallel(...)` / `session.pipeline(...)` / `session.parallelResumable(...)` | Host code | The workflow must be reproducible, testable, and resumable. | + +Automatic delegation depends on agent descriptions and confidence scoring. It fits "the user describes the goal and the runtime chooses specialists." Fixed release flows, batch reviews, and migrations are better expressed explicitly with orchestration. + +## Observability And Takeover + +Long-running agents must be observable and interruptible. A3S Code's core observation surfaces are: + +- `stream()` for incremental events. +- `runs()`, `runSnapshot()`, and `runEvents()` for active and historical runs. +- `toolNames()` / `toolDefinitions()` for the visible tool surface. +- `activeTools()` for currently running tool-call snapshots. +- `cancelRun(runId)` to interrupt an active turn. +- `traceEvents()` for compaction, delegation, tools, and verification evidence. +- `verificationSummaryText` for release or review gates. + +A host platform can translate those events into WebSocket/SSE streams, audit +records, debug panels, and workflow node states. + +## Relationship To A3S Box + +A3S Code owns the agent loop, tools, delegation, state, and verification. A3S Box owns stronger runtime isolation: MicroVMs, OCI workloads, networking, and TEE. If "the model may run shell, but the process must be isolated" is a requirement, run the tool execution through A3S Box or expose isolated capabilities through MCP. + +Typical composition: + +```text +A3S Code session + -> permission policy and HITL + -> MCP tool adapter + -> A3S Box isolated workload + -> typed result and verification evidence +``` + +## When To Use It + +Use this shape for: + +- Long-running engineering agents for release checks, dependency upgrades, and repository maintenance. +- Work that naturally splits into explore / review / verify / implement roles. +- Automatic delegation with retained permission gates, auditability, and verification evidence. +- Cron-based reports with recoverable context. +- Agents that need to connect to managed workflows, hosted sessions, or external collaboration channels. + +Avoid it when: + +- A single deterministic API call is enough; use a tool contract instead. +- The task requires full GUI automation but has no structured tool interface; provide MCP or browser tools first, then let A3S Code schedule them. +- The task needs strong OS-level isolation but only has a local shell enabled; integrate A3S Box. + +## Reading Order + +1. [Filesystem-First](/guide/filesystem-first) +2. [Agent Directory](/guide/agent-dir) +3. [agents/ Role Directory](/guide/filesystem-agents) +4. [tools/ Tool Directory](/guide/filesystem-tools) +5. [schedules/ Schedule Directory](/guide/filesystem-schedules) +6. [Tasks](/guide/tasks) +7. [Teams](/guide/teams) diff --git a/website/docs/v8.5.1/en/guide/examples/_meta.json b/website/docs/v8.5.1/en/guide/examples/_meta.json new file mode 100644 index 00000000..6814f12c --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/_meta.json @@ -0,0 +1,22 @@ +[ + "index", + "quick-start", + "streaming", + "direct-tools", + "structured-output", + "batch", + "planning", + "orchestration", + "external-tasks", + "lane-queue", + "memory", + "auto-compact", + "prompt-slots", + "ripgrep-context", + "model-switching", + "skills", + "skill-tool", + "hooks", + "security", + "git-worktree" +] diff --git a/website/docs/v8.5.1/en/guide/examples/auto-compact.mdx b/website/docs/v8.5.1/en/guide/examples/auto-compact.mdx new file mode 100644 index 00000000..efac509f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/auto-compact.mdx @@ -0,0 +1,173 @@ +--- +title: 'Auto Compact' +description: 'Let the runtime keep long sessions inside the context budget automatically' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Auto Compact + +A3S Code can keep long conversations inside the model's context budget for you. +Enable `autoCompact` and the runtime watches context usage; once it crosses +`autoCompactThreshold`, older turns are compacted into a running summary so the +agent stays coherent across many steps without you managing tokens by hand. +Continuation handles the other direction: when a single response is truncated by +length, the runtime automatically continues generating to assemble the full reply. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let options = SessionOptions::new() + .with_auto_compact(true) + .with_auto_compact_threshold(0.75) + .with_continuation(true) + .with_max_continuation_turns(3); + let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; + + for step in 0..50 { + session + .send( + &format!("Step {step}: continue refactoring the parser"), + None, + ) + .await?; + } + + println!("history turns: {}", session.history().len()); + if let Some(memory) = session.memory() { + println!("recent memory: {:#?}", memory.get_recent(5).await?); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + // Compact older turns once context fills past the threshold. + autoCompact: true, + autoCompactThreshold: 0.75, + // Auto-continue a single response that the model truncates by length. + continuationEnabled: true, + maxContinuationTurns: 3, +}); + +// Run a long, multi-step task. The runtime compacts older turns as needed; +// you never touch the token math. +for (let i = 0; i < 50; i++) { + await session.send(`Step ${i}: continue refactoring the parser`); +} + +// Inspect what the session is currently carrying. +console.log('history turns:', session.history().length); +console.log('recent memory:', await session.memoryRecent(5)); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +opts = SessionOptions() +# Compact older turns once context fills past the threshold. +opts.auto_compact = True +opts.auto_compact_threshold = 0.75 +# Auto-continue a single response that the model truncates by length. +opts.continuation_enabled = True +opts.max_continuation_turns = 3 +session = agent.session("/repo", opts) + +# Run a long, multi-step task. The runtime compacts older turns as needed; +# you never touch the token math. +for i in range(50): + session.send(f"Step {i}: continue refactoring the parser") + +# Inspect what the session is currently carrying. +print("history turns:", len(session.history())) +print("recent memory:", session.memory_recent(5)) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + options := &code.SessionOptions{ + AutoCompact: code.Ptr(true), + AutoCompactThreshold: code.Ptr(float32(0.75)), + ContinuationEnabled: code.Ptr(true), + MaxContinuationTurns: code.Ptr(uint32(3)), + } + session, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + for step := 0; step < 50; step++ { + prompt := fmt.Sprintf("Step %d: continue refactoring the parser", step) + if _, err := session.Run(ctx, prompt); err != nil { + log.Fatal(err) + } + } + + history, err := session.History(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("history turns:", len(history)) +} +``` + + + + +Use auto-compaction for long sessions where you want the runtime to manage context +pressure for you. `autoCompactThreshold` / `auto_compact_threshold` is a fraction of +the context window (0.0–1.0, default 0.8) at which compaction kicks in; lower it to +compact earlier. Inspect the live session with the SDK's history API. Rust, +Node.js, Python, and Go also expose direct recent-memory queries +(`memoryRecent` / `memory_recent` / `MemoryRecent`). diff --git a/website/docs/v8.5.1/en/guide/examples/batch.mdx b/website/docs/v8.5.1/en/guide/examples/batch.mdx new file mode 100644 index 00000000..ab7cd5e2 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/batch.mdx @@ -0,0 +1,194 @@ +--- +title: 'Batch' +description: 'Compose deterministic SDK helpers for grouped, model-free operations' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Batch + +There is no `batch()` method in the SDK. When a host workflow or agent turn has a +clear list of independent, deterministic steps, compose them directly from the +session's deterministic helpers (`readFile`, `grep`, `glob`, `ls`, `git`) and +aggregate the results yourself. That keeps the "batch" fully under your control: +no model calls, explicit ordering, and no destructive operations unless you +invoke them. + +The example below reads package metadata, the changelog, and the release script, +then reports version mismatches without editing any files. + + + + +```rust +use a3s_code_core::Agent; +use serde_json::Value; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/path/to/project") + .build() + .await?; + + let (package, changelog, release_script) = tokio::try_join!( + session.read_file("package.json"), + session.read_file("CHANGELOG.md"), + session.read_file("scripts/release.sh"), + )?; + + let package: Value = serde_json::from_str(&package)?; + let version = package["version"].as_str().unwrap_or_default(); + let mut mismatches = Vec::new(); + if !changelog.contains(version) { + mismatches.push(format!("CHANGELOG.md is missing {version}")); + } + if !release_script.contains(version) { + mismatches.push(format!("release.sh is missing {version}")); + } + + if mismatches.is_empty() { + println!("All files agree on the version."); + } else { + println!("{}", mismatches.join("\n")); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/path/to/project'); + +// Node helpers are async — group independent reads with Promise.all. +const [pkg, changelog, releaseScript] = await Promise.all([ + session.readFile('package.json'), + session.readFile('CHANGELOG.md'), + session.readFile('scripts/release.sh'), +]); + +const pkgVersion = JSON.parse(pkg).version; +const mismatches = []; +if (!changelog.includes(pkgVersion)) + mismatches.push(`CHANGELOG.md is missing ${pkgVersion}`); +if (!releaseScript.includes(pkgVersion)) + mismatches.push(`release.sh is missing ${pkgVersion}`); + +console.log( + mismatches.length ? mismatches.join('\n') : 'All files agree on the version.', +); + +session.close(); +``` + + + + +```python +import json +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session("/path/to/project", SessionOptions()) + +# Python helpers are synchronous — call them in sequence, no await. +pkg = session.read_file("package.json") +changelog = session.read_file("CHANGELOG.md") +release_script = session.read_file("scripts/release.sh") + +pkg_version = json.loads(pkg)["version"] +mismatches = [] +if pkg_version not in changelog: + mismatches.append(f"CHANGELOG.md is missing {pkg_version}") +if pkg_version not in release_script: + mismatches.append(f"release.sh is missing {pkg_version}") + +print("\n".join(mismatches) if mismatches else "All files agree on the version.") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/project", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + pkg, err := session.ReadFile(ctx, "package.json", nil) + if err != nil { + log.Fatal(err) + } + changelog, err := session.ReadFile(ctx, "CHANGELOG.md", nil) + if err != nil { + log.Fatal(err) + } + releaseScript, err := session.ReadFile(ctx, "scripts/release.sh", nil) + if err != nil { + log.Fatal(err) + } + + var metadata struct { + Version string `json:"version"` + } + if err := json.Unmarshal([]byte(pkg), &metadata); err != nil { + log.Fatal(err) + } + var mismatches []string + if !strings.Contains(changelog, metadata.Version) { + mismatches = append(mismatches, "CHANGELOG.md is missing "+metadata.Version) + } + if !strings.Contains(releaseScript, metadata.Version) { + mismatches = append(mismatches, "release.sh is missing "+metadata.Version) + } + if len(mismatches) == 0 { + fmt.Println("All files agree on the version.") + } else { + fmt.Println(strings.Join(mismatches, "\n")) + } +} +``` + + + + +Do not mix destructive operations into these grouped reads. Any destructive host +workflow (writes, `git` commits, `bash`) should sit behind explicit application +confirmation or your automation gates, constrained with `permissionPolicy` / +`permission_policy` so no unexpected step runs silently. + +When the steps are **not** independent — each one depends on the previous result +and you want an agent to drive them — use +[`session.pipeline(...)`](/guide/examples/orchestration) instead, which runs +the work in stages where each stage receives the prior stage's output. diff --git a/website/docs/v8.5.1/en/guide/examples/direct-tools.mdx b/website/docs/v8.5.1/en/guide/examples/direct-tools.mdx new file mode 100644 index 00000000..b9d61171 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/direct-tools.mdx @@ -0,0 +1,306 @@ +--- +title: 'Direct Tools' +description: 'Run deterministic host tools without spending an LLM turn' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Direct Tools + +`session.tool(name, args)` (and the typed helpers like `glob`, `grep`, `readFile`) +run a host tool directly, with no model call in the loop. Use them for tests, +migrations, and host-driven workflows where you want deterministic results +instead of an agent turn. + +Direct calls are host control-plane calls. Apply your own product authorization +before invoking them; `permissionPolicy` gates model-selected tool calls inside +an agent turn, not whether your application code is allowed to call an SDK +helper. + + + + +```rust +use a3s_code_core::Agent; +use serde_json::{json, Value}; +use std::collections::HashSet; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let allowed = HashSet::from(["search", "read", "generate_object"]); + let check = |name: &str| -> a3s_code_core::Result<()> { + allowed + .contains(name) + .then_some(()) + .ok_or_else(|| { + a3s_code_core::CodeError::Security(format!( + "direct tool not allowed here: {name}" + )) + }) + }; + + check("search")?; + let files = session.glob("**/*.rs").await?; + println!("glob found {} Rust files", files.len()); + + let matches = session.grep("Agent::new").await?; + println!("grep found {} matching lines", matches.lines().count()); + + check("read")?; + let readme = session.read_file("README.md").await?; + println!("README is {} bytes", readme.len()); + + let raw = session + .tool("read", json!({ "file_path": "Cargo.toml" })) + .await?; + println!("Cargo.toml via tool(): {} bytes", raw.output.len()); + println!("session exposes {} tools", session.tool_definitions().len()); + + check("generate_object")?; + let structured = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["count", "language"], + "properties": { + "count": { "type": "integer" }, + "language": { "type": "string" } + } + }, + "prompt": "How many Rust files are in this project?", + "schema_name": "file_stats" + }), + ) + .await?; + if structured.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "generate_object".into(), + message: structured.output, + }); + } + let generated: Value = serde_json::from_str(&structured.output)?; + println!("structured output: {}", generated["object"]); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const allowedDirectTools = new Set(['glob', 'grep', 'read', 'generate_object']); +function assertDirectToolAllowed(name: string) { + if (!allowedDirectTools.has(name)) { + throw new Error(`direct tool not allowed here: ${name}`); + } +} + +// Glob: list files by pattern +assertDirectToolAllowed('glob'); +const files = await session.glob('**/*.ts'); +console.log(`glob found ${files.length} TypeScript files`); + +// Grep: search file contents +assertDirectToolAllowed('grep'); +const matches = await session.grep('Agent.create'); +const matchCount = matches.split('\n').filter(Boolean).length; +console.log(`grep found ${matchCount} matching lines`); + +// Read a file +assertDirectToolAllowed('read'); +const readme = await session.readFile('README.md'); +console.log(`README is ${readme.length} bytes`); + +// Direct tool call by name +assertDirectToolAllowed('read'); +const raw = await session.tool('read', { file_path: 'package.json' }); +console.log(`package.json via tool(): ${raw.output.length} bytes`); + +// Inspect available tool schemas +const schemas = session.toolDefinitions(); +console.log(`session exposes ${schemas.length} tools`); + +// Structured output: generate a schema-validated JSON object +assertDirectToolAllowed('generate_object'); +const structured = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['count', 'language'], + properties: { + count: { type: 'integer' }, + language: { type: 'string' }, + }, + }, + prompt: 'How many TypeScript files are in this project?', + schema_name: 'file_stats', +}); +if (structured.exitCode !== 0) { + throw new Error(structured.output); +} +console.log('structured output:', JSON.parse(structured.output).object); + +session.close(); +``` + + + + +```python +import json + +from a3s_code import Agent + +agent = Agent.create("agent.acl") +session = agent.session('.') + +ALLOWED_DIRECT_TOOLS = {'glob', 'grep', 'read', 'generate_object'} + + +def assert_direct_tool_allowed(name: str) -> None: + if name not in ALLOWED_DIRECT_TOOLS: + raise RuntimeError(f'direct tool not allowed here: {name}') + +# Glob: list files by pattern +assert_direct_tool_allowed('glob') +files = session.glob('**/*.py') +print(f'glob found {len(files)} Python files') + +# Grep: search file contents +assert_direct_tool_allowed('grep') +matches = session.grep('Agent.create') +match_count = len([line for line in matches.splitlines() if line]) +print(f'grep found {match_count} matching lines') + +# Read a file +assert_direct_tool_allowed('read') +readme = session.read_file('README.md') +print(f'README is {len(readme)} bytes') + +# Direct tool call by name +assert_direct_tool_allowed('read') +raw = session.tool('read', {'file_path': 'pyproject.toml'}) +print(f'pyproject.toml via tool(): {len(raw.output)} bytes') + +# Inspect available tool schemas +schemas = session.tool_definitions() +print(f'session exposes {len(schemas)} tools') + +# Structured output: generate a schema-validated JSON object +assert_direct_tool_allowed('generate_object') +structured = session.tool('generate_object', { + 'schema': { + 'type': 'object', + 'required': ['count', 'language'], + 'properties': { + 'count': {'type': 'integer'}, + 'language': {'type': 'string'}, + }, + }, + 'prompt': 'How many Python files are in this project?', + 'schema_name': 'file_stats', +}) +if structured.exit_code != 0: + raise RuntimeError(structured.output) +print('structured output:', json.loads(structured.output)['object']) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func must[T any](value T, err error) T { + if err != nil { + log.Fatal(err) + } + return value +} + +func main() { + ctx := context.Background() + agent := must(code.Create(ctx, "agent.acl")) + defer agent.Close(context.Background()) + session := must(agent.Session(ctx, ".", nil)) + defer session.Close(context.Background()) + + files := must(session.Glob(ctx, "**/*.go")) + fmt.Printf("glob found %d Go files\n", len(files)) + + matches := must(session.Grep(ctx, "code.Create")) + fmt.Printf("grep found %d matching lines\n", len(strings.Split(matches, "\n"))) + + readme := must(session.ReadFile(ctx, "README.md", nil)) + fmt.Printf("README is %d bytes\n", len(readme)) + + raw := must(session.Tool(ctx, "read", map[string]any{ + "file_path": "go.mod", + })) + fmt.Printf("go.mod via Tool(): %d bytes\n", len(raw.Output)) + + schemas := must(session.ToolDefinitions(ctx)) + fmt.Printf("session exposes %d tools\n", len(schemas)) + + structured := must(session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"count", "language"}, + "properties": map[string]any{ + "count": map[string]any{"type": "integer"}, + "language": map[string]any{"type": "string"}, + }, + }, + "prompt": "How many Go files are in this project?", + "schema_name": "file_stats", + })) + if structured.ExitCode != 0 { + log.Fatal(structured.Output) + } + var generated struct { + Object map[string]any `json:"object"` + } + if err := json.Unmarshal([]byte(structured.Output), &generated); err != nil { + log.Fatal(err) + } + fmt.Println("structured output:", generated.Object) +} +``` + + + + +Direct tools execute under the session workspace and should be treated as +privileged host operations. Most calls (`read`, `glob`, `grep`) are purely +deterministic; `generate_object` is the exception — it still calls the model +to fill a schema-validated JSON object, but you drive it explicitly rather than +through a free-form agent turn. + +A runnable version ships at +`sdk/node/examples/basic/test_generate_object.ts` (Python: +`sdk/python/examples/test_generate_object.py`). The Go direct-tool +contract is covered by `sdk/go/session_test.go`. diff --git a/website/docs/v8.5.1/en/guide/examples/external-tasks.mdx b/website/docs/v8.5.1/en/guide/examples/external-tasks.mdx new file mode 100644 index 00000000..71d9e9c1 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/external-tasks.mdx @@ -0,0 +1,250 @@ +--- +title: 'External Tasks' +description: 'Fulfill agent-queued work from outside the agent process and report structured evidence back' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# External Tasks + +Some work cannot run inside the agent process: it belongs to a separate worker, a CI +runner, or a human in another system. When a lane is routed to an external handler, the +tools on that lane are **queued** instead of executed — they wait as external tasks. Your +host code drains the pending queue, does the work however it likes, and reports the +outcome back with `completeExternalTask`. Reach for this only when an outside worker is +genuinely part of your architecture. + +External tasks are produced by the [lane queue](/guide/examples/lane-queue): you must +register at least one `external` (or `hybrid`) lane handler first, otherwise every task +runs in-process and there is nothing to drain. + + + + +```rust +use a3s_code_core::{ + queue::{ + ExternalTaskResult, LaneHandlerConfig, SessionLane, SessionQueueConfig, + TaskHandlerMode, + }, + Agent, SessionOptions, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_queue_config(SessionQueueConfig::default()), + ) + .build() + .await?; + session + .set_lane_handler( + SessionLane::Execute, + LaneHandlerConfig { + mode: TaskHandlerMode::External, + timeout_ms: 300_000, + }, + ) + .await?; + + for task in session.pending_external_tasks().await { + println!( + "pending: {} on {:?} ({})", + task.task_id, task.lane, task.command_type + ); + let completed = session + .complete_external_task( + &task.task_id, + ExternalTaskResult { + success: true, + result: json!({ + "summary": "worker completed the test run", + "command": "npm run build", + "exit_code": 0 + }), + error: None, + }, + ) + .await; + println!("completed: {completed}"); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { queueConfig: {} }); + +// Route a lane to an external worker so its tools are queued, not executed. +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); + +// Drain the tasks waiting for the host to fulfill. +const pending = await session.pendingExternalTasks(); + +for (const task of pending) { + console.log( + `pending: ${task.task_id} on ${task.lane} (${task.command_type})`, + ); + + try { + // ...the host does the real work here (run CI, call a service, ask a human)... + const ok = await session.completeExternalTask(task.task_id, { + success: true, + result: { + summary: 'worker completed the test run', + command: 'npm run build', + exitCode: 0, + }, + }); + console.log('completed:', ok); + } catch (err) { + await session.completeExternalTask(task.task_id, { + success: false, + error: String(err), + }); + } +} + +session.close(); +``` + + + + +```python +import os +from a3s_code import Agent, SessionOptions, SessionQueueConfig + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.queue_config = SessionQueueConfig() +session = agent.session(os.getcwd(), opts) + +# Route a lane to an external worker so its tools are queued, not executed. +session.set_lane_handler("execute", "external", 300000) + +# Drain the tasks waiting for the host to fulfill. +pending = session.pending_external_tasks() + +for task in pending: + print(f"pending: {task['task_id']} on {task['lane']} ({task['command_type']})") + + try: + # ...the host does the real work here (run CI, call a service, ask a human)... + ok = session.complete_external_task( + task["task_id"], + success=True, + result={ + "summary": "worker completed the test run", + "command": "npm run build", + "exit_code": 0, + }, + ) + print("completed:", ok) + except Exception as err: + session.complete_external_task( + task["task_id"], + success=False, + error=str(err), + ) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + QueueConfig: &code.SessionQueueConfig{}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + err = session.SetLaneHandler(ctx, code.LaneExecute, code.LaneHandlerConfig{ + Mode: "external", + TimeoutMS: 300_000, + }) + if err != nil { + log.Fatal(err) + } + + pending, err := session.PendingExternalTasks(ctx) + if err != nil { + log.Fatal(err) + } + for _, task := range pending { + fmt.Println("pending:", task.TaskID, task.Lane, task.CommandType) + completed, err := session.CompleteExternalTask( + ctx, + task.TaskID, + code.ExternalTaskResult{ + Success: true, + Result: json.RawMessage( + `{"summary":"worker completed the test run","exit_code":0}`, + ), + }, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println("completed:", completed) + } +} +``` + + + + +Notes: + +- Each pending task carries `task_id`, `session_id`, `lane`, `command_type`, `payload`, and + `timeout_ms`. Pass the `task_id` back to `completeExternalTask` / `complete_external_task` + to match the completion to the right task. +- The result shape is `{ success, result?, error? }`. `result` holds any + JSON-serializable payload; `error` is an optional message for failures. +- On success, return compact structured evidence (a summary plus the key facts), not raw + logs only — the agent reasons over the result, so keep it small and machine-readable. +- `completeExternalTask` / `complete_external_task` returns `true` when the task was found + and completed, `false` otherwise. In Python these queue methods are synchronous; in Node + `pendingExternalTasks` and `completeExternalTask` return promises. +- Go uses `PendingExternalTasks` and `CompleteExternalTask`; both accept the + caller's `context.Context`. diff --git a/website/docs/v8.5.1/en/guide/examples/git-worktree.mdx b/website/docs/v8.5.1/en/guide/examples/git-worktree.mdx new file mode 100644 index 00000000..b4fa464f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/git-worktree.mdx @@ -0,0 +1,232 @@ +--- +title: 'Git Worktree' +description: 'Drive git and git worktrees through the session git tool' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Git Worktree + +The session `git` tool runs git as a privileged host operation. It accepts a +structured command object (`command`, plus `subcommand`/`name`/`path` for +worktrees) and returns a tool result with `output` and `exitCode`. This example +inspects a repo, then creates, lists, and removes a worktree directly through the +tool surface. + + + + +```rust +use a3s_code_core::Agent; +use serde_json::json; +use std::path::Path; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder("/path/to/repo").build().await?; + + let status = session.tool("git", json!({ "command": "status" })).await?; + println!("{}", status.output); + + let worktree = Path::new("/path/to/repo").join("wt-feature-auth"); + let created = session + .tool( + "git", + json!({ + "command": "worktree", + "subcommand": "create", + "name": "feature-auth", + "path": worktree + }), + ) + .await?; + if created.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "git".into(), + message: format!("create failed: {}", created.output), + }); + } + + let list = session + .tool( + "git", + json!({ "command": "worktree", "subcommand": "list" }), + ) + .await?; + println!("{}", list.output); + + let removed = session + .tool( + "git", + json!({ + "command": "worktree", + "subcommand": "remove", + "path": worktree + }), + ) + .await?; + if removed.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "git".into(), + message: format!("remove failed: {}", removed.output), + }); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; +import * as path from 'path'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/path/to/repo'); + +// Inspect the repo +const status = await session.git({ command: 'status' }); +console.log(status.output); + +// Create a worktree on a new branch +const wtPath = path.join('/path/to/repo', 'wt-feature-auth'); +const created = await session.git({ + command: 'worktree', + subcommand: 'create', + name: 'feature-auth', + path: wtPath, +}); +if (created.exitCode !== 0) throw new Error(`create failed: ${created.output}`); + +// List worktrees +const list = await session.git({ command: 'worktree', subcommand: 'list' }); +console.log(list.output); + +// Remove the worktree when done +const removed = await session.git({ + command: 'worktree', + subcommand: 'remove', + path: wtPath, +}); +if (removed.exitCode !== 0) throw new Error(`remove failed: ${removed.output}`); + +session.close(); +``` + + + + +```python +import os +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session("/path/to/repo", SessionOptions()) + +# Inspect the repo +status = session.git({"command": "status"}) +print(status.output) + +# Create a worktree on a new branch +wt_path = os.path.join("/path/to/repo", "wt-feature-auth") +created = session.git({ + "command": "worktree", + "subcommand": "create", + "name": "feature-auth", + "path": wt_path, +}) +if created.exit_code != 0: + raise RuntimeError(f"create failed: {created.output}") + +# List worktrees +listing = session.git({"command": "worktree", "subcommand": "list"}) +print(listing.output) + +# Remove the worktree when done +removed = session.git({ + "command": "worktree", + "subcommand": "remove", + "path": wt_path, +}) +if removed.exit_code != 0: + raise RuntimeError(f"remove failed: {removed.output}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "path/filepath" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/repo", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + status, err := session.Git(ctx, code.GitOptions{Command: "status"}) + if err != nil { + log.Fatal(err) + } + fmt.Println(status.Output) + + worktree := filepath.Join("/path/to/repo", "wt-feature-auth") + created, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "create", + Name: "feature-auth", Path: worktree, + }) + if err != nil || created.ExitCode != 0 { + log.Fatalf("create failed: %v %s", err, created.Output) + } + list, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "list", + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(list.Output) + + removed, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "remove", Path: worktree, + }) + if err != nil || removed.ExitCode != 0 { + log.Fatalf("remove failed: %v %s", err, removed.Output) + } +} +``` + + + + +Pass a command object rather than positional arguments: `{ command: 'status' }`, +`{ command: 'diff' }`, or `{ command: 'worktree', subcommand: 'list' }`. Each +call returns a tool result, so check `exit_code` (Rust/Python), `exitCode` +(Node.js), or `ExitCode` (Go) before trusting the output. + +Direct git calls are privileged host operations. Put push, publish, and release +workflows behind application-level approval or automation gates. + +A runnable version ships at `sdk/node/examples/git/test_worktree_git.ts`. diff --git a/website/docs/v8.5.1/en/guide/examples/hooks.mdx b/website/docs/v8.5.1/en/guide/examples/hooks.mdx new file mode 100644 index 00000000..9699234b --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/hooks.mdx @@ -0,0 +1,209 @@ +--- +title: 'Lifecycle Hooks' +description: 'Register, count, and unregister lifecycle event callbacks that observe and gate agent activity.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Lifecycle Hooks + +Hooks let you observe and gate agent activity as it happens. You register a named +callback against a lifecycle event, the runtime invokes it at that point, and the +callback returns a decision such as `{ action: "continue" }`. Use hooks for +auditing, redaction, logging, or enforcing policy without changing the agent's +prompt. + +The lifecycle is symmetric: `registerHook` adds a callback by name, `hookCount` +tells you how many are active, and `unregisterHook` removes one by its name. + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{ + hooks::{ + Hook, HookConfig, HookEvent, HookEventType, HookHandler, HookMatcher, HookResponse, + }, + Agent, +}; + +struct ContinueHandler; + +impl HookHandler for ContinueHandler { + fn handle(&self, event: &HookEvent) -> HookResponse { + println!("observed {}", event.event_type()); + HookResponse::continue_() + } +} + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let hook = Hook::new("observe-env-read", HookEventType::PreToolUse) + .with_matcher(HookMatcher::path("**/.env*")) + .with_config(HookConfig { + priority: 100, + ..HookConfig::default() + }); + session.register_hook(hook)?; + session.register_hook_handler("observe-env-read", Arc::new(ContinueHandler))?; + + println!("active hooks: {}", session.hook_count()); + session + .send("Read the project README and summarize it.", None) + .await?; + + session.unregister_hook_handler("observe-env-read")?; + session.unregister_hook("observe-env-read")?; + println!("active hooks: {}", session.hook_count()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd()); + +// Register a named hook on a lifecycle event. The callback must NOT throw — +// always return a decision such as { action: 'continue' }. +session.registerHook( + 'observe-env-read', + 'pre_tool_use', + { pathPattern: '**/.env*' }, + { priority: 100 }, + () => ({ action: 'continue' }), +); + +console.log('active hooks:', session.hookCount()); // 1 + +await session.run('Read the project README and summarize it.'); + +// Remove the hook by name when you no longer need it. +session.unregisterHook('observe-env-read'); +console.log('active hooks:', session.hookCount()); // 0 + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session('.', SessionOptions()) + +# Register a named hook on a lifecycle event. The callback returns a decision. +session.register_hook( + 'observe-env-read', + 'pre_tool_use', + {'pathPattern': '**/.env*'}, + {'priority': 100}, + lambda: {'action': 'continue'}, +) + +print("active hooks:", session.hook_count()) # 1 + +session.run("Read the project README and summarize it.") + +# Remove the hook by name when you no longer need it. +session.unregister_hook('observe-env-read') +print("active hooks:", session.hook_count()) # 0 + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + priority := 100 + err = session.RegisterHookWithHandler(ctx, code.Hook{ + ID: "observe-env-read", + EventType: "pre_tool_use", + Matcher: &code.HookMatcher{PathPattern: "**/.env*"}, + Config: &code.HookConfig{Priority: &priority}, + }, func(_ context.Context, event json.RawMessage) (*code.HookResponse, error) { + fmt.Println("hook event:", string(event)) + return &code.HookResponse{Action: "continue"}, nil + }) + if err != nil { + log.Fatal(err) + } + count, err := session.HookCount(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("active hooks:", count) + + if _, err = session.Run(ctx, "Read the project README and summarize it."); err != nil { + log.Fatal(err) + } + + if _, err = session.UnregisterHook(ctx, "observe-env-read"); err != nil { + log.Fatal(err) + } +} +``` + + + + +Notes: + +- A hook callback returns a decision. Return `{ action: "continue" }` + (Node) / `{"action": "continue"}` (Python) / + `&code.HookResponse{Action: "continue"}` (Go) to let the agent proceed. +- Permanent denials return `block` with a reason. Temporary denials return + `retry` with a reason and `delayMs` (Node), `delay_ms` (Python), or `DelayMS` + (Go). The denied tool result exposes `error_kind.type = "hook_denied"`, plus + `retryable` and `retry_after_ms`, so hosts do not need to parse output text. +- The matcher (`{ pathPattern: '**/.env*' }`) scopes the hook to events whose + path matches the pattern, and `{ priority: 100 }` orders hooks on the same + event (lower values run first). +- Node hook callbacks must **not** throw — an uncaught throw can abort the + process. Keep the handler body total and always return a decision. +- `hookCount` / `hook_count` / `HookCount` reflects the number of currently + registered hooks, which is handy in tests to assert that registration and + cleanup happened. +- `unregisterHook` / `unregister_hook` / `UnregisterHook` takes the name you registered with. + Always tear down hooks you no longer need so they do not leak across runs. +- Validate the event path you depend on before using a hook as a production gate. diff --git a/website/docs/v8.5.1/en/guide/examples/index.mdx b/website/docs/v8.5.1/en/guide/examples/index.mdx new file mode 100644 index 00000000..857ddc2f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/index.mdx @@ -0,0 +1,24 @@ +--- +title: 'Examples' +description: 'Examples for the current A3S Code runtime and SDK surfaces' +--- + +# Examples + +These examples use the current A3S Code runtime concepts: ACL configuration, +environment-variable credentials, session APIs, streaming, structured output, +task-based and automatic subagent delegation, programmable orchestration +(`parallel` / `pipeline` / `parallelResumable`), `.a3s/agents`, skills, memory, +direct tools, workspace backends, verification, git workflows, +and optional MCP/queue infrastructure. + +Start with the [Quick Start](/guide/examples/quick-start), then explore by area: + +- **Sessions & runtime** — [Quick Start](/guide/examples/quick-start), [Streaming](/guide/examples/streaming), [Model switching](/guide/examples/model-switching), [Auto-compact](/guide/examples/auto-compact) +- **Structured & programmable** — [Structured output](/guide/examples/structured-output), [Orchestration](/guide/examples/orchestration), [Planning](/guide/examples/planning), [Batch](/guide/examples/batch) +- **Tools & context** — [Direct tools](/guide/examples/direct-tools), [ripgrep context](/guide/examples/ripgrep-context), [Prompt slots](/guide/examples/prompt-slots), [Git worktree](/guide/examples/git-worktree) +- **Skills & memory** — [Skills](/guide/examples/skills), [Skill tool](/guide/examples/skill-tool), [Memory](/guide/examples/memory), [Hooks](/guide/examples/hooks) +- **Security & verification** — [Security](/guide/examples/security) +- **MCP & queues** — [Lane queue](/guide/examples/lane-queue), [External tasks](/guide/examples/external-tasks) + +See the [Orchestration](/guide/examples/orchestration) example for fan-out (`parallel`), staged (`pipeline`), and resumable (`parallelResumable`) multi-agent workflows. diff --git a/website/docs/v8.5.1/en/guide/examples/lane-queue.mdx b/website/docs/v8.5.1/en/guide/examples/lane-queue.mdx new file mode 100644 index 00000000..a16c9668 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/lane-queue.mdx @@ -0,0 +1,214 @@ +--- +title: 'Lane Queue' +description: 'Route a lane to an external worker and drain its pending tasks explicitly.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Lane Queue + +By default A3S Code runs every task in-process, with no queue. The lane queue is **optional infrastructure**: register an external handler for a lane and the tools routed to that lane are queued for an outside worker instead of being executed by the agent. You then drain the pending tasks, run them however you like, and report results back. Reach for this only when an external worker is genuinely part of your architecture. + +The four lanes are `control`, `query`, `execute`, and `generate`. Each handler has a `mode` of `internal` (the default), `external`, or `hybrid`. + + + + +```rust +use a3s_code_core::{ + queue::{ + ExternalTaskResult, LaneHandlerConfig, SessionLane, SessionQueueConfig, + TaskHandlerMode, + }, + Agent, SessionOptions, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_queue_config(SessionQueueConfig::default()), + ) + .build() + .await?; + + session + .set_lane_handler( + SessionLane::Execute, + LaneHandlerConfig { + mode: TaskHandlerMode::External, + timeout_ms: 300_000, + }, + ) + .await?; + println!("queue active: {}", session.has_queue()); + + for task in session.pending_external_tasks().await { + println!( + "pending: {} {:?} {}", + task.task_id, task.lane, task.command_type + ); + session + .complete_external_task( + &task.task_id, + ExternalTaskResult { + success: true, + result: json!({ "note": "done by external worker" }), + error: None, + }, + ) + .await; + } + + println!("lane queue drained"); + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('./agent.acl'); +const session = agent.session(process.cwd(), { queueConfig: {} }); + +// Route the "execute" lane to an external worker. +// Tools on this lane are NOT run by the agent; they are queued for +// an outside worker to pick up and complete. +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); + +// hasQueue() is true because this session was created with queueConfig. +console.log('queue active:', session.hasQueue()); + +// Drain whatever is waiting for an external worker. +const pending = await session.pendingExternalTasks(); +for (const task of pending) { + console.log('pending:', task.task_id, task.lane, task.command_type); + + // ... hand off to your worker, run it, then report the outcome back: + await session.completeExternalTask(task.task_id, { + success: true, + result: { note: 'done by external worker' }, + }); +} + +console.log('lane queue drained'); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, SessionQueueConfig + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.queue_config = SessionQueueConfig() +session = agent.session(".", opts) + +# Route the "execute" lane to an external worker. +# Tools on this lane are NOT run by the agent; they are queued for +# an outside worker to pick up and complete. +session.set_lane_handler("execute", "external", 300000) + +# has_queue() is True because this session was created with queue_config. +print("queue active:", session.has_queue()) + +# Drain whatever is waiting for an external worker. +pending = session.pending_external_tasks() +for task in pending: + print("pending:", task["task_id"], task["lane"], task["command_type"]) + + # ... hand off to your worker, run it, then report the outcome back: + session.complete_external_task( + task["task_id"], + success=True, + result={"note": "done by external worker"}, + ) + +print("lane queue drained") +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + QueueConfig: &code.SessionQueueConfig{}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + err = session.SetLaneHandler(ctx, code.LaneExecute, code.LaneHandlerConfig{ + Mode: "external", + TimeoutMS: 300_000, + }) + if err != nil { + log.Fatal(err) + } + active, err := session.HasQueue(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("queue active:", active) + + pending, err := session.PendingExternalTasks(ctx) + if err != nil { + log.Fatal(err) + } + for _, task := range pending { + fmt.Println("pending:", task.TaskID, task.Lane, task.CommandType) + _, err = session.CompleteExternalTask(ctx, task.TaskID, code.ExternalTaskResult{ + Success: true, + Result: json.RawMessage(`{"note":"done by external worker"}`), + }) + if err != nil { + log.Fatal(err) + } + } +} +``` + + + + +Notes: + +- **The default path is queue-free.** `hasQueue()` / `has_queue()` / `HasQueue()` returns `true` only when the session was created with `queueConfig` / `queue_config` / `QueueConfig`. Without that option, setting a lane handler has no queue to mutate. +- Each pending task carries `task_id`, `session_id`, `lane`, `command_type`, `payload`, and `timeout_ms`. Pass the `task_id` back to `completeExternalTask` / `complete_external_task` / `CompleteExternalTask` once the work is done. +- The result shape is `{ success, result?, error? }` — `result` holds any JSON-serializable payload, and `error` is an optional message for failures. `completeExternalTask` / `complete_external_task` returns `true` if the task was found and completed, `false` otherwise. +- In Python these queue methods are synchronous; in Node `setLaneHandler`, `pendingExternalTasks`, and `completeExternalTask` return promises, while `hasQueue` is synchronous. +- Go queue calls accept a `context.Context`; Node returns promises; Python uses + synchronous methods. diff --git a/website/docs/v8.5.1/en/guide/examples/memory.mdx b/website/docs/v8.5.1/en/guide/examples/memory.mdx new file mode 100644 index 00000000..1db37110 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/memory.mdx @@ -0,0 +1,196 @@ +--- +title: 'Memory' +description: 'Remember task outcomes and recall them later by similarity, tags, or recency.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Memory + +Persistent memory lets a session record what worked (and what didn't) and pull +those facts back later. SDK sessions have a default file-backed store at +`/.a3s/memory`; the TUI defaults to `~/.a3s/memory` so its `/memory` +panel and live session browse the same durable store. Pass a `memoryStore` only +when you want to override the path or backend. You can write outcomes explicitly +with `rememberSuccess` / `rememberFailure`, then retrieve them with +`recallSimilar`, `recallByTags`, or `memoryRecent`. + +LLM extraction is also enabled by default. It runs after significant completed +turns to distill durable preferences, workflows, decisions, and failure lessons, +while trivial turns are skipped. +Default stores consolidate exact and conservative near-duplicate memories into a +canonical item, while conflict-like memories remain separate for future recall. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_file_memory("./.a3s/memory")) + .build() + .await?; + let memory = session.memory().expect("memory is configured"); + + memory + .remember_success( + "refactored auth module", + &["read".into(), "edit".into(), "bash".into()], + "all tests passed after extracting AuthService", + ) + .await?; + memory + .remember_failure( + "migration attempt", + "psql connection refused on port 5432", + &["bash".into()], + ) + .await?; + + let recent = memory.get_recent(10).await?; + let by_tags = memory + .recall_by_tags(&["read".into(), "edit".into()], 5) + .await?; + let similar = memory.recall_similar("auth refactor", 5).await?; + println!("{} {} {}", recent.len(), by_tags.len(), similar.len()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileMemoryStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), +}); + +// Record outcomes as the agent works. +await session.rememberSuccess( + 'refactored auth module', + ['read', 'edit', 'bash'], + 'all tests passed after extracting AuthService', +); +await session.rememberFailure( + 'migration attempt', + ['bash'], + 'psql connection refused on port 5432', +); + +// Recall later — by recency, by tool tags, or by semantic similarity. +const recent = await session.memoryRecent(10); +const byTags = await session.recallByTags(['read', 'edit'], 5); +const similar = await session.recallSimilar('auth refactor', 5); + +console.log(recent.length, byTags.length, similar.length); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, FileMemoryStore + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.memory_store = FileMemoryStore("./.a3s/memory") +session = agent.session("/repo", opts) + +# Record outcomes as the agent works. Python helpers are synchronous — no await. +session.remember_success( + "refactored auth module", + ["read", "edit", "bash"], + "all tests passed after extracting AuthService", +) +session.remember_failure( + "migration attempt", + ["bash"], + "psql connection refused on port 5432", +) + +# Recall later — by recency, by tool tags, or by semantic similarity. +recent = session.memory_recent(10) +by_tags = session.recall_by_tags(["read", "edit"], 5) +similar = session.recall_similar("auth refactor", 5) + +print(len(recent), len(by_tags), len(similar)) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + options := &code.SessionOptions{FileMemoryDir: "./.a3s/memory"} + + first, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + if err := first.RememberSuccess( + ctx, + "auth verification rule", + []string{"cargo", "test"}, + "Authentication changes must pass cargo test.", + ); err != nil { + log.Fatal(err) + } + if err := first.Close(ctx); err != nil { + log.Fatal(err) + } + + second, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + defer second.Close(context.Background()) + items, err := second.RecallSimilar( + ctx, + "verification required for authentication changes", + 5, + ) + if err != nil { + log.Fatal(err) + } + for _, item := range items { + fmt.Println(item.Content) + } +} +``` + + + + +All four SDKs expose explicit memory write/query helpers. In Go these are +`RememberSuccess`, `RememberFailure`, `RecallSimilar`, `RecallByTags`, +`MemoryRecent`, and the working/short-term memory methods. The remember methods +take a short task description, the tools involved (also used as tags), and the +outcome text. If the default file store cannot be created, the session falls +back to in-process memory and exposes an init warning. diff --git a/website/docs/v8.5.1/en/guide/examples/model-switching.mdx b/website/docs/v8.5.1/en/guide/examples/model-switching.mdx new file mode 100644 index 00000000..9fdcb4d9 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/model-switching.mdx @@ -0,0 +1,367 @@ +--- +title: 'Model Switching' +description: 'Choose the model per session, and override it per worker agent for cost and capability tuning.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Model Switching + +A session runs against whatever model you pass in the `model` option. Declare +the models your agent can reach once, then pick one per session — a fast model +for high-volume, low-stakes work and a stronger model for review. Use this when +you want to balance cost against capability without changing any of your prompts. + +## Declaring models + +Models are configured in your agent file. Each provider lists the models it +exposes, and `default_model` is used when a session does not set `model`. + +```acl +default_model = "provider/fast-model" + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "fast-model" { tool_call = true } + models "review-model" { tool_call = true } +} +``` + +## Per-session model + +The `model` option is set when you open the session. Everything that session +runs — `send`, `run`, `task`, `parallel`, `pipeline` — uses that model. One +agent configuration can drive different model choices for different sessions. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let fast = agent + .session_builder("/repo") + .options(SessionOptions::new().with_model("provider/fast-model")) + .build() + .await?; + let draft = fast + .send("Draft a short README intro for this project.", None) + .await? + .text; + println!("draft: {draft}"); + fast.close().await; + + let review = agent + .session_builder("/repo") + .options(SessionOptions::new().with_model("provider/review-model")) + .build() + .await?; + let critique = review + .send(&format!("Critique this README intro:\n{draft}"), None) + .await?; + println!("critique: {}", critique.text); + + review.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +// A fast model for high-volume, low-stakes work. +const fast = agent.session('/repo', { model: 'provider/fast-model' }); +const draft = await fast.run('Draft a short README intro for this project.'); +console.log('draft:', draft); +await fast.close(); + +// A stronger model for review / higher-stakes reasoning. +const review = agent.session('/repo', { model: 'provider/review-model' }); +const critique = await review.run(`Critique this README intro:\n${draft}`); +console.log('critique:', critique); +await review.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +# A fast model for high-volume, low-stakes work. +fast_opts = SessionOptions() +fast_opts.model = 'provider/fast-model' +fast = agent.session('/repo', fast_opts) +draft = fast.run('Draft a short README intro for this project.') +print('draft:', draft) +fast.close() + +# A stronger model for review / higher-stakes reasoning. +review_opts = SessionOptions() +review_opts.model = 'provider/review-model' +review = agent.session('/repo', review_opts) +critique = review.run(f'Critique this README intro:\n{draft}') +print('critique:', critique) +review.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + fast, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/fast-model", + }) + if err != nil { + log.Fatal(err) + } + draft, err := fast.Run(ctx, "Draft a short README intro for this project.") + if err != nil { + log.Fatal(err) + } + fmt.Println("draft:", draft.Text) + fast.Close(ctx) + + review, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/review-model", + }) + if err != nil { + log.Fatal(err) + } + defer review.Close(context.Background()) + critique, err := review.Run(ctx, "Critique this README intro:\n"+draft.Text) + if err != nil { + log.Fatal(err) + } + fmt.Println("critique:", critique.Text) +} +``` + + + + +## Per-worker-agent model override + +Worker agents are registered with their own spec. Give a worker its own `model` +so it runs on a different (often smaller, cheaper) model than the session that +delegates to it. The orchestrating session keeps its own `model`; only the +delegated work runs on the worker's model. + + + + +```rust +use a3s_code_core::{ + subagent::ModelConfig as WorkerModelConfig, Agent, SessionOptions, WorkerAgentSpec, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let mut scout = WorkerAgentSpec::read_only( + "scout", + "Reads files and reports findings.", + ); + scout.model = Some(WorkerModelConfig::from_model_ref( + "provider/fast-model", + )); + let session = agent + .session_builder("/repo") + .options( + SessionOptions::new() + .with_model("provider/review-model") + .with_worker_agent(scout), + ) + .build() + .await?; + + let findings = session + .tool( + "task", + json!({ + "agent": "scout", + "description": "List public APIs", + "prompt": "List every public API in src/." + }), + ) + .await?; + let plan = session + .send( + &format!("Given these findings, propose a refactor:\n{}", findings.output), + None, + ) + .await?; + println!("{}", plan.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + // Orchestrator stays on the stronger model. + model: 'provider/review-model', + // High-volume exploration runs on the cheaper model. + workerAgents: [ + { + name: 'scout', + description: 'Reads files and reports findings.', + model: 'provider/fast-model', + }, + ], +}); + +// Delegate exploration to the cheaper worker, then reason on the strong model. +const findings = await session.task({ + agent: 'scout', + description: 'List public APIs', + prompt: 'List every public API in src/.', +}); +const plan = await session.run( + `Given these findings, propose a refactor:\n${findings.output}`, +); +console.log(plan); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, WorkerAgentSpec + +agent = Agent.create("agent.acl") + +opts = SessionOptions() +# Orchestrator stays on the stronger model. +opts.model = 'provider/review-model' +# High-volume exploration runs on the cheaper model. +scout = WorkerAgentSpec( + name='scout', + description='Reads files and reports findings.', +) +scout.model = 'provider/fast-model' +opts.worker_agents = [scout] +session = agent.session('/repo', opts) + +# Delegate exploration to the cheaper worker, then reason on the strong model. +findings = session.task({ + "agent": "scout", + "description": "List public APIs", + "prompt": "List every public API in src/.", +}) +plan = session.run(f'Given these findings, propose a refactor:\n{findings.output}') +print(plan) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/review-model", + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + findings, err := session.Task(ctx, code.DelegateTaskOptions{ + Agent: "explore", + Description: "List public APIs", + Prompt: "List every public API in src/.", + Model: "provider/fast-model", + }) + if err != nil { + log.Fatal(err) + } + plan, err := session.Run( + ctx, + "Given these findings, propose a refactor:\n"+findings.Output, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(plan.Text) +} +``` + + + + +Notes: + +- The `model` value is an identifier string your runtime resolves to one of the + models declared in your agent file — there are no hard-coded model names in the + SDK. +- A worker agent's `model` applies only to that agent's delegated work. The + session's own `send`/`run`/`task` calls still use the session `model`. +- A worker without a `model` inherits the session `model`, so you only override + the agents where a different model actually pays off. + +A runnable version showing the `model` option on a session ships at +`sdk/node/examples/basic/test_api_alignment.ts`. diff --git a/website/docs/v8.5.1/en/guide/examples/orchestration.mdx b/website/docs/v8.5.1/en/guide/examples/orchestration.mdx new file mode 100644 index 00000000..db6dc20c --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/orchestration.mdx @@ -0,0 +1,797 @@ +--- +title: 'Orchestration' +description: 'Fan out independent work with session.parallel, build per-item chains with session.pipeline, and resume journaled runs with session.parallelResumable.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Orchestration + +This page shows the programmable orchestration primitives in A3S Code: `session.parallel` for fan-out, `session.pipeline` for per-item multi-stage chains, and `session.parallelResumable` for journaled runs that survive a crash. `parallel` also takes an optional token budget so a whole fan-out shares one ledger. Use orchestration when you have several independent subagent tasks (parallel), or one transformation that flows through ordered stages per input (pipeline). + +For the conceptual model behind these primitives, see [Orchestration](/guide/orchestration). + +## Fan-out with `session.parallel` + +`parallel` takes an array of `AgentStepSpec` and runs them concurrently, returning one `StepOutcome` per spec **in input order** (not completion order). Each spec routes to a named subagent (`explore`, `plan`, `review`, `verification`, `general`, ...). Set `outputSchema` / `output_schema` on a spec to get a schema-validated `structured` result back. + + + + +```rust +use a3s_code_core::{Agent, AgentStepSpec}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow(); + + let outcomes = workflow + .parallel(vec![ + AgentStepSpec::new( + "langs", + "general", + "list languages", + "Name three systems programming languages.", + ) + .with_max_steps(2), + AgentStepSpec::new( + "verdict", + "general", + "classify", + "Is Rust memory-safe without a GC? Answer yes or no.", + ) + .with_max_steps(2), + ]) + .await; + + for outcome in outcomes { + println!( + "[parallel] {}: success={}", + outcome.task_id, outcome.success + ); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +// Independent steps; outcomes come back in input order, not completion order. +const outcomes = await session.parallel([ + { + taskId: 'langs', + agent: 'general', + description: 'list', + prompt: 'Name three systems languages.', + maxSteps: 2, + }, + { + taskId: 'safe', + agent: 'general', + description: 'classify', + prompt: 'Is Rust memory-safe without a GC? yes/no.', + maxSteps: 2, + }, +]); + +for (const o of outcomes) { + console.log(`[parallel] ${o.taskId}: success=${o.success}`); +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +# Independent steps; outcomes come back in input order, not completion order. +outcomes = session.parallel([ + { + "task_id": "langs", + "agent": "general", + "description": "list languages", + "prompt": "Name three systems programming languages, comma-separated.", + "max_steps": 2, + }, + { + "task_id": "verdict", + "agent": "general", + "description": "classify", + "prompt": "Is Rust memory-safe without a GC? Answer yes or no.", + "max_steps": 2, + # Schema-validated structured output for this step. + "output_schema": { + "type": "object", + "properties": {"memory_safe": {"type": "boolean"}}, + "required": ["memory_safe"], + }, + }, +]) + +for o in outcomes: + print(f"[parallel] {o['task_id']}: success={o['success']} structured={o.get('structured')}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + result, err := session.Parallel(ctx, []code.AgentStepSpec{ + { + TaskID: "langs", + Description: "list languages", + Agent: "general", + Prompt: "Name three systems programming languages, comma-separated.", + MaxSteps: &maxSteps, + }, + { + TaskID: "verdict", + Description: "classify", + Agent: "general", + Prompt: "Is Rust memory-safe without a GC? Answer yes or no.", + MaxSteps: &maxSteps, + OutputSchema: json.RawMessage(`{ + "type":"object", + "properties":{"memory_safe":{"type":"boolean"}}, + "required":["memory_safe"] + }`), + }, + }, nil) + if err != nil { + log.Fatal(err) + } + for _, outcome := range result.Outcomes { + fmt.Printf( + "[parallel] %s: success=%t structured=%s\n", + outcome.TaskID, + outcome.Success, + outcome.Structured, + ) + } +} +``` + + + + +Outcomes are dictionaries in Python, objects in Node, and `StepOutcome` values +in Go. The `maxParallelTasks` / `max_parallel_tasks` / `MaxParallelTasks` +session option caps concurrency; extra specs queue, and the outcome array is +still returned in full, in order. + +## Per-item chains with `session.pipeline` + +`pipeline` takes a list of input `items` and an ordered list of `stages`. Each item flows through the stages independently — there is **no barrier between stages**, so a fast item can reach stage 2 while a slow item is still in stage 1. A stage callback receives a `ctx`: the first stage sees `ctx.item`, later stages see `ctx.previous` (the prior `StepOutcome`, whose `.output` you build on). Return the next spec to continue, or `null` / `None` to stop that item's chain early. + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{Agent, AgentStepSpec, PipelineStage}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow(); + + let stages: Vec> = vec![ + Arc::new(|_, item| { + Some( + AgentStepSpec::new( + "summarize", + "general", + "summarize", + format!("In one sentence, what is {item}?"), + ) + .with_max_steps(2), + ) + }), + Arc::new(|previous, _| { + previous.map(|outcome| { + AgentStepSpec::new( + "classify", + "general", + "classify", + format!( + "Reply YES or NO: does this describe a programming language?\n\n{}", + outcome.output + ), + ) + .with_max_steps(2) + }) + }), + ]; + + let results = workflow + .pipeline(vec!["the Rust programming language".to_string()], stages) + .await; + for result in results.into_iter().flatten() { + println!("[pipeline] {}", result.output); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +// Stage 2 builds on stage 1's output. A stage callback MUST NOT throw — +// return null to stop this item's chain. +const results = await session.pipeline( + ['the Rust programming language'], + [ + (ctx) => ({ + taskId: 'sum', + agent: 'general', + description: 'summarize', + prompt: `In one sentence, what is ${ctx.item}?`, + maxSteps: 2, + }), + (ctx) => ({ + taskId: 'cls', + agent: 'general', + description: 'classify', + prompt: `Reply YES or NO: does this describe a programming language?\n\n${ctx.previous.output}`, + maxSteps: 2, + }), + ], +); + +for (const r of results) { + console.log( + `[pipeline] final=${r === null ? null : JSON.stringify(r.output.slice(0, 60))}`, + ); +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +# Each item chains through stages; stage 2 builds on stage 1. Return None from a +# stage (or raise — caught and treated as None) to stop that item's chain. +results = session.pipeline( + ["the Rust programming language"], + [ + lambda ctx: { + "task_id": "summarize", + "agent": "general", + "description": "summarize", + "prompt": f"In one sentence, what is {ctx['item']}?", + "max_steps": 2, + }, + lambda ctx: { + "task_id": "classify", + "agent": "general", + "description": "classify", + "prompt": "Reply with one word YES or NO: does this describe a " + f"programming language?\n\n{ctx['previous']['output']}", + "max_steps": 2, + }, + ], +) + +for r in results: + print(f"[pipeline] final={None if r is None else r['output'][:60]!r}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + results, err := session.Pipeline( + ctx, + []any{"the Rust programming language"}, + []code.PipelineStage{ + func( + _ context.Context, + stage code.PipelineContext, + ) (*code.AgentStepSpec, error) { + return &code.AgentStepSpec{ + TaskID: "summarize", + Agent: "general", + Description: "summarize", + Prompt: fmt.Sprintf( + "In one sentence, what is %v?", + stage.Item, + ), + MaxSteps: &maxSteps, + }, nil + }, + func( + _ context.Context, + stage code.PipelineContext, + ) (*code.AgentStepSpec, error) { + return &code.AgentStepSpec{ + TaskID: "classify", + Agent: "general", + Description: "classify", + Prompt: "Reply YES or NO: does this describe a programming language?\n\n" + + stage.Previous.Output, + MaxSteps: &maxSteps, + }, nil + }, + }, + ) + if err != nil { + log.Fatal(err) + } + for _, outcome := range results { + if outcome != nil { + fmt.Println(outcome.Output) + } + } +} +``` + + + + +Key difference from `parallel`: stages are ordered and dependent, but items do +**not** wait for each other between stages. Node stage callbacks must never +throw — return `null` on error; Python stages may raise (caught as `None`); Go +stages return `(*AgentStepSpec, error)`. + +## Resumable runs with `session.parallelResumable` + +`parallelResumable` is `parallel` with a journal. It takes the `specs` first and +a stable `workflowId` second; each step's outcome is journaled to the session's +store, so if the process crashes mid-run you can call it again with the same +`workflowId` and completed steps are replayed from the journal instead of +re-executed. It **requires a session store** — pass `sessionStore`, +`session_store`, or Go's `FileSessionStoreDir` when opening the session. + + + + +Rust exposes the same checkpoint behavior through a named `workflow.phase`. +When a session store is configured, every phase is a resumable barrier. + +```rust +use a3s_code_core::{Agent, AgentStepSpec, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_file_session_store("./.a3s/sessions") + .with_session_id("nightly-audit-session"), + ) + .build() + .await?; + let workflow = session.workflow(); + + let outcomes = workflow + .phase( + "nightly-audit", + vec![ + AgentStepSpec::new( + "deps", + "general", + "audit dependencies", + "Check manifests for outdated dependencies.", + ) + .with_max_steps(2), + AgentStepSpec::new( + "tests", + "verification", + "run tests", + "Run the test suite and summarize failures.", + ) + .with_max_steps(2), + ], + ) + .await; + + for outcome in outcomes { + println!("{}:{}", outcome.task_id, outcome.success); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +// parallelResumable journals to the session store; it throws without one. +const session = agent.session('.', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +// Signature is (specs, workflowId): specs first, stable workflowId second. +const outcomes = await session.parallelResumable( + [ + { + taskId: 'deps', + agent: 'general', + description: 'audit deps', + prompt: 'Check manifests for outdated dependencies.', + maxSteps: 2, + }, + { + taskId: 'tests', + agent: 'verification', + description: 'run tests', + prompt: 'Run the test suite and summarize failures.', + maxSteps: 2, + }, + ], + 'nightly-audit', +); + +// After an interrupted run, restart with the same workflowId. Completed steps +// are loaded from the journal; a fully successful run removes its checkpoint. +console.log(outcomes.map((o) => `${o.taskId}:${o.success}`).join(' ')); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, FileSessionStore + +agent = Agent.create("agent.acl") +# parallel_resumable journals to the session store; it raises without one. +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session(".", opts) + +# Signature is (specs, workflow_id): specs first, stable workflow_id second. +outcomes = session.parallel_resumable( + [ + {"task_id": "deps", "agent": "general", "description": "audit deps", "prompt": "Check manifests for outdated dependencies.", "max_steps": 2}, + {"task_id": "tests", "agent": "verification", "description": "run tests", "prompt": "Run the test suite and summarize failures.", "max_steps": 2}, + ], + "nightly-audit", +) + +# After an interruption, restart with the same workflow_id. Completed steps +# are loaded from the journal; a fully successful run removes its checkpoint. +print(" ".join(f"{o['task_id']}:{o['success']}" for o in outcomes)) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + options := &code.SessionOptions{ + SessionID: "nightly-audit-session", + FileSessionStoreDir: "./.a3s/sessions", + } + session, err := agent.Session(ctx, ".", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + outcomes, err := session.ParallelResumable(ctx, []code.AgentStepSpec{ + { + TaskID: "deps", + Agent: "general", + Description: "audit deps", + Prompt: "Check manifests for outdated dependencies.", + MaxSteps: &maxSteps, + }, + { + TaskID: "tests", + Agent: "verification", + Description: "run tests", + Prompt: "Run the test suite and summarize failures.", + MaxSteps: &maxSteps, + }, + }, "nightly-audit") + if err != nil { + log.Fatal(err) + } + for _, outcome := range outcomes { + fmt.Printf("%s:%t ", outcome.TaskID, outcome.Success) + } +} +``` + + + + +## Budgeted fan-out with `session.parallel` + +Pass a token budget and every child agent feeds **one shared ledger**. With a +budget, `parallel` resolves to `{ outcomes, budget }` +(the ledger snapshot) instead of the plain outcomes array; once the cap is hit, a +step that starts afterwards is denied (`success: false`). It is a soft cap — a +wide fan-out can race a few in-flight turns past it; the in-flight work is never +force-killed. + + + + +```rust +use a3s_code_core::{Agent, AgentStepSpec}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow_with_token_budget(Some(50_000)); + + let outcomes = workflow + .parallel(vec![ + AgentStepSpec::new("a", "general", "question one", "Reply: ready.") + .with_max_steps(2), + AgentStepSpec::new("b", "general", "question two", "Reply: go.") + .with_max_steps(2), + ]) + .await; + for outcome in outcomes { + println!("{}: success={}", outcome.task_id, outcome.success); + } + + if let Some(budget) = workflow.budget_snapshot() { + println!( + "spent {} / {} tokens", + budget.consumed_tokens, + budget.limit_tokens.unwrap_or_default() + ); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +const specs = [ + { + taskId: 'a', + agent: 'general', + description: 'q1', + prompt: 'Reply with one word: ready.', + maxSteps: 2, + }, + { + taskId: 'b', + agent: 'general', + description: 'q2', + prompt: 'Reply with one word: go.', + maxSteps: 2, + }, +]; + +// With a budget, parallel() resolves to { outcomes, budget } — all children +// share one ledger. (Without it, parallel(specs) returns the plain array.) +const { outcomes, budget } = await session.parallel(specs, 50_000); +for (const o of outcomes) + console.log(`[budget] ${o.taskId}: success=${o.success}`); +console.log(`spent ${budget.consumedTokens} / ${budget.limitTokens} tokens`); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +specs = [ + {"task_id": "a", "agent": "general", "description": "q1", "prompt": "Reply with one word: ready.", "max_steps": 2}, + {"task_id": "b", "agent": "general", "description": "q2", "prompt": "Reply with one word: go.", "max_steps": 2}, +] + +# With a budget, parallel() returns {"outcomes", "budget"} (a shared ledger). +res = session.parallel(specs, budget_tokens=50_000) +for o in res["outcomes"]: + print(f"[budget] {o['task_id']}: success={o['success']}") +print(f"spent {res['budget']['consumed_tokens']} / {res['budget']['limit_tokens']} tokens") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + maxSteps := uint(2) + budgetTokens := uint64(50_000) + result, err := session.Parallel(ctx, []code.AgentStepSpec{ + {TaskID: "a", Agent: "general", Description: "q1", Prompt: "Reply with one word: ready.", MaxSteps: &maxSteps}, + {TaskID: "b", Agent: "general", Description: "q2", Prompt: "Reply with one word: go.", MaxSteps: &maxSteps}, + }, &budgetTokens) + if err != nil { + log.Fatal(err) + } + for _, outcome := range result.Outcomes { + fmt.Printf("[budget] %s: success=%t\n", outcome.TaskID, outcome.Success) + } + if result.Budget != nil && result.Budget.LimitTokens != nil { + fmt.Printf( + "spent %d / %d tokens\n", + result.Budget.ConsumedTokens, + *result.Budget.LimitTokens, + ) + } +} +``` + + + + +Notes: + +- All three primitives return outcomes aligned to input order. Go uses + `StepOutcome`; Node uses objects; Python uses dictionaries. +- Set `outputSchema` / `output_schema` / `OutputSchema` on a spec to get a + parsed result back in `structured` / `Structured`. +- `maxSteps` / `max_steps` / `MaxSteps` caps the steps per subagent; + `maxParallelTasks` / `max_parallel_tasks` / `MaxParallelTasks` caps fan-out + concurrency. +- Pass a token budget to `parallel` to cap a whole fan-out against one shared + ledger. In Go, pass `*uint64` as the third `Parallel` argument and read + `ParallelResult.Budget`. It is a soft cap (usage is recorded after each call). +- Node pipeline stage callbacks must never throw — return `null` on error. + Python stages may raise (caught as `None`); Go stages return an error. + +Runnable Node.js and Python versions ship at +`sdk/node/examples/orchestration/parallel-pipeline.mjs` and +`sdk/python/examples/orchestration_workflow.py`; the Rust and Go tabs above are +self-contained. diff --git a/website/docs/v8.5.1/en/guide/examples/planning.mdx b/website/docs/v8.5.1/en/guide/examples/planning.mdx new file mode 100644 index 00000000..38d4b11f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/planning.mdx @@ -0,0 +1,140 @@ +--- +title: 'Planning' +description: 'Make the agent plan before it acts with planningMode' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Planning + +Planning mode tells the session to produce a structured plan before it starts +calling tools. Use it for multi-step work (refactors, release reviews, audits) +where you want the agent to decompose the goal first instead of jumping straight +into edits. + +Set it through the session planning option: `PlanningMode` in Rust and Go, +`planningMode` in Node.js, and `planning_mode` in Python. The accepted values +are: + +| Value | Behavior | +| ------------ | --------------------------------------------------------------- | +| `"auto"` | The runtime detects from the message when a plan is worthwhile. | +| `"enabled"` | Force a plan on every request, even simple ones. | +| `"disabled"` | Skip planning entirely for the lowest-latency path. | + +`"auto"` is the default structured pre-analysis path. + + + + +```rust +use a3s_code_core::{Agent, PlanningMode, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_planning_mode(PlanningMode::Auto)) + .build() + .await?; + + let result = session + .send("Plan and complete the release-readiness review.", None) + .await?; + println!("{}", result.text); + println!("{} tool calls executed", result.tool_calls_count); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + planningMode: 'auto', +}); + +const result = await session.send( + 'Plan and complete the release-readiness review.', +); +console.log(result.text); +console.log(`${result.toolCallsCount} tool calls executed`); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.planning_mode = "auto" +session = agent.session("/repo", opts) + +result = session.send("Plan and complete the release-readiness review.") +print(result.text) +print(f"{result.tool_calls_count} tool calls executed") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PlanningMode: code.PlanningAuto, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "Plan and complete the release-readiness review.") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) + fmt.Printf("%d tool calls executed\n", result.ToolCallsCount) +} +``` + + + + +Planning state is attached to the run, so a host UI can render a task list from +run events and update completion as the agent works. Planning organizes the +work; verification commands still provide the completion evidence. + +Runnable session examples ship at +`sdk/node/examples/orchestration/parallel-pipeline.mjs` and +`sdk/python/examples/orchestration_workflow.py`. diff --git a/website/docs/v8.5.1/en/guide/examples/prompt-slots.mdx b/website/docs/v8.5.1/en/guide/examples/prompt-slots.mdx new file mode 100644 index 00000000..745b99ac --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/prompt-slots.mdx @@ -0,0 +1,366 @@ +--- +title: 'Prompt Slots' +description: "Customize the agent's persona, guidelines, and response style without overriding core behavior." +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Prompt Slots + +Prompt slots are session options that shape the agent's system prompt declaratively. Use +them for host-level behavior — persona, coding standards, output style — that should not +live inside each user prompt. The slots layer on top of the agent's built-in instructions, +so core tool behavior (reading, writing, running commands) is preserved. + +There are four slots: + +| Slot | Purpose | +| ---------------------------------- | ------------------------------------------ | +| `role` / `role` | The persona the agent adopts. | +| `guidelines` / `guidelines` | Standards and rules the agent must follow. | +| `responseStyle` / `response_style` | How the agent should format its replies. | +| `extra` / `extra` | Freeform instructions appended verbatim. | + +## Basic usage + +Set any subset of the slots when you open a session. They apply to every turn of that +session. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, SystemPromptSlots}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let slots = SystemPromptSlots::default() + .with_role("release-readiness reviewer") + .with_guidelines( + "Find blockers before improvements. Require command evidence for done claims.", + ) + .with_response_style("concise, findings first"); + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots(slots)) + .build() + .await?; + + let result = session.send("Is this repo ready to ship?", None).await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +async function main() { + // create() accepts an .acl file path or inline ACL string. + const agent = await Agent.create('agent.acl'); + + const session = agent.session('/repo', { + role: 'release-readiness reviewer', + guidelines: + 'Find blockers before improvements. Require command evidence for done claims.', + responseStyle: 'concise, findings first', + }); + + const result = await session.send('Is this repo ready to ship?'); + console.log(result.text); + + session.close(); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +def main(): + # create() accepts an .acl file path or inline ACL string. + agent = Agent.create('agent.acl') + + opts = SessionOptions() + opts.role = 'release-readiness reviewer' + opts.guidelines = 'Find blockers before improvements. Require command evidence for done claims.' + opts.response_style = 'concise, findings first' + session = agent.session('/repo', opts) + + result = session.send('Is this repo ready to ship?') + print(result.text) + + session.close() + +main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PromptSlots: &code.PromptSlots{ + Role: "release-readiness reviewer", + Guidelines: "Find blockers before improvements. " + + "Require command evidence for done claims.", + ResponseStyle: "concise, findings first", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "Is this repo ready to ship?") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +## Each slot in turn + +The four slots compose independently. A persona-only session, a reviewer with strict +guidelines, and a session that appends a freeform instruction all use the same option set. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, SystemPromptSlots}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let role_only = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default().with_role( + "You are a senior Rust developer who specializes in async programming.", + ), + )) + .build() + .await?; + role_only.close().await; + + let reviewer = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default() + .with_role("You are a Python code reviewer.") + .with_guidelines("Always check for type hints. Flag any use of `eval()`.") + .with_response_style("Reply in bullet points. Be concise."), + )) + .build() + .await?; + reviewer.close().await; + + let extra_only = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default().with_extra("Always end your response with '-- A3S'"), + )) + .build() + .await?; + extra_only.close().await; + + let file_manager = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default() + .with_role("You are a minimalist file manager.") + .with_guidelines("Only create files when explicitly asked."), + )) + .build() + .await?; + let result = file_manager + .send( + "Create test.txt with 'prompt slots work', then read it back.", + None, + ) + .await?; + println!("{}", result.text); + + file_manager.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +// 1. Custom role only. +let session = agent.session(workspace, { + role: 'You are a senior Rust developer who specializes in async programming.', +}); + +// 2. Role + guidelines + response style. +session = agent.session(workspace, { + role: 'You are a Python code reviewer.', + guidelines: 'Always check for type hints. Flag any use of `eval()`.', + responseStyle: 'Reply in bullet points. Be concise.', +}); + +// 3. Extra freeform instructions only. +session = agent.session(workspace, { + extra: "Always end your response with '-- A3S'", +}); + +// Core tool behavior is preserved regardless of the slots. +session = agent.session(workspace, { + role: 'You are a minimalist file manager.', + guidelines: 'Only create files when explicitly asked.', +}); +const result = await session.send( + "Create a file called test.txt with the content 'prompt slots work'. Then read it back.", +); +``` + + + + +```python +# 1. Custom role only. +opts = SessionOptions() +opts.role = 'You are a senior Rust developer who specializes in async programming.' +session = agent.session(workspace, opts) + +# 2. Role + guidelines + response style. +opts = SessionOptions() +opts.role = 'You are a Python code reviewer.' +opts.guidelines = 'Always check for type hints. Flag any use of `eval()`.' +opts.response_style = 'Reply in bullet points. Be concise.' +session = agent.session(workspace, opts) + +# 3. Extra freeform instructions only. +opts = SessionOptions() +opts.extra = "Always end your response with '-- A3S'" +session = agent.session(workspace, opts) + +# Core tool behavior is preserved regardless of the slots. +opts = SessionOptions() +opts.role = 'You are a minimalist file manager.' +opts.guidelines = 'Only create files when explicitly asked.' +session = agent.session(workspace, opts) +result = session.send( + "Create a file called test.txt with the content 'prompt slots work'. Then read it back.", +) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func open( + ctx context.Context, + agent *code.Agent, + slots code.PromptSlots, +) *code.Session { + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PromptSlots: &slots, + }) + if err != nil { + log.Fatal(err) + } + return session +} + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + roleOnly := open(ctx, agent, code.PromptSlots{ + Role: "You are a senior Rust developer who specializes in async programming.", + }) + roleOnly.Close(ctx) + + reviewer := open(ctx, agent, code.PromptSlots{ + Role: "You are a Python code reviewer.", + Guidelines: "Always check for type hints. Flag any use of `eval()`.", + ResponseStyle: "Reply in bullet points. Be concise.", + }) + reviewer.Close(ctx) + + extraOnly := open(ctx, agent, code.PromptSlots{ + Extra: "Always end your response with '-- A3S'", + }) + extraOnly.Close(ctx) + + fileManager := open(ctx, agent, code.PromptSlots{ + Role: "You are a minimalist file manager.", + Guidelines: "Only create files when explicitly asked.", + }) + defer fileManager.Close(context.Background()) + result, err := fileManager.Run( + ctx, + "Create test.txt with 'prompt slots work', then read it back.", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +Slots customize personality and house rules; they do not disable tools or change the +agent's core loop. Keep task-specific requests in the `send` message and reserve the slots +for behavior that should hold across every turn of the session. + +A runnable version ships at `sdk/node/examples/skills/test_prompt_slots.ts`. diff --git a/website/docs/v8.5.1/en/guide/examples/quick-start.mdx b/website/docs/v8.5.1/en/guide/examples/quick-start.mdx new file mode 100644 index 00000000..7971e6dc --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/quick-start.mdx @@ -0,0 +1,126 @@ +--- +title: 'Quick Start' +description: 'Create an agent, open a session, run one turn, and read the result.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Quick Start + +The smallest useful program: create an agent from an ACL file, open a session on a +project directory, run one turn with `send`, print the reply text, and inspect the +verification summary the runtime produced for that turn. + + + + +```rust +use a3s_code_core::Agent; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + // Agent::new accepts an .acl file path or inline ACL source. + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let result = session + .send("List the files in this directory.", None) + .await?; + println!("{}", result.text); + + // What the runtime checked while producing that turn. + println!("{}", result.verification_summary_text()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +// Agent.create accepts an .acl file path or inline ACL string. +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const result = await session.send('List the files in this directory.'); +console.log(result.text); + +// What the runtime checked while producing that turn. +console.log(result.verificationSummaryText); + +session.close(); +``` + + + + +```python +from a3s_code import Agent + +# Agent.create accepts an .acl file path or inline ACL string. +agent = Agent.create('agent.acl') +session = agent.session('.') + +result = session.send('List the files in this directory.') +print(result.text) + +# What the runtime checked while producing that turn. +print(result.verification_summary_text) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "List the files in this directory.") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) + fmt.Println(result.VerificationSummaryText) +} +``` + + + + +The agent factory accepts either an `.acl` file path or inline ACL source text +in all four SDKs. Always close the session when you are done so the runtime +can flush state and release resources. Go close operations accept a context +because they also stop the bridge-owned native resources. + +## Next steps + +- [Streaming](/guide/examples/streaming) — read tokens as they arrive +- [Sessions](/guide/sessions) — persist and resume conversations diff --git a/website/docs/v8.5.1/en/guide/examples/ripgrep-context.mdx b/website/docs/v8.5.1/en/guide/examples/ripgrep-context.mdx new file mode 100644 index 00000000..90125419 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/ripgrep-context.mdx @@ -0,0 +1,160 @@ +--- +title: 'Ripgrep Context Builder' +description: 'Use grep and glob to gather code context before asking the agent.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Ripgrep Context Builder + +Fast code search with `session.grep` and `session.glob` lets you gather the +relevant files and matching lines, then feed them into a prompt — a lightweight +retrieval step before the agent reasons. Use this when you want to scope the +agent to a specific slice of a large codebase instead of letting it explore from +scratch. + + + + +```rust +use a3s_code_core::Agent; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let files = session.glob("src/**/*.rs").await?; + let hits = session.grep("create_session").await?; + let context = format!( + "Files in scope:\n{}\n\nMatches for \"create_session\":\n{}", + files.join("\n"), + hits + ); + let answer = session + .send( + &format!( + "Using only this context, explain how create_session is wired up:\n\n{context}" + ), + None, + ) + .await?; + println!("{}", answer.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +// 1. Find candidate files by glob pattern (returns a list of paths). +const files = await session.glob('src/**/*.ts'); + +// 2. Search the workspace for the symbol we care about (returns ripgrep text). +const hits = await session.grep('createSession'); + +// 3. Build a context string and feed it into a focused prompt. +const context = [ + `Files in scope:\n${files.join('\n')}`, + `Matches for "createSession":\n${hits}`, +].join('\n\n'); + +const answer = await session.run( + `Using only this context, explain how createSession is wired up:\n\n${context}`, +); +console.log(answer); +``` + + + + +```python +from a3s_code import Agent + +agent = Agent.create("agent.acl") +session = agent.session('.') + +# 1. Find candidate files by glob pattern (returns a list of paths). +files = session.glob('src/**/*.ts') + +# 2. Search the workspace for the symbol we care about (returns ripgrep text). +hits = session.grep('createSession') + +# 3. Build a context string and feed it into a focused prompt. +context = '\n\n'.join([ + 'Files in scope:\n' + '\n'.join(files), + 'Matches for "createSession":\n' + hits, +]) + +answer = session.run( + f'Using only this context, explain how createSession is wired up:\n\n{context}' +) +print(answer) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + files, err := session.Glob(ctx, "sdk/go/**/*.go") + if err != nil { + log.Fatal(err) + } + hits, err := session.Grep(ctx, "Session") + if err != nil { + log.Fatal(err) + } + scope := "Files in scope:\n" + strings.Join(files, "\n") + + "\n\nMatches for \"Session\":\n" + hits + answer, err := session.Run( + ctx, + "Using only this context, explain how Session is wired up:\n\n"+scope, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(answer.Text) +} +``` + + + + +`glob` returns a list of matching file paths, while `grep` returns the raw +ripgrep output as a single string. Both run locally and return quickly, so you +can chain several searches to assemble context cheaply before spending a model +turn. Pair them with `session.readFile` when you need the full body of a file +rather than just the matching lines. diff --git a/website/docs/v8.5.1/en/guide/examples/security.mdx b/website/docs/v8.5.1/en/guide/examples/security.mdx new file mode 100644 index 00000000..cb2fd3de --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/security.mdx @@ -0,0 +1,441 @@ +--- +title: 'Security' +description: 'Gate privileged operations with a permission policy, a human-in-the-loop confirmation flow, and a security provider' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Security + +Every side effect an agent can produce — writing files, running `bash`, pushing +git — flows through a permission policy. Start from an `ask` or `deny` +fallback, then list the patterns that should be `allow`-ed, `deny`-ed, or sent +to the `ask` path. To keep a human in the loop, add a confirmation policy: +`ask` decisions pause on a `confirmation_required` event so your application +(or a person) can approve or reject each call. Use this whenever an agent runs +against a real repository. + + + + +```rust +use a3s_code_core::{ + hitl::{ConfirmationPolicy, TimeoutAction}, + permissions::PermissionPolicy, + Agent, AgentEvent, SessionOptions, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let permission_policy = PermissionPolicy::new() + .allow("read(*)") + .allow("search(*)") + .allow("ls(*)") + .allow("bash(git status:*)") + .deny("write(**/.env*)") + .deny("bash(rm -rf*)") + .ask("write(*)") + .ask("edit(*)") + .ask("bash(git push:*)") + .ask("bash(npm publish:*)") + .ask("bash(*)"); + let confirmation_policy = + ConfirmationPolicy::enabled().with_timeout(120_000, TimeoutAction::Reject); + + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_permission_policy(permission_policy) + .with_confirmation_policy(confirmation_policy), + ) + .build() + .await?; + + let (mut events, lifecycle) = session + .stream("Bump the version and push the release.", None) + .await?; + while let Some(event) = events.recv().await { + if let AgentEvent::ConfirmationRequired { + tool_id, + tool_name, + args, + .. + } = event + { + println!("[confirm] {tool_name}\n{args:#}"); + session + .confirm_tool_use( + &tool_id, + false, + Some("Rejected by the host review".into()), + ) + .await?; + } + } + let _ = lifecycle.await; + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session(process.cwd(), { + permissionPolicy: { + allow: ['read(*)', 'search(*)', 'ls(*)', 'bash(git status:*)'], + deny: ['write(**/.env*)', 'bash(rm -rf*)'], + ask: [ + 'write(*)', + 'edit(*)', + 'bash(git push:*)', + 'bash(npm publish:*)', + 'bash(*)', + ], + defaultDecision: 'ask', + }, + // Turn the `ask` patterns into a human-in-the-loop confirmation flow. + confirmationPolicy: { + enabled: true, + defaultTimeoutMs: 120000, + timeoutAction: 'reject', + }, +}); + +// Stream execution and resolve confirmations as they arrive. +const stream = await session.stream('Bump the version and push the release'); +while (true) { + const next = await stream.next(); + if (next.done || !next.value) break; + + const event = next.value; + if (event.type === 'confirmation_required') { + // Look up the pending request for richer display. + const [pending] = await session.pendingConfirmations(); + const toolId = pending?.toolId ?? event.toolId; + console.log(`[confirm] ${pending?.toolName ?? event.toolName}`); + console.log(JSON.stringify(pending?.args ?? {}, null, 2)); + + // In a real app, prompt the user here. + const approved = false; // deny risky operations by default + if (toolId) + await session.confirmToolUse(toolId, approved, 'Reviewed by host'); + } +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, PermissionPolicy, ConfirmationPolicy + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.permission_policy = PermissionPolicy( + allow=["read(*)", "search(*)", "ls(*)", "bash(git status:*)"], + deny=["write(**/.env*)", "bash(rm -rf*)"], + ask=["write(*)", "edit(*)", "bash(git push:*)", "bash(npm publish:*)", "bash(*)"], + default_decision="ask", + ) + # Turn the `ask` patterns into a human-in-the-loop confirmation flow. + opts.confirmation_policy = ConfirmationPolicy( + enabled=True, + default_timeout_ms=120_000, + timeout_action="reject", + ) + + session = agent.session(".", opts) + + # Stream execution and resolve confirmations as they arrive. + for event in session.stream("Bump the version and push the release"): + if event.event_type == "confirmation_required": + # Look up the pending request for richer display. + pending = session.pending_confirmations() + first = pending[0] if pending else {} + tool_id = first.get("tool_id") or event.tool_id + print(f"[confirm] {first.get('tool_name') or event.tool_name}") + + # In a real app, prompt the user here. + approved = False # deny risky operations by default + if tool_id: + session.confirm_tool_use(tool_id, approved, "Reviewed by host") + + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + enabled := true + timeoutMS := uint64(120_000) + session, err := agent.Session( + ctx, + ".", + &code.SessionOptions{ + PermissionPolicy: &code.PermissionPolicy{ + Allow: []string{"read(*)", "search(*)", "ls(*)", "bash(git status:*)"}, + Deny: []string{"write(**/.env*)", "bash(rm -rf*)"}, + Ask: []string{"write(*)", "edit(*)", "bash(git push:*)", "bash(npm publish:*)", "bash(*)"}, + DefaultDecision: "ask", + }, + ConfirmationPolicy: &code.ConfirmationPolicy{ + Enabled: &enabled, + DefaultTimeoutMS: &timeoutMS, + TimeoutAction: "reject", + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + stream, err := session.Stream(ctx, "Bump the version and push the release.", nil) + if err != nil { + log.Fatal(err) + } + for event := range stream.Events { + if event.Type != code.EventConfirmationRequired { + continue + } + pending, err := session.PendingConfirmations(ctx) + if err != nil { + log.Fatal(err) + } + if len(pending) == 0 { + continue + } + request := pending[0] + args, _ := json.MarshalIndent(request.Args, "", " ") + fmt.Printf("[confirm] %s\n%s\n", request.ToolName, args) + if _, err = session.ConfirmToolUse( + ctx, + request.ToolID, + false, + "Rejected by the Go host review", + ); err != nil { + log.Fatal(err) + } + } + if err := <-stream.Done; err != nil { + log.Fatal(err) + } +} +``` + + + + +## Add a security provider + +A `DefaultSecurityProvider` enables input taint tracking and output sanitisation, +screening tool I/O independently of the permission policy. Pass one through +`securityProvider` (Node), `security_provider` (Python), or set Go +`SessionOptions.DefaultSecurity` to `true`; omit it to disable security. + + + + +```rust +use a3s_code_core::{permissions::PermissionPolicy, Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_default_security() + .with_permission_policy( + PermissionPolicy::new() + .allow("bash(echo:*)") + .ask("bash(*)"), + ), + ) + .build() + .await?; + + let result = session + .send( + "Use bash to run exactly: echo screened-by-security-provider", + None, + ) + .await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, DefaultSecurityProvider } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session(process.cwd(), { + securityProvider: new DefaultSecurityProvider(), + permissionPolicy: { + allow: ['bash(echo:*)'], + ask: ['bash(*)'], + defaultDecision: 'ask', + }, +}); + +// A model-selected tool call runs through both the provider and the policy. +const result = await session.run( + 'Use bash to run exactly: echo screened-by-security-provider', +); +console.log(result.text); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, PermissionPolicy, DefaultSecurityProvider + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.security_provider = DefaultSecurityProvider() + opts.permission_policy = PermissionPolicy( + allow=["bash(echo:*)"], + ask=["bash(*)"], + default_decision="ask", + ) + + session = agent.session(".", opts) + + # A model-selected tool call runs through both the provider and the policy. + result = session.run( + "Use bash to run exactly: echo screened-by-security-provider" + ) + print(result.text) + + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + DefaultSecurity: code.Ptr(true), + PermissionPolicy: &code.PermissionPolicy{ + Allow: []string{"bash(echo:*)"}, + Ask: []string{"bash(*)"}, + DefaultDecision: "ask", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + result, err := session.Run( + ctx, + "Use bash to run exactly: echo screened-by-security-provider", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + +Arbitrary custom `SecurityProvider` trait implementations remain a Rust-native +extension point; all SDKs expose the built-in provider shown here. + + + + +## Notes + +- `defaultDecision` is the fallback for any pattern not matched by `allow` / `deny` + / `ask` (one of `allow`, `deny`, or `ask`). Prefer `ask` for real repositories + and open up only what automation needs. +- A `confirmationPolicy` with `enabled: true` is what turns `ask` decisions into a + pausing `confirmation_required` event. Resolve each one with + `session.confirmToolUse(toolId, approved, reason?)`; if no answer arrives within + `defaultTimeoutMs`, `timeoutAction` (`reject`) decides the outcome. +- Keep release and publish actions (`bash(git push*)`, `bash(npm publish*)`) on the + `ask` or `deny` path unless automation owns the final step. +- Direct host calls such as `session.tool()`, `session.bash()`, and `session.git()` + are privileged host operations initiated by your application code. Authorize + them in the host before calling the SDK; the permission policy above gates + model-selected tool calls inside `send`, `run`, and `stream`. + +A runnable confirmation loop ships at +`sdk/node/examples/streaming/hitl_confirmation_loop.ts` and +`sdk/python/examples/hitl_confirmation_loop.py`. diff --git a/website/docs/v8.5.1/en/guide/examples/skill-tool.mdx b/website/docs/v8.5.1/en/guide/examples/skill-tool.mdx new file mode 100644 index 00000000..f9a1e347 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/skill-tool.mdx @@ -0,0 +1,200 @@ +--- +title: 'Skill Tool' +description: 'Invoke a registered skill as a callable tool' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Skill Tool + +Skills surface to the model as two core tools: `search_skills` (find a skill by intent) +and `Skill` (invoke a skill by name). A tool-kind skill runs its handler; an +instruction-kind skill returns its body for the model to apply. You can let the model +call these tools during a run, or invoke a skill directly from the SDK with +`session.tool('Skill', ...)`. + +Register a skill directory (a folder of `SKILL.md` files) via `skillDirs` / `skill_dirs`, +or pass inline skills through session options. A3S Code no longer ships default +embedded skills, so `builtinSkills` / `builtin_skills` is only a compatibility +flag. Registered skills become visible through `Skill` and `search_skills`. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_skill_dirs(["./skills"])) + .build() + .await?; + + println!("{:?}", session.tool_names()); + let run = session + .send( + "Search available skills, then apply the most relevant one.", + None, + ) + .await?; + println!("{}", run.text); + + let result = session + .tool( + "Skill", + json!({ + "skill_name": "release-review", + "prompt": "Review this release patch for blockers and verification gaps." + }), + ) + .await?; + if result.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "Skill".into(), + message: result.output, + }); + } + println!("{}", result.output); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { + skillDirs: ['./skills'], // a folder of SKILL.md files +}); + +// Skill and search_skills are core tools — confirm they're on the surface. +console.log(session.toolNames()); + +// Option A: let the model search for and apply a skill during a run. +const run = await session.run( + 'Search available skills, then apply the most relevant one.', +); +console.log(run.text); + +// Option B: invoke a skill directly as a callable tool. +// Canonical args: { skill_name, prompt? }. +const result = await session.tool('Skill', { + skill_name: 'release-review', + prompt: 'Review this release patch for blockers and verification gaps.', +}); +if (result.exitCode !== 0) { + throw new Error(result.output); +} +console.log(result.output); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.skill_dirs = ['./skills'] # a folder of SKILL.md files +session = agent.session('.', opts) + +# Skill and search_skills are core tools — confirm they're on the surface. +print(session.tool_names()) + +# Option A: let the model search for and apply a skill during a run. +run = session.run('Search available skills, then apply the most relevant one.') +print(run.text) + +# Option B: invoke a skill directly as a callable tool. +# Canonical args: { skill_name, prompt? }. +result = session.tool('Skill', { + 'skill_name': 'release-review', + 'prompt': 'Review this release patch for blockers and verification gaps.', +}) +if result.exit_code != 0: + raise RuntimeError(result.output) +print(result.output) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + SkillDirs: []string{"./skills"}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + names, err := session.ToolNames(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println(names) + run, err := session.Run( + ctx, + "Search available skills, then apply the most relevant one.", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(run.Text) + + result, err := session.Tool(ctx, "Skill", map[string]any{ + "skill_name": "release-review", + "prompt": "Review this release patch for blockers and verification gaps.", + }) + if err != nil { + log.Fatal(err) + } + if result.ExitCode != 0 { + log.Fatal(result.Output) + } + fmt.Println(result.Output) +} +``` + + + + +A `SKILL.md` declares its `kind` in frontmatter (`tool`, `instruction`, or `agent`). +For a tool-kind skill, the `Skill` tool runs the skill's handler and returns its output; +for an instruction-kind skill, it returns the body for the model to apply. Skill +administration is handled by SDK registration, skill directories, or project files — not +by a model-visible management tool. + +A runnable version ships at `sdk/node/examples/skills/test_custom_skills_agents.ts`. diff --git a/website/docs/v8.5.1/en/guide/examples/skills.mdx b/website/docs/v8.5.1/en/guide/examples/skills.mdx new file mode 100644 index 00000000..57c8e24a --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/skills.mdx @@ -0,0 +1,188 @@ +--- +title: 'Skills & Custom Agents' +description: 'Load project skills and custom subagents from filesystem conventions.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Skills & Custom Agents + +A3S Code does not ship default embedded skills. Load your own skills and +subagents from directories on disk. Use `skillDirs` for Markdown skills and +`agentDirs` for worker/subagent definitions. `registerAgentDir` can add more +agent definition directories after the session exists; skill directories are +loaded when the session is created. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; +use std::path::Path; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/path/to/project") + .options( + SessionOptions::new() + .with_skill_dirs(["./.a3s/skills"]) + .with_agent_dir("./.a3s/agents"), + ) + .build() + .await?; + + session.register_agent_dir(Path::new("./team/shared-agents"))?; + println!("Tools: {:?}", session.tool_names()); + println!("Skills: {:?}", session.skill_names()); + + let result = session + .send( + "Use the project conventions skill to scaffold a new module.", + None, + ) + .await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +// Add project skills and agents explicitly. +const session = agent.session('/path/to/project', { + skillDirs: ['./.a3s/skills'], + agentDirs: ['./.a3s/agents'], +}); + +// You can register more agent definition directories after the session exists. +session.registerAgentDir('./team/shared-agents'); + +// Inspect what the session loaded. +console.log('Tools:', session.toolNames()); +console.log('Commands:', session.listCommands()); + +// The agent now has access to the project skills. +const result = await session.run( + 'Use the project conventions skill to scaffold a new module.', +); +console.log(result.text); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +# Add project skills and agents explicitly. +opts = SessionOptions() +opts.skill_dirs = ['./.a3s/skills'] +opts.agent_dirs = ['./.a3s/agents'] + +session = agent.session('/path/to/project', opts) + +# You can register more agent definition directories after the session exists. +session.register_agent_dir('./team/shared-agents') + +# Inspect what the session loaded. +print('Tools:', session.tool_names()) +print('Commands:', session.list_commands()) + +# The agent now has access to the project skills. +result = session.run( + 'Use the project conventions skill to scaffold a new module.', +) +print(result.text) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/project", &code.SessionOptions{ + SkillDirs: []string{"./.a3s/skills"}, + AgentDirs: []string{"./.a3s/agents"}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + if _, err := session.RegisterAgentDir(ctx, "./team/shared-agents"); err != nil { + log.Fatal(err) + } + tools, err := session.ToolNames(ctx) + if err != nil { + log.Fatal(err) + } + skills, err := session.SkillNames(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("Tools:", tools) + fmt.Println("Skills:", skills) + + result, err := session.Run( + ctx, + "Use the project conventions skill to scaffold a new module.", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +## Skill Registry Behavior + +The default effective skill registry contains no embedded skills. +`builtinSkills: true` / `builtin_skills = True` is accepted for compatibility, +but it currently adds nothing. Add skills with `skillDirs` / `skill_dirs`, +inline skills, or an explicit custom `SkillRegistry`. + +For day-to-day projects, keep durable reusable behavior in `.a3s/skills` or a +configured skill directory. + +Custom subagents loaded from `agentDirs` can be referenced by name in +[`session.parallel(...)`](/guide/examples/orchestration) and +[`session.pipeline(...)`](/guide/examples/orchestration) alongside the built-in +registry agents (`explore`, `plan`, `general`, `verification`, `review`). + +A runnable version ships at `sdk/node/examples/skills/test_custom_skills_agents.ts`. diff --git a/website/docs/v8.5.1/en/guide/examples/streaming.mdx b/website/docs/v8.5.1/en/guide/examples/streaming.mdx new file mode 100644 index 00000000..4de8655c --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/streaming.mdx @@ -0,0 +1,209 @@ +--- +title: 'Streaming' +description: 'Read incremental AgentEvent values as a turn runs' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Streaming + +`session.stream(prompt)` yields incremental events as the turn runs, so you can render text as it arrives and react to tool activity in real time. Use it when you want a live UI or a CLI that prints output token-by-token instead of waiting for the full result from `send` or `run`. + +Each event carries the stable envelope fields `version`, `type`, `payload`, and optional `metadata`. The common kinds are `agent_start`, `text_delta` / `reasoning_delta`, `tool_start` / `tool_end`, `agent_end`, and `error`. Verification data is carried by the `agent_end` payload and convenience fields. + + + + +```rust +use a3s_code_core::{Agent, AgentEvent, CodeError, PlanningMode, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_planning_mode(PlanningMode::Disabled)) + .build() + .await?; + + let (mut events, lifecycle) = session + .stream( + "Use the bash tool to run the tests, then summarize the result.", + None, + ) + .await?; + + while let Some(event) = events.recv().await { + match event { + AgentEvent::TextDelta { text } => print!("{text}"), + AgentEvent::ToolStart { name, .. } => println!("\n[tool:start] {name}"), + AgentEvent::ToolEnd { + name, exit_code, .. + } => println!("\n[tool:end] {name} exit={exit_code}"), + AgentEvent::End { + verification_summary, + .. + } => println!("\n[verification] {verification_summary:?}"), + AgentEvent::Error { message } => return Err(CodeError::Llm(message)), + _ => {} + } + } + lifecycle + .await + .map_err(|error| CodeError::Internal(error.into()))??; + println!("\n[stream] complete"); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { planningMode: 'disabled' }); + +const stream = await session.stream( + 'Use the bash tool to run the tests, then summarize the result.', +); + +while (true) { + const next = await stream.next(); + if (next.done || !next.value) break; + + const event = next.value; + if (event.type === 'text_delta' && event.text) { + process.stdout.write(event.text); + } else if (event.type === 'tool_start') { + console.log(`\n[tool:start] ${event.toolName ?? 'unknown'}`); + } else if (event.type === 'tool_end') { + console.log( + `\n[tool:end] ${event.toolName ?? 'unknown'} exit=${event.exitCode ?? 0}`, + ); + } else if (event.type === 'agent_end') { + console.log(`\n[verification] ${event.verificationSummaryText ?? ''}`); + } else if (event.type === 'error') { + throw new Error(event.error ?? 'stream error'); + } +} + +console.log('\n[stream] complete'); +session.close(); +``` + + + + +```python +import os + +from a3s_code import Agent, SessionOptions + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.planning_mode = "disabled" + session = agent.session(".", opts) + + prompt = "Use the bash tool to run the tests, then summarize the result." + + try: + for event in session.stream(prompt): + if event.type == "text_delta" and event.text: + print(event.text, end="", flush=True) + elif event.type == "tool_start": + print(f"\n[tool:start] {event.tool_name or 'unknown'}") + elif event.type == "tool_end": + print(f"\n[tool:end] {event.tool_name or 'unknown'} exit={event.exit_code or 0}") + elif event.type == "agent_end": + print(f"\n[verification] {event.verification_summary_text or ''}") + elif event.type == "error": + raise RuntimeError(event.error or "stream error") + print("\n[stream] complete") + finally: + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + PlanningMode: code.PlanningDisabled, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + stream, err := session.Stream( + ctx, + "Use the bash tool to run the tests, then summarize the result.", + nil, + ) + if err != nil { + log.Fatal(err) + } + for event := range stream.Events { + switch event.Type { + case code.EventTextDelta: + var payload struct { + Text string `json:"text"` + } + if err := event.DecodePayload(&payload); err != nil { + log.Fatal(err) + } + fmt.Print(payload.Text) + default: + // Forward event.Payload and event.Metadata to the UI as needed. + } + } + if err := <-stream.Done; err != nil { + log.Fatal(err) + } +} +``` + + + + +Notes: + +- Rust receives `AgentEvent` values from a Tokio channel, Node.js iterates with `stream.next()`, Python uses a synchronous iterator, and Go drains `stream.Events` before reading the terminal error from `stream.Done`. Canceling the Go context also requests cancellation of the active native run. +- All four SDKs expose the canonical event type. Node.js and Python add convenience projections; Go keeps `Payload` and `Metadata` as `json.RawMessage` and provides `DecodePayload`. Unknown future event types remain lossless in every SDK. +- Streamed events can also include human-in-the-loop confirmation signals (`confirmation_required`, `confirmation_received`, `confirmation_timeout`) when a confirmation policy is enabled. + +Runnable streaming examples ship under `sdk/node/examples/streaming/`. +A complete human-in-the-loop confirmation loop ships at +`sdk/node/examples/streaming/hitl_confirmation_loop.ts`, with the +matching Python version under `sdk/python/examples/` and Go stream +coverage in `sdk/go/session_test.go`. diff --git a/website/docs/v8.5.1/en/guide/examples/structured-output.mdx b/website/docs/v8.5.1/en/guide/examples/structured-output.mdx new file mode 100644 index 00000000..9cb1252d --- /dev/null +++ b/website/docs/v8.5.1/en/guide/examples/structured-output.mdx @@ -0,0 +1,599 @@ +--- +title: 'Structured Output' +description: 'Generate schema-validated JSON values with the generate_object tool.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Structured Output + +The built-in `generate_object` tool asks the configured LLM for a JSON value, +validates the response against a JSON Schema you supply, and returns the +validated value only on a zero-exit result. Use it whenever you need +machine-readable results: extraction, classification, config generation, or +feeding another program. + +You can call it directly through `session.tool('generate_object', ...)`. The +tool result carries the validated object as JSON on `result.output` — parse it +and read the `object` field. The same tool also supports agent-driven +invocation, where the model decides to call it during a `send`. + +## Direct tool call + +The simplest path: call `generate_object` directly, check the tool exit code, +and parse the validated object out of the result. + + + + +```rust +use a3s_code_core::Agent; +use serde_json::{json, Value}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["name", "age", "skills"], + "properties": { + "name": { "type": "string" }, + "age": { "type": "integer", "minimum": 0 }, + "skills": { + "type": "array", + "items": { "type": "string" }, + "minItems": 1 + } + } + }, + "prompt": "Extract: \"Alice is 28, skilled in Rust, TypeScript, and Python.\"", + "schema_name": "developer", + "mode": "tool" + }), + ) + .await?; + if result.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "generate_object".into(), + message: result.output, + }); + } + + let value: Value = serde_json::from_str(&result.output)?; + println!("{}", value["object"]); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['name', 'age', 'skills'], + properties: { + name: { type: 'string' }, + age: { type: 'integer', minimum: 0 }, + skills: { + type: 'array', + items: { type: 'string' }, + minItems: 1, + }, + }, + }, + prompt: 'Extract: "Alice is 28, skilled in Rust, TypeScript, and Python."', + schema_name: 'developer', + mode: 'tool', +}); + +if (result.exitCode !== 0) { + throw new Error(result.output); +} + +const { object } = JSON.parse(result.output); +console.log(object); +// { name: "Alice", age: 28, skills: ["Rust", "TypeScript", "Python"] } + +session.close(); +``` + + + + +```python +import json +from a3s_code import Agent + +agent = Agent.create('agent.acl') +session = agent.session('.') + +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["name", "age", "skills"], + "properties": { + "name": {"type": "string"}, + "age": {"type": "integer", "minimum": 0}, + "skills": { + "type": "array", + "items": {"type": "string"}, + "minItems": 1, + }, + }, + }, + "prompt": 'Extract: "Alice is 28, skilled in Rust, TypeScript, and Python."', + "schema_name": "developer", + "mode": "tool", +}) + +if result.exit_code != 0: + raise RuntimeError(result.output) + +obj = json.loads(result.output)["object"] +print(obj) +# {"name": "Alice", "age": 28, "skills": ["Rust", "TypeScript", "Python"]} + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"name", "age", "skills"}, + "properties": map[string]any{ + "name": map[string]any{"type": "string"}, + "age": map[string]any{"type": "integer", "minimum": 0}, + "skills": map[string]any{ + "type": "array", "items": map[string]any{"type": "string"}, + "minItems": 1, + }, + }, + }, + "prompt": `Extract: "Alice is 28, skilled in Rust, TypeScript, and Python."`, + "schema_name": "developer", + "mode": "tool", + }) + if err != nil { + log.Fatal(err) + } + if result.ExitCode != 0 { + log.Fatal(result.Output) + } + var value struct { + Object map[string]any `json:"object"` + } + if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + log.Fatal(err) + } + fmt.Println(value.Object) +} +``` + + + + +The validated value lives on the `object` key of the parsed output. When +`result.exitCode` (Node) / `result.exit_code` (Python) is zero, fields declared +in `required` have passed runtime validation. If the model cannot satisfy the +schema after repair attempts, the tool reports a non-zero exit code. + +## Enum classification + +Constrain a field to a fixed set with `enum`. This turns a free-form model +classification into a schema-gated result. + + + + +```rust +use serde_json::{json, Value}; + +let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["sentiment", "confidence"], + "properties": { + "sentiment": { + "type": "string", + "enum": ["positive", "negative", "neutral"] + }, + "confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + } + } + }, + "prompt": "Classify sentiment: \"This is the worst product I have ever used.\"", + "schema_name": "sentiment" + }), + ) + .await?; +let value: Value = serde_json::from_str(&result.output)?; +println!( + "{} {}", + value["object"]["sentiment"], + value["object"]["confidence"] +); +``` + + + + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['sentiment', 'confidence'], + properties: { + sentiment: { type: 'string', enum: ['positive', 'negative', 'neutral'] }, + confidence: { type: 'number', minimum: 0, maximum: 1 }, + }, + }, + prompt: 'Classify sentiment: "This is the worst product I have ever used."', + schema_name: 'sentiment', +}); + +const { object } = JSON.parse(result.output); +console.log(object.sentiment, object.confidence); // "negative" 0.97 +``` + + + + +```python +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["sentiment", "confidence"], + "properties": { + "sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + }, + }, + "prompt": 'Classify sentiment: "This is the worst product I have ever used."', + "schema_name": "sentiment", +}) + +obj = json.loads(result.output)["object"] +print(obj["sentiment"], obj["confidence"]) # "negative" 0.97 +``` + + + + +```go +result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"sentiment", "confidence"}, + "properties": map[string]any{ + "sentiment": map[string]any{ + "type": "string", "enum": []string{"positive", "negative", "neutral"}, + }, + "confidence": map[string]any{ + "type": "number", "minimum": 0, "maximum": 1, + }, + }, + }, + "prompt": `Classify sentiment: "This is the worst product I have ever used."`, + "schema_name": "sentiment", +}) +if err != nil { + return err +} +var value struct { + Object map[string]any `json:"object"` +} +if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + return err +} +fmt.Println(value.Object["sentiment"], value.Object["confidence"]) +``` + + + + +## Nested schemas and arrays + +Schemas can nest objects and arrays within the runtime's bounded schema-depth +limit, and the runtime validates the whole structure. This models real config +files, manifests, or API payloads in one call. + + + + +```rust +use serde_json::{json, Value}; + +let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["items"], + "properties": { + "items": { + "type": "array", + "minItems": 3, + "maxItems": 5, + "items": { + "type": "object", + "required": ["name", "category"], + "properties": { + "name": { "type": "string" }, + "category": { + "type": "string", + "enum": ["fruit", "vegetable", "grain"] + } + } + } + } + } + }, + "prompt": "List 3 food items with their categories.", + "schema_name": "food_list" + }), + ) + .await?; +let value: Value = serde_json::from_str(&result.output)?; +let items = value["object"]["items"].as_array().unwrap(); +println!("{} {:?}", items.len(), items); +``` + + + + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['items'], + properties: { + items: { + type: 'array', + minItems: 3, + maxItems: 5, + items: { + type: 'object', + required: ['name', 'category'], + properties: { + name: { type: 'string' }, + category: { type: 'string', enum: ['fruit', 'vegetable', 'grain'] }, + }, + }, + }, + }, + }, + prompt: 'List 3 food items with their categories.', + schema_name: 'food_list', +}); + +const { items } = JSON.parse(result.output).object; +console.log( + items.length, + items.map((i) => i.name), +); +``` + + + + +```python +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["items"], + "properties": { + "items": { + "type": "array", + "minItems": 3, + "maxItems": 5, + "items": { + "type": "object", + "required": ["name", "category"], + "properties": { + "name": {"type": "string"}, + "category": {"type": "string", "enum": ["fruit", "vegetable", "grain"]}, + }, + }, + }, + }, + }, + "prompt": "List 3 food items with their categories.", + "schema_name": "food_list", +}) + +items = json.loads(result.output)["object"]["items"] +print(len(items), [i["name"] for i in items]) +``` + + + + +```go +result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"items"}, + "properties": map[string]any{ + "items": map[string]any{ + "type": "array", "minItems": 3, "maxItems": 5, + "items": map[string]any{ + "type": "object", + "required": []string{"name", "category"}, + "properties": map[string]any{ + "name": map[string]any{"type": "string"}, + "category": map[string]any{ + "type": "string", + "enum": []string{"fruit", "vegetable", "grain"}, + }, + }, + }, + }, + }, + }, + "prompt": "List 3 food items with their categories.", + "schema_name": "food_list", +}) +if err != nil { + return err +} +var value struct { + Object struct { + Items []map[string]any `json:"items"` + } `json:"object"` +} +if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + return err +} +fmt.Println(len(value.Object.Items), value.Object.Items) +``` + + + + +## Agent-driven invocation + +You can also let the agent decide when to use structured output. Ask it to call `generate_object` during a `send`; it gathers context first, then emits the object. + + + + +```rust +let result = session + .send( + "Use generate_object to extract the movie title, year, and genre from: \ + The movie \"Inception\" was released in 2010 and is a sci-fi thriller.", + None, + ) + .await?; + +println!( + "tool calls: {}, tokens: {}", + result.tool_calls_count, + result.usage.total_tokens +); +``` + + + + +```ts +const result = await session.send( + 'Use the generate_object tool to extract the following into an object ' + + 'with fields "title" (string), "year" (integer), "genre" (string): ' + + 'The movie "Inception" was released in 2010 and is a sci-fi thriller.', +); + +console.log( + `tool calls: ${result.toolCallsCount}, tokens: ${result.totalTokens}`, +); +``` + + + + +```python +result = session.send( + 'Use the generate_object tool to produce a JSON object with schema ' + '{"type":"object","required":["language","paradigm"],"properties":' + '{"language":{"type":"string"},"paradigm":{"type":"string"}}} ' + 'for: "Rust is a systems programming language with a focus on safety."' +) + +print(f"tool calls: {result.tool_calls_count}, tokens: {result.total_tokens}") +``` + + + + +```go +result, err := session.Run( + ctx, + "Use generate_object to extract the movie title, year, and genre from: "+ + `The movie "Inception" was released in 2010 and is a sci-fi thriller.`, +) +if err != nil { + return err +} +fmt.Printf( + "tool calls: %d, tokens: %d\n", + result.ToolCallsCount, + result.Usage.TotalTokens, +) +``` + + + + +## Schema validation coverage + +The built-in validator supports: + +- `type` (including nullable arrays like `["string", "null"]`) +- `required`, `properties`, `additionalProperties` +- `enum`, `const` +- `$ref` with local `$defs` or `definitions` +- `allOf`, `anyOf`, `oneOf` +- `minLength`, `maxLength`, `pattern` +- `minimum`, `maximum`, `exclusiveMinimum`, `exclusiveMaximum` +- `minItems`, `maxItems`, `items` +- Nested object and array validation, including root arrays and scalar values + +## Notes + +- The validated value is on the `object` key of the parsed `result.output`. `auto` selects forced-tool mode when the provider declares that capability and otherwise uses the prompt+schema fallback. Explicit `strict` and `json` modes use provider-native response formats only when supported, then safely fall back to forced-tool or prompt mode. +- List every field you depend on in `required` — the runtime enforces it, so missing or mistyped fields fail validation instead of silently returning partial data. +- `generate_object` is a built-in tool registered independently of built-in skills. +- Direct `session.tool(...)` calls are host control-plane calls. Use `permissionPolicy` when you let the model choose tools inside `send` / `run` / `stream`; use host authorization before direct SDK calls. + +A runnable version ships at `sdk/node/examples/basic/test_generate_object.ts` and `sdk/python/examples/test_generate_object.py`. diff --git a/website/docs/v8.5.1/en/guide/filesystem-agents.mdx b/website/docs/v8.5.1/en/guide/filesystem-agents.mdx new file mode 100644 index 00000000..5bd96826 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-agents.mdx @@ -0,0 +1,75 @@ +--- +title: 'agents/ Role Directory' +description: 'Define worker agents for task and automatic delegation' +--- + +# agents/ Role Directory + +`agents/` stores worker/subagent definitions. Prefer `.a3s/agents/` for new A3S projects. Migration projects may still read `.claude/agents/`, but new docs and projects should use `.a3s/agents/`. + +```text +repo/ +└── .a3s/ + └── agents/ + ├── explorer.md + ├── security-reviewer.md + └── verification-runner.md +``` + +These files are not the main AgentDir. They are invoked by the model-visible +`task` tool, the `session.task(...)` and `session.tasks(...)` host helpers, or +`autoDelegation`. The parent session still owns final synthesis, verification, +and permission boundaries. + +## Agent File + +```md +--- +name: security-reviewer +description: Use for permission, secret, and external side-effect review +tools: Read, Search, Bash(rg *) +disallowedTools: + - Write + - Bash(git push *) +--- + +Review security risks first. Return blockers, evidence paths, and required verification. +``` + +`name` is the call name. `description` drives automatic routing. The body describes the worker role. Tool fields narrow visible capabilities; do not rely on the worker merely promising not to do risky things. + +## Manual Delegation + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + maxParallelTasks: 4, +}); + +await session.task({ + agent: 'security-reviewer', + description: 'Review release side effects', + prompt: 'Check changed auth, permission, and external API paths.', +}); +``` + +Fixed flows are better as manual delegation or programmable orchestration. Automatic delegation fits goals where the parent agent should choose specialists. + +## Automatic Delegation + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, +}); +``` + +Automatic delegation depends on descriptions and confidence scoring. Write descriptions that say when to use the agent, not just what the role is called. + +## Practices + +- Keep one role per file. +- Make descriptions useful for routing and bodies useful for execution. +- Ask workers to return evidence, risks, and next steps. +- Keep publish, delete, push, and other high-risk permissions out of default workers. +- Use `workerAgents` or `registerWorkerAgent()` for dynamic one-off workers. diff --git a/website/docs/v8.5.1/en/guide/filesystem-config.mdx b/website/docs/v8.5.1/en/guide/filesystem-config.mdx new file mode 100644 index 00000000..e2357e09 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-config.mdx @@ -0,0 +1,93 @@ +--- +title: 'agent.acl' +description: 'File-based runtime config for models, providers, queues, skill dirs, and worker agent dirs' +--- + +# agent.acl + +`agent.acl` is the runtime configuration entry point for filesystem-first agents. It turns model, provider, queue, storage, skill directory, and worker agent directory policy into versioned configuration. + +An SDK host may pass any `.acl` file explicitly with `Agent.create("agent.acl")`. +The `a3s code` TUI discovers `.a3s/config.acl` from the workspace and then +`~/.a3s/config.acl`; it does not require a root `agent.acl`. An AgentDir-local +`agent.acl` serves that durable agent. The format is the same; discovery and +scope are different. + +## Basic Config + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} +``` + +`apiKey` / `api_key` and `baseUrl` / `base_url` are accepted aliases. The +runtime does not hard-code model names; `default_model` and session-level +`model` overrides must match the `provider/model-id` values declared here. + +Inject tokens through environment variables. Do not commit credentials. Once `agent.acl` lives in the repo, it is product behavior and should be reviewed like code. + +## Directory Discovery + +```acl +skill_dirs = ["./.a3s/skills"] +agent_dirs = ["./.a3s/agents"] +project_doc_max_bytes = 32768 +project_doc_fallback_filenames = ["TEAM_GUIDE.md"] + +auto_delegation { + enabled = true + min_confidence = 0.72 + max_tasks = 4 + auto_parallel = false +} +``` + +`skill_dirs` points at reusable skills. `agent_dirs` points at worker/subagent definitions. `project_doc_max_bytes` bounds the combined root-to-workspace instruction chain; fallback filenames are checked after `AGENTS.override.md` and `AGENTS.md`. Automatic delegation decides whether the model may choose a worker; it does not remove parent-session permission policy, tool visibility, or verification requirements. + +## Session Storage + +```acl +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +`sessions_dir` is the local file-session persistence path used when the session +does not receive an explicit SDK `sessionStore`. `storage_backend = "memory"` +keeps sessions ephemeral. `storage_url` is parsed as custom storage metadata, +but it does not create a local `FileSessionStore` by itself. + +## Inside AgentDir + +AgentDir `agent.acl` is optional. When present, `AgentDir::load` parses it into `CodeConfig` and combines it with `instructions.md`, `skills/`, `tools/`, and `schedules/`. + +```text +release-agent/ +├── instructions.md +├── agent.acl +├── skills/ +├── tools/ +└── schedules/ +``` + +Good AgentDir config includes the agent's default model, providers, limits, queue policy, and private skill dirs. Keep environment differences outside the file; use env vars or host injection for development, staging, and production differences. + +## Boundaries + +- Config decides what can be connected and how the runtime starts; it does not bypass permission gates. +- Prefer workspace-relative or AgentDir-relative paths. +- Tune automatic delegation together with high-quality `agents/` descriptions. +- High-risk tools should still go through HITL or allow-lists. diff --git a/website/docs/v8.5.1/en/guide/filesystem-first.mdx b/website/docs/v8.5.1/en/guide/filesystem-first.mdx new file mode 100644 index 00000000..6e7294db --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-first.mdx @@ -0,0 +1,68 @@ +--- +title: 'Filesystem-First' +description: 'Persist roles, config, skills, tools, schedules, and team definitions as reviewable filesystem conventions' +--- + +# Filesystem-First + +A3S Code treats the filesystem as the first product interface for agents. Roles, tools, skills, schedules, and team definitions that need to last are written as files first, then loaded by convention. The goal is not fewer knobs; the goal is to make agent behavior reviewable, versioned, reusable, and portable. + +Filesystem-first has two common shapes: + +```text +repo/ +├── AGENTS.md # durable project instructions +├── agent.acl # model, provider, queue, and delegation config +└── .a3s/ + ├── agents/ # worker agents for task / autoDelegation + └── skills/ # reusable skills + +release-agent/ +├── instructions.md # role slot for one durable agent +├── agent.acl # runtime config for that agent +├── skills/ # private skills +├── tools/ # MCP / script tool specs +└── schedules/ # recurring turns +``` + +The first shape is a workspace convention for interactive development, team delegation, and project knowledge. The second shape is an AgentDir convention for long-running agents, scheduled work, and directory-scoped tools. + +## Path Map + +| Path | Purpose | Use it when | +| ------------------------- | --------------------------------------------------------- | ------------------------------------------------------------------------------------ | +| `AGENTS.md` | Stable workspace instructions | Code style, verification commands, safety boundaries, and release flow must persist. | +| `instructions.md` | Role slot for an AgentDir main agent | A single directory should load as a durable agent. | +| `agent.acl` | Model, provider, queue, skill dirs, and delegation config | Runtime policy should be versioned with the repo or agent. | +| `.a3s/agents/` | Worker/subagent definitions | A parent agent needs `task` or automatic delegation. | +| `.a3s/skills/`, `skills/` | Reusable skills | Many tasks share checklists, domain flow, or operating rules. | +| `tools/` | AgentDir MCP or script tools | A connector or constrained script belongs to scheduled agent sessions. | +| `schedules/` | Recurring turns | Daily reports, patrols, sync jobs, and regression checks should run on cron. | + +These conventions are not a new prompt system. `AGENTS.md`, `instructions.md`, and skills enter A3S Code's context composition path; tool visibility, permission gates, HITL, response contracts, and verification remain harness-controlled. + +## Loading Order + +A typical session parses `agent.acl`, binds a workspace, then loads project instructions, skills, agent definitions, direct tools, MCP connections, and runtime policy. `serve_agent_dir` loads an AgentDir by combining `instructions.md`, local `agent.acl`, `skills/`, `tools/`, and `schedules/`, then creates one independent session for every schedule. + +The closer a path is to a specific agent, the more local its meaning should be. Root `AGENTS.md` describes the whole project, `.a3s/agents/*.md` describes one worker, and AgentDir `instructions.md` describes the directory's main agent. + +## Design Rules + +- Conventions discover and assemble behavior; they do not bypass safety. +- Files should be code-reviewable. Model, provider, tool, schedule, and role changes should be visible in diffs. +- Never commit secrets. Use environment variables, host connections, or a secret manager. +- Keep one-off experiments in SDK options; commit files when behavior must be reused, audited, or migrated. +- AgentDir is the main-agent directory; `.a3s/agents/` is the worker-agent definition directory. + +## Reading Order + +1. [Convention Over Configuration](/guide/convention-over-configuration) +2. [AGENTS.md](/guide/agents-md) +3. [instructions.md](/guide/filesystem-instructions) +4. [agent.acl](/guide/filesystem-config) +5. [Agent Directory](/guide/agent-dir) +6. [agents/ Role Directory](/guide/filesystem-agents) +7. [skills/ Skill Directory](/guide/filesystem-skills) +8. [tools/ Tool Directory](/guide/filesystem-tools) +9. [schedules/ Schedule Directory](/guide/filesystem-schedules) diff --git a/website/docs/v8.5.1/en/guide/filesystem-instructions.mdx b/website/docs/v8.5.1/en/guide/filesystem-instructions.mdx new file mode 100644 index 00000000..054fd6d4 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-instructions.mdx @@ -0,0 +1,50 @@ +--- +title: 'instructions.md' +description: 'The AgentDir main-agent role slot and its boundary with AGENTS.md' +--- + +# instructions.md + +`instructions.md` is the role file for an AgentDir main agent. It is the only required file in an AgentDir, and its body is injected as `SystemPromptSlots.role` into every scheduled session. + +It has a different scope from `AGENTS.md`: + +| File | Scope | Good for | +| ----------------- | ------------------------- | --------------------------------------------------------------------------- | +| `AGENTS.md` | Workspace or subdirectory | Project rules, verification commands, code style, and safety requirements. | +| `instructions.md` | One AgentDir main agent | Role identity, work goals, output preferences, and durable task boundaries. | + +`instructions.md` is plain Markdown with no frontmatter. It does not override harness boundaries, response format, tool permissions, or verification requirements. + +## Recommended Shape + +```md +You are a release-readiness agent for this repository. + +Responsibilities: + +- Track release blockers and risky changes. +- Separate shipped changes from follow-up work. +- Never invent CI status, versions, or owners. + +Output: + +- Start with blockers. +- Include evidence paths. +- End with required verification commands. +``` + +Keep it short, stable, and reviewable. Put project commands in `AGENTS.md`, reusable process in `skills/`, and recurring prompts in `schedules/*.md`. `instructions.md` should answer: who is this durable agent, and how does it work by default? + +## Where It Is Used + +`serve_agent_dir` reads `instructions.md` at startup and applies it to every enabled schedule session. With a `SessionStore`, history comes from the store, but the current `instructions.md` is reloaded on every boot, so role edits take effect after restart. + +Interactive sessions can use SDK prompt slots for temporary roles. Commit `instructions.md` when the role should be reviewed, reused, or shared by several schedules. + +## Do Not Put + +- Secrets, tokens, private endpoints, or personal paths. +- Instructions to bypass safety or auto-approve high-risk actions. +- Large project manuals; link to docs or skills instead. +- Schedule prompts; every recurring task belongs in `schedules/*.md`. diff --git a/website/docs/v8.5.1/en/guide/filesystem-schedules.mdx b/website/docs/v8.5.1/en/guide/filesystem-schedules.mdx new file mode 100644 index 00000000..55fdcc97 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-schedules.mdx @@ -0,0 +1,74 @@ +--- +title: 'schedules/ Schedule Directory' +description: 'Declare cron turns, independent sessions, and recoverable context with Markdown schedule files' +--- + +# schedules/ Schedule Directory + +`schedules/` declares recurring turns for AgentDir. Each Markdown file is one schedule: frontmatter defines cron metadata, and the body is the prompt sent to the agent on every fire. + +```text +release-agent/ +└── schedules/ + ├── daily.md + └── weekly-risk-review.md +``` + +## Schedule File + +```md +--- +cron: '0 9 * * *' +name: daily-release-check +enabled: true +--- + +Summarize merged changes since the last run, inspect release risks, +and report only blockers plus required verification. +``` + +| Field | Required | Default | Meaning | +| --------- | -------- | --------- | ---------------------------------------------------- | +| `cron` | yes | - | 5-field or 6-field cron expression. | +| `name` | no | file stem | Schedule name and session id suffix. | +| `enabled` | no | `true` | Set `false` to keep the file but pause the schedule. | + +5-field cron is normalized to 6 fields by prefixing `0` seconds. Times are evaluated in UTC. + +## Independent Sessions + +Every schedule uses a stable session id: `schedule:`. Repeated fires of the same schedule accumulate context; different schedules stay isolated. Every session receives AgentDir `instructions.md`, `agent.acl`, `skills/`, and `tools/`. + +Daily and weekly schedules can share one AgentDir without sharing chat history. Use external storage, A3S Memory, or explicit tools when schedules need shared state. + +## Serve Daemon + +```rust +use a3s_code_core::config::AgentDir; +use a3s_code_core::serve::serve_agent_dir; +use a3s_code_core::Agent; +use tokio_util::sync::CancellationToken; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let agent_dir = AgentDir::load("./release-agent")?; + let agent = Agent::from_config(agent_dir.config.clone()).await?; + let cancel = CancellationToken::new(); + + serve_agent_dir(&agent, &agent_dir, "./workspace", None, cancel).await?; + Ok(()) +} +``` + +`serve_agent_dir` runs every enabled schedule until the cancellation token fires. Graceful cancellation lets an in-flight turn finish the current loop iteration. + +## Recovery And Limits + +With a `SessionStore`, daemon restart restores the conversation history for existing `schedule:` sessions. Current `instructions.md`, `skills/`, and `tools/` are re-applied on every boot. + +Know the limits: + +- Recovery restores history; it does not catch up missed fires during downtime. +- The store directory is a trust boundary for recovered history and workspace state. +- Invalid cron fails when the scheduler is built and names the offending schedule. +- An AgentDir with no enabled schedules returns immediately. diff --git a/website/docs/v8.5.1/en/guide/filesystem-skills.mdx b/website/docs/v8.5.1/en/guide/filesystem-skills.mdx new file mode 100644 index 00000000..df5e9633 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-skills.mdx @@ -0,0 +1,67 @@ +--- +title: 'skills/ Skill Directory' +description: 'Organize reusable skills, checklists, and domain process in workspace or AgentDir skill directories' +--- + +# skills/ Skill Directory + +`skills/` stores reusable skills. Skills express stable process, checklists, domain terms, and tool-use guidance. A skill is not a worker agent and does not start an independent session. + +A3S Code commonly uses two locations: + +```text +repo/.a3s/skills/ # workspace-level skills +release-agent/skills/ # AgentDir-private skills +``` + +Workspace skills are loaded through `skillDirs` or `agent.acl` `skill_dirs`. +Agent definitions use `agentDirs`; do not put skill directories in `agentDirs`. +AgentDir-private skills are injected when `serve_agent_dir` creates schedule +sessions. + +## Skill File + +```md +--- +name: release-readiness +description: Check whether a repository is ready to release +allowed-tools: read(*), search(*), bash(pnpm test*), bash(cargo test*) +--- + +Always inspect: + +- package or crate version changes +- migration compatibility +- release notes +- required verification commands + +Return blockers first, then risks, then follow-up work. +``` + +Frontmatter helps discovery and filtering. The body describes how to execute. +Keep `allowed-tools` minimal. When the skill is invoked through the `Skill` +tool, omitted `allowed-tools` grants no tools, so the invocation remains +fail-secure. Skills guide the model; they should not widen permissions. + +## When To Use skills/ + +Use skills for repeated review checklists, product or protocol process, release and migration workflows, recommended tool order, and shared background for several worker agents. + +Do not use skills for independent roles that should be invoked by `task`; put those in `agents/`. Do not use skills for recurring jobs; put those in `schedules/`. Do not use skills to declare external capabilities; use `tools/` or MCP. + +## Loading + +```ts +const session = agent.session('/repo', { + skillDirs: ['./.a3s/skills'], +}); +``` + +The model can use `search_skills` to find relevant skills. File skills and inline skills share discovery semantics. As directories grow, keep `name` and `description` searchable and unambiguous. + +## Maintenance + +- Keep skill files short and stable. +- Use minimal runnable examples, not long logs. +- Put agent-specific skills in that AgentDir's `skills/`. +- Put repository-wide skills in `.a3s/skills/` and document their boundary in `AGENTS.md`. diff --git a/website/docs/v8.5.1/en/guide/filesystem-tools.mdx b/website/docs/v8.5.1/en/guide/filesystem-tools.mdx new file mode 100644 index 00000000..02690b61 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/filesystem-tools.mdx @@ -0,0 +1,67 @@ +--- +title: 'tools/ Tool Directory' +description: 'Declare AgentDir MCP and script tools while preserving permissions, HITL, and allow-list boundaries' +--- + +# tools/ Tool Directory + +`tools/` declares directory-scoped tools for AgentDir. Each `tools/.md` describes one model-visible capability. A3S Code currently supports `kind: mcp` and `kind: script`. + +```text +release-agent/ +└── tools/ + ├── github.md + └── search-auth.md +``` + +Tool definitions come from the filesystem, but visibility, execution, confirmation, and audit still belong to the harness, permission policy, and AgentDir loader. A file existing is not unlimited permission. + +## MCP Tool + +```md +--- +kind: mcp +name: github +transport: stdio +command: npx +args: ['-y', '@modelcontextprotocol/server-github'] +env: + GITHUB_TOKEN: '${GITHUB_TOKEN}' +--- + +GitHub issues and pull request tools. +``` + +Every enabled schedule session connects the MCP server at startup and receives namespaced `mcp__github__*` tools. Inject secrets through environment variables, not the tool file. + +## Script Tool + +```md +--- +kind: script +name: search-auth +path: scripts/search-auth.js +allowed_tools: [grep, glob, read] +limits: + timeoutMs: 30000 + maxToolCalls: 30 + maxOutputBytes: 65536 +--- + +Find authentication-related files and return an evidence list. +``` + +`kind: script` exposes a pre-parameterized QuickJS `program` call as a model-visible tool. The source must define `async function run(ctx, inputs)`. It has no filesystem, network, process, or environment access; it can only call allow-listed tools through `ctx.tool(...)`. + +## Safety Boundary + +- `allowed_tools` is the script's internal capability boundary; keep it minimal. +- Unknown `kind`, workspace-escaping paths, duplicate tool names, and illegal limits should fail at load time. +- Do not let untrusted directories declare high-privilege MCP servers or scripts. +- High-risk tools should still use HITL, allow-lists, and audit. + +## Current Scope + +`tools/` is installed by `serve_agent_dir` per schedule session. It serves durable agents and recurring work. Normal interactive sessions should use host direct tools, MCP connections, or SDK `session.tool(...)` registration. + +If a capability is a project-wide connector, prefer host config or MCP. If it belongs only to one durable scheduled agent, put it in that AgentDir's `tools/`. diff --git a/website/docs/v8.5.1/en/guide/hooks.mdx b/website/docs/v8.5.1/en/guide/hooks.mdx new file mode 100644 index 00000000..f3486627 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/hooks.mdx @@ -0,0 +1,144 @@ +--- +title: 'Hooks' +description: 'Lifecycle interception and policy callbacks' +--- + +# Hooks + +Hooks register lifecycle callbacks inside a session. The registration lifecycle +is `registerHook()`, `hookCount()`, and `unregisterHook()`. + +## Events + +The Node.js and Python registration helpers accept these stable names: + +```text +pre_tool_use +post_tool_use +generate_start +generate_end +session_start +session_end +skill_load +skill_unload +pre_prompt +post_response +on_error +pre_run_control +post_run_control +``` + +Rust Core also exposes `permission_request`, `pre_compact`, and `post_compact` +as first-class lifecycle points. The Go bridge can register the same serialized +enum names. Those three names are not yet accepted by the Node.js or Python +string parsers, so use a Rust or Go host when policy must intercept permission +or compaction directly. + +Core also defines perception, memory, planning, reasoning, rate-limit, +confirmation, success, and intent events for specialized harnesses. Treat an +event as a policy boundary only when the protected runtime path consumes its +decision. + +## Registration Example + +```ts +session.registerHook( + 'release-publish-observer', + 'pre_tool_use', + { tool: 'bash', commandPattern: 'npm publish|twine upload|cargo publish' }, + { priority: 50, timeoutMs: 1000 }, + () => ({ action: 'continue' }), +); +``` + +A handler may return `{ action: 'continue' }`, `{ action: 'skip' }`, +`{ action: 'block', reason }`, `{ action: 'retry', reason, delayMs }`, or +`null`/`undefined` to continue. Validate the specific event path you depend on +before using a hook as a production gate. + +## Lifecycle Governance + +The gating events are `pre_tool_use`, `permission_request`, `pre_compact`, +`pre_prompt`, `pre_planning`, and `pre_run_control`. A handler failure or timeout at one of these +points fails closed. Other events are observational or advisory and continue +when their handler infrastructure fails. + +`pre_tool_use` can replace arguments before the tool reaches confirmation or +side effects: + +```ts +() => ({ + action: 'continue', + modified: { + updatedInput: { file_path: 'approved/release.txt', content: 'ready\n' }, + }, +}); +``` + +Core accepts `updatedInput`, `updated_input`, `args`, or a direct argument +object inside `modified` (and also accepts the Codex-style +`hookSpecificOutput` wrapper). The rewritten object is validated against the +tool's JSON Schema again. Invalid rewritten arguments are rejected before any +tool side effect. + +`pre_prompt` can return `modified.prompt` and optional +`modified.additionalContext` (or `additional_context`). The resulting prompt, +including the bounded hook-context block, replaces the actual user message +sent to the model. `permission_request` can return `decision: 'allow'` or +`decision: 'deny'`; `pre_compact` can block compaction. `session_start` and +`session_end` provide matching lifecycle observations for host cleanup and +audit. + +`pre_run_control` gates a typed `steer` or `interrupt` request before it enters +the active Run inbox. `post_run_control` observes its durable accepted, +applied, settled, or rejected receipt. A retry with the same request ID and +payload reuses the recorded admission result instead of firing a second gating +decision; a conflicting payload with that ID is rejected. Post-control Hook +results are observational and cannot rewrite an already issued receipt. + +## Denial Feedback + +Use `block` when retrying the same invocation cannot succeed without changing +the request, arguments, or policy context. Use `retry` for a temporary +condition and include both a reason and suggested delay: + +```ts +() => ({ + action: 'retry', + reason: 'The policy backend is temporarily unavailable.', + delayMs: 1000, +}); +``` + +Python callbacks use `delay_ms`; Go callbacks return +`&code.HookResponse{Action: "retry", Reason: "...", DelayMS: 1000}`. +The current invocation is denied rather than automatically scheduled. The +model receives the explanation and explicit retry guidance, while direct SDK +callers receive a structured tool error: + +```json +{ + "type": "hook_denied", + "reason": "The policy backend is temporarily unavailable.", + "retryable": true, + "retry_after_ms": 1000 +} +``` + +A `block` response uses the same error type with `retryable: false` and +`retry_after_ms: null`. Rust callers that need the retry explanation can use +`HookEngine::fire_outcome()`; the existing `fire()` API retains its legacy +`HookResult` projection. + +## Propagation + +Delegation and automatic subagent fan-out use the `task` tool. When a product +depends on hook behavior across delegated runs, cover that product path with an +integration test. + +## Management + +```ts +console.log(session.hookCount()); +session.unregisterHook('release-publish-observer'); +``` diff --git a/website/docs/v8.5.1/en/guide/index.mdx b/website/docs/v8.5.1/en/guide/index.mdx new file mode 100644 index 00000000..fb05f418 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/index.mdx @@ -0,0 +1,391 @@ +--- +title: 'Overview' +description: 'Install A3S Code, understand its main features, and choose an SDK' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# A3S Code + +A3S Code is the Rust runtime behind the `a3s code` terminal application. You can +also embed it in an IDE, runner, service, or desktop application. It handles the +agent loop, context, tool calls, permission checks, child tasks, asynchronous +Workspace retrieval, durable evidence, and session recovery. + +Version 8.5.1 keeps the thin `local-code` harness and separates workspace search +planes: exact `grep` (with optional in-tree trigram candidate pruning) never +opens durable zvec FTS, while `bm25` remains the ranked lexical path. Session +stores recover corrupt WALs under a cross-process flock so concurrent writers +cannot mint colliding sequences. + +## Design rules + +| Rule | Meaning | +| --------------------------- | ------------------------------------------------------------------------------------------------------ | +| Thin default | Core `default` = `local-code`. Advanced evaluation, server, and headless search are explicit features. | +| Grep ≠ zvec | Exact `grep` is match authority; trigram pruning is fail-open only. Ranked retrieval uses `bm25`. | +| One delegation path | Multi-item fan-out uses `task` / `session.tasks`. Do not restore `parallel_task`. | +| Active-only memory | Durable serving is `active_recall`. Candidate shadow mode is refused. | +| Evidence before Gate claims | Incomplete or retention-gapped evidence cannot satisfy Gate evaluation. | +| Host owns product policy | Reviewer rubrics, Cloud audit, and UI approvals stay outside Core. | + +Install the [`a3s` CLI](https://github.com/A3S-Lab/a3s) when you want to work in +a terminal. When building a product, use the Rust crate, Node.js package, +Python package, or Go module with its matching native bridge. They emit the +same events, so every UI does not need its own agent loop. + +## Choose an entry point + +| Entry | Use it when | Repository | +| --------------------------------- | -------------------------------------------------------------- | ----------------------------------------------- | +| Rust / Node.js / Python / Go SDKs | Adding a coding agent to an IDE, runner, service, or custom UI | [A3S-Lab/Code](https://github.com/A3S-Lab/Code) | +| `a3s code` | Running a coding agent directly in your terminal | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | +| `a3s-tui` | Building a terminal UI; it does not include the agent runtime | [A3S-Lab/TUI](https://github.com/A3S-Lab/TUI) | +| A3S Flow | Saving and resuming flows used by `DynamicWorkflowRuntime` | [A3S-Lab/Flow](https://github.com/A3S-Lab/Flow) | + +## What it includes + +| Area | What A3S Code provides | +| --------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Agent sessions | Create a Workspace-bound `AgentSession` with `SessionBuilder`; send, run, stream, steer, interrupt, cancel, save, resume, and close it. Concurrent transcript operations fail immediately instead of racing. | +| User interfaces | [`a3s code`](/guide/tui) renders the event stream in a terminal. Applications can render the same `AgentEvent` stream in their own interface. | +| Project files | [Filesystem-first](/guide/filesystem-first) explains `AGENTS.md`, ACL config, `.a3s/agents/`, `skills/`, `tools/`, and `schedules/`. | +| Tools | Built-in files, binary-safe local downloads, search, shell, Git, web, batch, structured output, QuickJS, Skills, MCP, and child-task tools. Model calls pass through argument, permission, confirmation, cancellation, deterministic result projection, and evidence. | +| Commands | [Commands](/guide/commands) covers TUI slash commands and custom `/command` handlers registered by an application. | +| Child tasks | Use the model-visible `task` tool or the host-side `session.task(...)` and `session.tasks(...)` helpers with built-in and custom agents. One item is focused; multiple independent items fan out concurrently. | +| Scheduling | One [Agent-wide priority scheduler](/guide/tasks#agent-wide-priority-scheduler) shares local execution capacity across sessions, direct tools, detached children, and host workflows, with FIFO ordering, aging, cancellation, and occupancy snapshots. | +| Workflows | Use [`session.parallel`](/guide/orchestration), pipelines, phases, checkpoints, loop limits, and budget records for fixed, recoverable workflows. | +| Scheduled work | Run scheduled AgentDir turns through `serveAgentDir` / `serve_agent_dir`; each schedule keeps a stable `schedule:` session. | +| Access control | Apply permission rules, user confirmation, budgets, Workspace path checks, tool timeouts, lifecycle hooks, and output cleanup during execution. | +| Workspaces | [Workspace backends](/guide/workspace-backends) support local files, application-provided workspaces, optional S3-compatible storage, and remote Git services. The native Harness can isolate conversations in detached Git worktrees. | +| Retrieval | [Workspace retrieval](/guide/context#workspace-retrieval) adds an asynchronous session-owned text catalog, zvec-rust FTS/BM25, optional host embeddings, A3S Memory exact vectors, hybrid RRF, and optional deterministic CPU reranking without a vector database service. | +| Events | `EventEnvelopeV1` is shared by Rust, Node.js, Python, and Go. Unknown event payloads and metadata are preserved. | +| Save and resume | Session snapshots, run events, traces, artifacts, loop/workflow checkpoints, and memory stores make sessions recoverable. | +| Verification | [Verification](/guide/verification) runs named checks and returns reports, summaries, artifacts, traces, and replay data. | + +## What is new in v8.5.1 + +- Default `local-code` **trigram pruning for `grep`** (`CODE-G1`): literal + needles build a fail-open candidate cache under `.a3s-code/grep-trigram` + without opening durable zvec FTS. Exact regex matching stays in Code. +- **Session-store WAL flock recovery**: concurrent writers re-read the durable + max sequence under a cross-process flock; hosts can quarantine a corrupt WAL + and continue from durable snapshots. +- Carries forward the 8.4 thin harness (Active-only memory, unified `task`, + `update_plan`, SDK capabilities v2) and the 8.3 durability/trust kernel. + +## Earlier v8.4.0 additions + +- Library and SDK **defaults are thin**: `a3s-code-core` defaults to + `local-code` (bundled zvec FTS); Node/Python/Go SDK crates default to + `zvec-rust-fts-bundled`. Enable `advanced-harness`, `server`, and/or + `headless-search` explicitly when product embeds need them. +- Model-visible **`parallel_task` is removed**; multi-item fan-out uses the + `task` tool. Matching Node/Python/Go helpers are gone. +- Durable memory serving is **Active-only** (`active_recall`). Candidate + shadow mode is removed; extraction may still write Candidates until the host + activates them. +- Built-in **`update_plan`** checklist tool plus host + `set_output_language` / `outputLanguage` across Rust and SDKs. +- SDK capabilities ship as **`a3s-code/sdk-capabilities/v2`** with + `tier: baseline | advanced`. +- Gate-mode evaluation fail-closes on incomplete evidence; first-principles E2E + and harness wrap-up runbooks live under `manual/FIRST_PRINCIPLES_E2E.md` and + `manual/HARNESS_CONVERGENCE.md`. +- Carries forward the 8.3 durability/trust kernel (negotiable session stores, + typed tool-result trust, workspace source snapshots, fallible FFI init, and + host-owned immutable-content / checkpoint hooks). + +## Earlier v8.3.0 additions + +- Session-store durability is negotiable (KRN-6). Built-in memory and file + adapters advertise exact guarantees such as aggregate CAS, append-only WAL, + writer lease fencing, optional AES-256-GCM encryption at rest, commit watch + notifications, and reference-aware artifact GC. +- Every tool result carries a typed trust label (KRN-5): trusted, workspace + data, or external. Sessions expose secret-free `model_middleware_health` + counters on Rust and all four SDKs. +- Workspace retrieval binds results to a tamper-evident source snapshot + (KRN-4). Persistent BM25/zvec indexing is release-qualified on Windows and + Linux, including stripping Windows `\\?\` verbatim paths before native opens. +- Node.js and Python FFI runtime initialization is fallible (KRN-9). + `TASK_ADMISSION_AT_CAPACITY` maps consistently across SDKs. +- Linux arm64 Python wheels ship as `manylinux_2_39_aarch64` (glibc 2.39+); + x86_64 Linux remains `manylinux_2_28`. + +## Earlier v8.2.0 additions + +- `steer` adds a newer instruction to the active Run at its next safe point; + `interrupt` cooperatively stops provider, Tool, workflow, and delegated work. + Idempotent receipts and optional expected-turn fields prevent duplicate or + stale UI actions. +- `pre_run_control` can gate a control request and `post_run_control` observes + every durable receipt transition. `run_control_applied` is available through + the shared event protocol and persisted Run history. +- The default prompt now composes a compact operating loop, a runtime authority + contract, repository Tool schemas, and safety boundaries. It treats files, + Tool output, and web content as untrusted data and requires evidence before a + completion claim without replacing host permissions or approvals. +- Real-provider release tests exercise both configured DeepSeek models across + tools and Hook rewrites, long-horizon coding, SubAgents, Skills, PTC, + replayable dynamic workflows, and live steer/interrupt behavior. +- Dynamic workflows no longer need a duplicate permission grant for their + private `program` implementation step, while inner Tool calls remain fully + governed. QuickJS `ctx.readFile()` now returns file text; `ctx.read()` keeps + the line-numbered Tool result for audit-oriented code. + +## Earlier v8.1 additions + +- `web_search` is powered by `a3s-search` v3.1.0. Google, Baidu, Bing, and + Brave browser engines use Moli by default; Chrome/Chromium and Lightpanda + remain explicit compatibility backends. The first-use runtime is selected in + this order: package sidecar, verified per-user cache, system executable, and + finally a digest-pinned HTTPS download. +- The Moli installer uses an atomic staging/receipt protocol and a + cross-process lock at `~/.cache/a3s-code/moli` (or `A3S_CODE_MOLI_CACHE_DIR`). + A second A3S Code process waits for the first install and reuses the same + executable. Set `autoDownloadMoli: false` when a host must fail closed rather + than access the network. +- Rust, Node.js, Python, and Go expose the same `sdk_capabilities` inventory, + state-graph operations, Moli diagnostics, and typed search configuration. + Use the inventory to feature-detect a build instead of parsing package files. +- Native Node packages and Python wheels include their target Moli sidecar and + provenance metadata. Linux musl packages explicitly carry an + `MOLI_UNAVAILABLE` marker because upstream Moli does not publish a musl + binary; those hosts must provide a system Moli executable or choose another + backend. + +## Earlier v7.0 additions + +- Session-owned Workspace Retrieval builds one bounded text catalog + asynchronously, reuses incremental BM25 postings, and can publish exact + in-memory vector partitions without delaying session construction or adding + a vector database. +- Dense semantics are explicitly host-enabled. Exact search, glob, BM25, Code + Intelligence, RRF, and the optional deterministic reranker run locally on + CPU; a host can inject an in-process CPU embedding callback when semantic + search is useful. +- Rust, Node.js, Python, and Go expose typed line, fixed-window, and recursive + chunking, readiness and batching metrics, optional deterministic reranking, + cancellation, current-source digest verification, and bounded cleanup. + Non-text assets are rejected before chunking or embedding. +- Product builds use zvec-rust for lexical FTS/BM25 and A3S Memory for exact + semantic vectors. Native lexical handles are bounded and temporary; semantic + vectors are released with the owning session. +- Model-bound run evidence records the effective capability and policy + identity, retrieval generation, input shape, repeated Tool-result context, + and normalized usage without retaining new prompt, source, vector, + credential, or endpoint plaintext. + +Go consumers must use the v7 module path: +`github.com/A3S-Lab/Code/sdk/go/v8`. + +v6.9 introduced the shared priority/FIFO scheduler, bounded personal and +project instruction chains, governed lifecycle hooks, isolated Harness Git +worktrees, deterministic Tool-result evidence, and exact cognitive-package +generation bindings. + +v6.8 presents one compact `search` schema for grep, glob, and dependency-free +BM25 ranking, plus one model-visible `task` schema for focused or bounded +fan-out delegation. Legacy direct host aliases remain readable without +consuming model schema tokens. The same release adds exact, replay-safe run +admission for headless hosts and exposes its snapshot/replay result through the +Node.js, Python, and Go SDKs. + +The v6.7 local `download` tool streams HTTP(S) resources through SSRF-safe +redirect validation into adjacent temporary files, supports adaptive 1–4 +connection Range transfers, and can verify a 64-character +`expected_sha256` before atomic promotion. It is a permission- and HITL-governed +workspace mutation and is not registered for remote workspace backends. See +[Tools](/guide/tools#binary-safe-local-downloads). + +Search uses a structurally gated cascade: a Moli tier included in the default +Core feature set, HTTP/RSS engines, then native APIs. A completed +cascade that remains below the structural retrieval requirements fails closed +instead of presenting weak candidates as successful evidence. No external +semantic verifier or reranker is required. Delegated contexts share bulkheads, +retry budgets, and identical-request coalescing; see +[Tools](/guide/tools#structurally-gated-web-search). + +A run roughly follows this path: + +```text +Agent / AgentSession + -> collect project context + -> optional plan + -> select tools or child tasks + -> check permission and ask the user when needed + -> execution + -> publish events and verification results + -> save the session +``` + +## Install + +For the interactive terminal workspace, run the installer for your platform: + + + + +```bash +curl --proto '=https' --tlsv1.2 -LsSf \ + https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.sh | sh +``` + + + + +```powershell +[Net.ServicePointManager]::SecurityProtocol = [Net.ServicePointManager]::SecurityProtocol -bor [Net.SecurityProtocolType]::Tls12 +irm https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.ps1 | iex +``` + + + + +The scripts select the release archive for the current system and architecture +and verify its SHA-256 digest. You can also use +`brew install a3s-lab/tap/a3s` or `cargo install a3s`. + +Install an SDK when embedding A3S Code: + +```bash +npm install @a3s-lab/code +pip install a3s-code +cargo add a3s-code-core +go get github.com/A3S-Lab/Code/sdk/go/v8 +``` + +The v8.5.1 Python package supports CPython 3.10–3.14 through one +`cp310-abi3` wheel per platform. Intel Macs use the `macosx_12_0_x86_64` wheel +and require macOS 12 or newer. See [Python wheel platforms](/api/#python-wheel-platforms-v820) +for the bootstrap flow and the `No module named pip` repair command. + +## Configure + +A3S Code uses ACL. Keep real API keys, private model endpoints, local config +paths, and tenant/user identifiers out of commits. Commit templates that resolve +credentials from the environment. + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +storage_backend = "file" +sessions_dir = ".a3s/sessions" + +search { + headless { + backend = "moli" + auto_download_moli = true + max_tabs = 4 + } +} +``` + +`auto_parallel = false` disables automatic parallel child-agent fan-out only. +Manual `task` calls and SDK `session.tasks(...)` fan-out remain available unless +manual delegation is disabled separately. + +The Moli sidecar is selected automatically by the SDK package. For a source +checkout or a minimal Core build, set `A3S_CODE_MOLI_EXECUTABLE` to a verified +executable, or let the first search call download the pinned release into the +shared cache. + +## Use the TUI + +Run `a3s code` from the workspace you want the agent to inspect: + +```bash +a3s code +a3s code resume +a3s code update +``` + +The TUI discovers config from `A3S_CONFIG_FILE`, then `.a3s/config.acl` walking +upward from the current directory, then `~/.a3s/config.acl`. + +Common first-run flow: + +```text +/init # inspect the repository and create or update AGENTS.md +/model # pick a configured provider or account-backed model +/effort # choose low, medium, high, xhigh, max, or ultracode +/ide # open the workspace tree and terminal editor +/help # open the full command and shortcut guide +``` + +## Rust Runtime Quick Start + +```rust +use a3s_code_core::{Agent, AgentEvent, SessionOptions}; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/path/to/workspace") + .options( + SessionOptions::new() + .with_planning(true) + .with_max_parallel_tasks(4) + .with_tool_timeout(120_000), + ) + .build() + .await?; + + let result = session + .send("Find the authentication entry points.", None) + .await?; + println!("{}", result.text); + + let (mut rx, _handle) = session + .stream("Summarize the test strategy.", None) + .await?; + while let Some(event) = rx.recv().await { + match event { + AgentEvent::TextDelta { text } => print!("{text}"), + AgentEvent::End { .. } => break, + _ => {} + } + } + + Ok(()) +} +``` + +Rust construction is async-first. The synchronous `Agent::session` method only +works when memory and the other required resources are already initialized; +options that require async setup return +`CodeError::AsyncSessionBuildRequired`. A session-option MCP manager always uses +the async path while discovering tools. + +Calls such as `session.tool(...)` come from your application, not the model. +Check access in your application before exposing them to users. + +## Continue reading + +- [A3S Code TUI](/guide/tui) explains installation, config discovery, slash commands, and effort profiles. +- [SDKs and APIs](/api/#go-module-and-bridge) explains Go module and bridge installation; the Go examples then stay beside Node.js and Python throughout the shared guides. +- [Filesystem-first](/guide/filesystem-first) covers `AGENTS.md`, ACL, AgentDir, Skills, tools, and schedules. +- [API contract](/guide/api-contract) lists the Node.js API covered by integration tests. +- [Sessions](/guide/sessions) covers creation, streaming, run state, saving, resuming, and cancellation. +- [Tools](/guide/tools) covers direct calls, typed errors, structured output, and QuickJS programs. +- [Workspace retrieval](/guide/context#workspace-retrieval) covers opt-in semantics, chunking, lifecycle, quality metrics, and safe CPU-only defaults. +- [Tasks](/guide/tasks) and [orchestration](/guide/orchestration) cover child agents and fixed workflows. +- [Security](/guide/security), [hooks](/guide/hooks), and [verification](/guide/verification) cover checks before, during, and after execution. +- [Memory](/guide/memory) and [persistence](/guide/persistence) cover reusable facts and session recovery. diff --git a/website/docs/v8.5.1/en/guide/isolation.mdx b/website/docs/v8.5.1/en/guide/isolation.mdx new file mode 100644 index 00000000..b9e090f6 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/isolation.mdx @@ -0,0 +1,100 @@ +--- +title: 'Isolation' +description: 'Workspace, context, and delegated-task isolation' +--- + +# Isolation + +Isolation starts with the session boundary: each session binds to one +workspace, and each delegated child run receives bounded context. Direct host +tool calls are privileged host operations; gate them in the host application +before exposing them to users. + +## Workspace Boundary + +```ts +const session = agent.session('/repo'); +``` + +Relative file, search, shell, and git operations are evaluated from the session +workspace. Security providers and hooks are session options; validate the exact +policy path you depend on before using them as a production boundary. + +## Delegated Context + +`task` and automatic subagent delegation isolate child reasoning. The parent +receives compact results instead of full transcripts. This avoids prompt +pollution and makes evidence review easier. + +## Storage Isolation + +Use separate memory and session store directories for different products, tenants, or test suites: + +```ts +import { FileMemoryStore, FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + +## Native Harness Worktrees + +When `a3s code harness` points at a Git repository, every admitted protocol +conversation receives its own temporary detached worktree at the source +repository's `HEAD`. The source worktree must be clean at admission, so local +host changes are never silently omitted. Different conversation sessions do +not share mutable files, and removing a Harness session removes its temporary +worktree without applying anything to the source checkout. + +Tracked and untracked non-ignored content is snapshotted through an isolated +temporary Git index before and after each run. After the run becomes terminal, +Core captures one binary-capable, full-index unified diff and pins the result +tree under a private Git ref. The immutable evidence belongs to that exact run; +a conflicting second capture is rejected. A resumed persisted conversation +restores its latest captured result tree into the new detached worktree. + +Non-Git workspaces continue to use the configured shared path and cannot +produce this Git change-set evidence. A dirty Git source fails isolation +admission instead of falling back to shared writes. + +## Bounded Change-set Protocol + +Post an exact run identity to `/v1/agent/changes` after its event stream reaches +`completed`, `failed`, or `cancelled`: + +```json +{ + "schema": "a3s.code.agent-change-set-request.v1", + "identity": { + "schema": "a3s.code.agent-run-identity.v1", + "protocol": "a3s.code.agent.v1", + "agent_release_identity": "sha256:", + "session_id": "conversation-018f4f86", + "run_id": "run-018f4f86-attempt-1" + } +} +``` + +The response uses `a3s.code.agent-change-set.v1` and includes the same identity, +terminal state, `base_tree`, `result_tree`, `patch_digest`, `patch_bytes`, +`observed_at_ms`, and a `patch_base64` payload with format +`git_unified_diff_v1` and encoding `base64`. The raw patch is limited to 4 MiB. + +Change capture settles just after the worker, so a terminal run can briefly +return `a3s.code.agent_protocol.change_set_pending`; retry the same query. +`a3s.code.agent_protocol.change_set_unavailable` means the workspace was not +Git-compatible or capture could not satisfy the protocol bound. + +The caller, not the endpoint, decides whether and where to apply the patch. +Before decoding or applying it, verify the declared byte count and +`sha256:` digest and retain both Git tree identities. This keeps concurrent +conversation writes isolated while still giving the host a deterministic +merge or review artifact. + +## External Harness + +Hooks and permission policy are exposed as integration points. Test live +organization policy with your own harness before treating it as a production +boundary. diff --git a/website/docs/v8.5.1/en/guide/lane-queue.mdx b/website/docs/v8.5.1/en/guide/lane-queue.mdx new file mode 100644 index 00000000..8c1b6d51 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/lane-queue.mdx @@ -0,0 +1,58 @@ +--- +title: 'Lane Queue' +description: 'Optional external and hybrid dispatch infrastructure' +--- + +# Lane Queue + +Ordinary sessions are queue-free. Lane queues are optional advanced +infrastructure for applications that need external or hybrid dispatch. They are +separate from `task` and automatic subagent delegation. + +Do not confuse this optional dispatch layer with the always-on +[Agent-wide task scheduler](/guide/tasks#agent-wide-priority-scheduler). The task scheduler uses an +`a3s-lane` priority queue internally to share local execution capacity across +sessions; a Lane queue hands selected work to local, external, or hybrid +handlers. + +## When To Use It + +Use a queue when you need: + +- external workers to complete selected tasks +- hybrid routing between local and remote executors +- queue metrics, dead letters, or dispatch observability +- multi-machine work distribution + +## Configure A Session + +```ts +const session = agent.session('/repo', { + queueConfig: { + enableDlq: true, + enableMetrics: true, + }, +}); + +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); +``` + +Lane names are `control`, `query`, `execute`, and `generate`. + +## External Completion + +```ts +const pending = await session.pendingExternalTasks(); + +if (pending.length > 0) { + await session.completeExternalTask(pending[0].id, { + success: true, + result: { summary: 'completed by worker' }, + }); +} +``` + +Keep queue use explicit. Do not introduce it for normal local agent sessions. diff --git a/website/docs/v8.5.1/en/guide/limits.mdx b/website/docs/v8.5.1/en/guide/limits.mdx new file mode 100644 index 00000000..6e1ffe9e --- /dev/null +++ b/website/docs/v8.5.1/en/guide/limits.mdx @@ -0,0 +1,189 @@ +--- +title: 'Limits' +description: 'Runtime limits, compaction, timeouts, and circuit breakers' +--- + +# Limits + +Limit options are session-level controls for long-running work, noisy tools, and +provider failures. + +## Session Options + +```ts +const session = agent.session('/repo', { + maxToolRounds: 24, + maxParseRetries: 3, + toolTimeoutMs: 120000, + circuitBreakerThreshold: 4, + autoCompact: true, + autoCompactThreshold: 0.75, + continuationEnabled: true, + maxContinuationTurns: 3, +}); +``` + +## Option Intent + +- `maxToolRounds` is the tool-iteration budget for a turn. +- `maxParseRetries` is the malformed tool-call recovery budget. +- `toolTimeoutMs` is the per-tool timeout in milliseconds. +- `circuitBreakerThreshold` is the repeated provider failure threshold. +- `autoCompact` and `autoCompactThreshold` enable context compaction behavior. +- `continuationEnabled` and `maxContinuationTurns` control continuation injection. + +## Practical Defaults + +Use strict limits for CI, release, and user-facing automation. Use larger budgets for exploratory local coding sessions, but keep verification commands explicit and required when the task has side effects. + +## Retention Limits + +A session keeps its run history, trace events, and subagent task snapshots in +memory. `SessionRetentionLimits` applies conservative finite defaults so a +long-lived session cannot grow those stores without bound. Override any +individual FIFO cap, or explicitly set `unbounded: true` to opt into unlimited +retention. + +Five independent caps: + +| Field | Effect when capped | +| ----------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | +| `max_runs_retained` | When a new run pushes past the cap, the **oldest** run and all of its events are dropped. | +| `max_events_per_run` | The oldest events in a run are FIFO-dropped. The run snapshot's `event_count` is **not** decremented — it stays the cumulative total ever recorded. | +| `max_event_bytes_per_run` | Oldest run events are dropped until the serialized-byte cap is met; an oversized single event is not retained. | +| `max_trace_events` | The oldest event in the trace sink is dropped on each new write past the cap. | +| `max_terminal_subagent_tasks` | The oldest **terminal** (completed / failed / cancelled) subagent task snapshot is dropped past the cap. **Running tasks are never dropped.** | + +All caps are soft: enforcement drops the oldest entry on insert and never +returns an error. + +```ts +const session = agent.session('/repo', { + retentionLimits: { + maxRunsRetained: 100, + maxEventsPerRun: 5000, + maxEventBytesPerRun: 16 * 1024 * 1024, + maxTraceEvents: 20000, + maxTerminalSubagentTasks: 500, + }, +}); +``` + +```python +opts = SessionOptions() +opts.retention_limits = { + 'max_runs_retained': 100, + 'max_events_per_run': 5000, + 'max_event_bytes_per_run': 16 * 1024 * 1024, + 'max_trace_events': 20000, + 'max_terminal_subagent_tasks': 500, +} +session = agent.session('/repo', opts) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + RetentionLimits: &code.RetentionLimits{ + MaxRunsRetained: code.Ptr(uint(100)), + MaxEventsPerRun: code.Ptr(uint(5_000)), + MaxEventBytesPerRun: code.Ptr(uint(16 * 1024 * 1024)), + MaxTraceEvents: code.Ptr(uint(20_000)), + MaxTerminalSubagentTasks: code.Ptr(uint(500)), + }, +}) +``` + +Rust uses the corresponding `SessionRetentionLimits` builder methods. + +## Budget Guard + +`BudgetGuard` (CHANGELOG [3.3.0] "BudgetGuard") is a host-supplied cost / quota +contract. The framework does not enforce budgets itself — it defines the +decision points and consults a guard the host plugs in. Three hooks are wired at +the LLM / tool call site: + +- `check_before_llm` — before each LLM call. +- `record_after_llm` — after each successful LLM call, with the actual provider + usage, so the host keeps its running spend total accurate. +- `check_before_tool` — before each tool call. + +Each `check_*` returns one of three decisions: + +- `Allow` — proceed normally, no event. +- `SoftLimit { resource, consumed, limit, message }` — emits an + `AgentEvent::BudgetThresholdHit { kind: "soft" }` and **proceeds**. In-session + hooks can react (auto-compact, swap to a cheaper model next turn). +- `Deny { resource, reason }` — aborts the call with + `CodeError::BudgetExhausted`. **The session stays open** — the caller can retry + later or after the host re-allocates budget. + +### Node — `session.setBudgetGuard({...})` + +Each callback takes a **single `ctx` object** (not positional arguments) and +returns a decision dict (or `null` / `{ decision: 'allow' }` to allow): + +```ts +session.setBudgetGuard({ + checkBeforeLlm: (ctx) => { + // ctx.sessionId, ctx.estimatedTokens + if (overMonthlyCap(ctx.sessionId)) { + return { + decision: 'deny', + resource: 'llm_tokens', + reason: 'monthly cap', + }; + } + return { decision: 'allow' }; + }, + recordAfterLlm: (ctx) => { + // ctx.sessionId, ctx.usage — usage keys are camelCase: + // promptTokens, completionTokens, totalTokens, cacheReadTokens, cacheWriteTokens + addSpend(ctx.sessionId, ctx.usage.totalTokens); + }, + checkBeforeTool: (ctx) => { + // ctx.sessionId, ctx.toolName + return { decision: 'allow' }; + }, + timeoutMs: 5000, // optional, default 5000 +}); +``` + +The Node bridge **fails closed**: a `check_*` callback that does not return +within `timeoutMs`, or returns something unreadable, is treated as a **deny** — +a budget control must never silently disable itself when the guard stalls +(CHANGELOG [3.3.0] Fixed "Node BudgetGuard fail-open"). + +The callback **MUST NOT throw.** Due to a napi-rs constraint a thrown exception +aborts the host process at return-value conversion. Wrap your logic in +try/catch and return a decision (e.g. a deny) instead of throwing. Hangs are +handled safely by the fail-closed timeout (CHANGELOG [3.3.0] Known limitations). + +### Python — `budget_guard` session option + +Python supplies a `BudgetGuard`-shaped object on the `budget_guard` +`SessionOptions` field. Methods that aren't defined behave as Allow / no-op. +Python callbacks use **positional arguments** and the framework catches any +exception they raise (a `check_*` that raises defaults to Allow): + +```python +class MyBudgetGuard: + def check_before_llm(self, session_id, est_tokens): + if over_monthly_cap(session_id): + return {'decision': 'deny', 'resource': 'llm_tokens', 'reason': 'monthly cap'} + return {'decision': 'allow'} + + def record_after_llm(self, session_id, usage): + # usage is a dict with snake_case keys: + # total_tokens, cache_read_tokens (plus prompt_tokens, completion_tokens, cache_write_tokens) + add_spend(session_id, usage['total_tokens']) + + def check_before_tool(self, session_id, tool_name): + return {'decision': 'allow'} + +opts = SessionOptions() +opts.budget_guard = MyBudgetGuard() +session = agent.session('/repo', opts) +``` + +The decision return dict `{"decision": "deny", "resource": ..., "reason": ...}` +(and `"soft"` / `"allow"`) is the same shape on both SDKs. diff --git a/website/docs/v8.5.1/en/guide/mcp.mdx b/website/docs/v8.5.1/en/guide/mcp.mdx new file mode 100644 index 00000000..0c4356bf --- /dev/null +++ b/website/docs/v8.5.1/en/guide/mcp.mdx @@ -0,0 +1,260 @@ +--- +title: 'MCP' +description: 'Model Context Protocol server integration' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# MCP + +MCP connects A3S Code sessions to external tool servers. A stdio server can be +added to a live session, inspected, called through its registered tools, and +removed. Tools from a server are registered into the session and named +`mcp____`. + +## Add A Server + +Use the object-shaped `addMcp(...)` API for new code. It maps directly to the +core `McpServerConfig` shape and avoids the positional overload's argument +ordering. + + + + +```rust +use a3s_code_core::mcp::{McpServerConfig, McpTransportConfig}; +use std::collections::HashMap; + +let count = session + .add_mcp_server(McpServerConfig { + name: "echo".into(), + transport: McpTransportConfig::Stdio { + command: "node".into(), + args: vec![ + "tools/mcp_echo_server.mjs".into(), + "example-value".into(), + ], + }, + enabled: true, + env: HashMap::new(), + oauth: None, + tool_timeout_secs: 30, + }) + .await?; + +println!("registered tools: {count}"); +``` + + + + +```ts +const count = await session.addMcp({ + name: 'echo', + transport: { + type: 'stdio', + command: process.execPath, + args: ['tools/mcp_echo_server.mjs', 'example-value'], + }, + timeoutMs: 30000, +}); + +console.log('registered tools:', count); +``` + + + + +```python +count = session.add_mcp({ + 'name': 'echo', + 'transport': { + 'type': 'stdio', + 'command': 'python', + 'args': ['tools/mcp_echo_server.py', 'example-value'], + }, + 'timeout_ms': 30000, +}) +print('registered tools:', count) +``` + + + + +```go +count, err := session.AddMCPServer(ctx, code.MCPServerConfig{ + Name: "echo", + Transport: code.MCPTransport{ + Type: "stdio", + Command: "node", + Args: []string{"tools/mcp_echo_server.mjs", "example-value"}, + }, + ToolTimeoutSecs: 30, +}) +if err != nil { + return err +} +fmt.Println("registered tools:", count) +``` + + + + +Python exposes the object-shaped API as `session.add_mcp({...})`. +`addMcpServer(...)` / `add_mcp_server(...)` and +`addMcpServerConfig(...)` / `add_mcp_server_config(...)` remain compatibility +aliases. + +For remote servers, use a nested transport object with `type: 'http'` or +`type: 'streamable-http'`. Supply credentials from the host environment or a +secret manager, and validate the server before relying on it in production docs +or release notes. + +## Inspect And Remove + + + + +```rust +use serde_json::json; + +println!("{:#?}", session.mcp_status().await); +println!( + "{:#?}", + session + .tool_names() + .into_iter() + .filter(|name| name.starts_with("mcp__")) + .collect::>() +); +let result = session + .tool("mcp__echo__echo", json!({ "message": "docs mcp ok" })) + .await?; +println!("{}", result.output); +session.remove_mcp_server("echo").await?; +``` + + + + +```ts +console.log(await session.mcpStatus()); +console.log(session.toolNames().filter((name) => name.startsWith('mcp__'))); +await session.tool('mcp__echo__echo', { message: 'docs mcp ok' }); +await session.removeMcpServer('echo'); +``` + + + + +```python +print(session.mcp_status()) +print([name for name in session.tool_names() if name.startswith('mcp__')]) +session.tool('mcp__echo__echo', {'message': 'docs mcp ok'}) +session.remove_mcp_server('echo') +``` + + + + +```go +status, err := session.MCPStatus(ctx) +if err != nil { + return err +} +names, err := session.ToolNames(ctx) +if err != nil { + return err +} +result, err := session.Tool(ctx, "mcp__echo__echo", map[string]any{ + "message": "docs mcp ok", +}) +if err != nil { + return err +} +fmt.Println(status, names, result.Output) +if err := session.RemoveMCPServer(ctx, "echo"); err != nil { + return err +} +``` + + + + +## Ownership And Session Isolation + +MCP capability discovery and live mutation have different owners: + +- The agent-global manager owns servers loaded from global configuration. +- A manager supplied through Rust `SessionOptions` is an inherited, read-only + capability source for that session. +- Every built session owns a new private manager for live `addMcp` and + `removeMcpServer` operations. + +Session assembly reads tool definitions from inherited sources without merging +session configuration back into those managers. A locally added server can +shadow an inherited server with the same fully qualified tool names, but only +inside that session. Removing the local server reveals the inherited tools +again. It does not unregister, disconnect, or otherwise mutate the global or +host-owned source, and sibling sessions remain unchanged. + +Delegated child agents receive the ordered capability sources so they can call +the same MCP tools without taking ownership. `session.close()` disconnects only +the private manager's servers. `agent.close()` and `disconnectIdleMcp(...)` +remain responsible for the agent-global manager. + +For Rust, a host-supplied MCP source requires async session construction so +discovery happens without blocking: + +```rust +let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_mcp(manager)) + .build() + .await?; +``` + +## Idle Disconnect + +A connected stdio MCP server holds resources — file descriptors and a +background worker — even while it sits idle. In a long-lived cluster session +those quiet servers accumulate. `disconnectIdleMcp` (CHANGELOG [3.3.0] "MCP idle +disconnect") reaps them on demand: it disconnects every global MCP server whose +last activity is older than the threshold, releasing the FDs and workers, **while +keeping each server's registered config** so a later tool call reconnects on +demand. + +This is an agent-level method (it operates only on the agent's global MCP manager, +backed by `McpManager::disconnect_idle(threshold_ms)`). It returns the list of +disconnected server names. + +```ts +// Reap servers idle longer than 5 minutes. Returns disconnected names. +const dropped = await agent.disconnectIdleMcp(5 * 60 * 1000); +console.log('disconnected:', dropped); +``` + +```python +# Reap servers idle longer than 5 minutes. Returns disconnected names. +dropped = agent.disconnect_idle_mcp(5 * 60 * 1000) +print('disconnected:', dropped) +``` + +The current Go bridge exposes session-local add, status, call, and remove plus +agent-level `RefreshMCPTools`; it does not yet expose the global idle-disconnect +sweeper. + +Hosts running thousands of long-lived sessions should call this periodically +from a sweeper (e.g. every 60s with a 5-min threshold). A subsequent tool call +on a disconnected server re-establishes the connection from its retained config +— no re-registration needed. + +Disconnect also purges orphan timestamps: a `touch()`-without-connect entry +(recorded for a server that was never actually connected) is swept on each +`disconnect_idle` call so the activity map cannot grow unbounded over a +long-running manager's lifetime (CHANGELOG [3.3.0] Fixed "MCP timestamp leak"). + +## Security + +Treat external tool servers as privileged integrations and keep secrets in +environment variables. diff --git a/website/docs/v8.5.1/en/guide/memory.mdx b/website/docs/v8.5.1/en/guide/memory.mdx new file mode 100644 index 00000000..336e333e --- /dev/null +++ b/website/docs/v8.5.1/en/guide/memory.mdx @@ -0,0 +1,122 @@ +--- +title: 'Memory' +description: 'Session memory stores and explicit recall' +--- + +# Memory + +Memory records reusable facts about previous work. It should help the harness recall patterns without flooding every prompt. + +## Default Store + +Every session gets a memory store by default. Plain SDK sessions use a +file-backed store at `/.a3s/memory` if you do not pass one. The +`a3s code` TUI uses the same `memory_dir` setting that its `/memory` panel +browses; by default that is `~/.a3s/memory`, so memories carry across projects +unless you configure a project-local path. Set `memory_dir` in config or pass a +typed store object to override the backend for one session. If the file store +cannot be created, the session falls back to an in-memory store and exposes an +init warning. + +## Override Stores + +```ts +import { FileMemoryStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), +}); +``` + +## LLM Extraction + +LLM memory extraction is enabled by default when memory is available. The agent +submits every completed non-empty session turn to the active model. The model, +not a keyword list or tool-type heuristic, decides whether the turn contains +anything that can change a future answer or action. It must return an empty +`items` array when nothing qualifies. Tool calls and tool results are context +for that judgment; they are never mechanically copied into long-term memory. + +Automatic output is deliberately narrow. The runtime accepts only `semantic` +memories for durable facts, preferences, and decisions, or `procedural` +memories for reusable workflows and failure lessons. Each item must include a +validated `source`, `importance`, `confidence`, `scope`, and a non-empty +future-value `reason`. Stored metadata also records the workspace, session ID, +and `a3s.memory.durable.v1` schema. The runtime rejects malformed output and +obvious API keys, tokens, password assignments, and private keys; these checks +validate structure and safety rather than judging semantic value. + +The extraction prompt includes a small set of related existing memories. The +model can use `supersedes` for a directly replaced or consolidated memory and +`conflicts_with` for a contradiction that should remain visible. The runtime +accepts only relation IDs it supplied to the model. It removes accepted +superseded items, preserves conflicts, and includes relation annotations in +future recall. The storage layer otherwise merges only normalized exact +duplicates; it does not infer semantic equivalence from term overlap. + +The TUI registers extraction before publishing the final event, then processes +completed turns in FIFO order in the +background. A slow extraction does not delay the final response, and a later +turn is queued rather than dropped. Graceful session close waits up to its +bounded extraction deadline for work accepted before close, preventing an +immediate UI shutdown from silently losing the completed turn. + +`rememberSuccess` and `rememberFailure` remain explicit SDK operations for a +host that intentionally wants to store those records. The normal agent runtime +does not call them for each tool result. + +## Store Hygiene + +Default memory stores return the canonical item for normalized exact duplicate +content, raising importance and preserving useful tags, provenance, and +relation metadata. Distinct but related wording remains separate unless the LLM +explicitly consolidates it through `supersedes`. + +Automatic pruning removes stale, low-importance memories when configured, but it +hard-protects curated memories: pinned/protected items, frequently recalled +items, consolidated memories, and memories carrying `supersedes` or +`conflicts_with` relation metadata. + +You can tune extraction limits in `config.acl`: + +```acl +memory { + llmExtraction = true + llmExtractionMaxItems = 5 + llmExtractionMaxInputChars = 8000 +} +``` + +## Write Memory + +```ts +await session.rememberSuccess( + 'release preflight', + ['bash', 'grep'], + 'Release checks passed after provider verification', +); + +await session.rememberFailure( + 'provider verification', + 'missing PROVIDER_BASE_URL', + ['bash'], +); +``` + +## Recall + +```ts +const related = await session.recallSimilar('release provider verification', 5); +const recent = await session.memoryRecent(10); +const tagged = await session.recallByTags(['release'], 10); +``` + +Use recalled memory as supporting context. Verification evidence still comes from current commands and traces. + +## Durable memory modes + +Production durable-memory bindings must use **Active recall** +(`DurableMemorySession::active_recall`). Candidate-only **shadow** mode +(`ShadowCandidates` / `::shadow`) is **removed** (`HARNESS-CONV4` / `CAP-GA1`). +Extraction may still write evidence-backed Candidate nodes; hosts activate them +explicitly before recall can admit them. diff --git a/website/docs/v8.5.1/en/guide/multi-machine.mdx b/website/docs/v8.5.1/en/guide/multi-machine.mdx new file mode 100644 index 00000000..eb9d793f --- /dev/null +++ b/website/docs/v8.5.1/en/guide/multi-machine.mdx @@ -0,0 +1,256 @@ +--- +title: 'Multi-Machine' +description: 'Placing orchestration steps across machines via the AgentExecutor seam' +--- + +# Multi-Machine + +A3S Code runs multi-agent orchestration as a _grammar_ expressed in code, then +places the resulting steps wherever you want them to run. The split is drawn +along the **framework / host boundary**, introduced in `[3.4.0]`: + +- The **framework** owns the orchestration grammar and the serializable data + contracts. It never decides where a step runs. +- The **host** owns placement, transport, and scheduling — which node executes + a step, how the spec gets there, and how concurrency maps to a cluster. + +The single point of contact between the two is one trait, `AgentExecutor`. + +## The framework / host boundary + +The framework's contract is two serializable types: + +- `AgentStepSpec` — _what_ to run, independent of _where_: `task_id`, `agent`, + `description`, `prompt`, and optional `max_steps`, `parent_session_id`, + `output_schema`. +- `StepOutcome` — the result of running one spec: `task_id`, `session_id`, + `agent`, `output`, `success`, and optional `structured`. + +Both serialize cleanly, so a host may ship a spec to another node and persist an +outcome in a checkpoint. The combinators that compose specs are written purely +against the `AgentExecutor` trait and never observe where a step ran, so the +same orchestration scales from one process to a cluster without changing. + +## The AgentExecutor seam + +`AgentExecutor` is the boundary between the grammar and the host: + +```text +combinators (parallel / pipeline / resumable) + -> AgentExecutor::execute_step(spec) -> StepOutcome + ├─ in-box TaskExecutor: runs the step as a local child agent + └─ host executor: places the step on a remote node +``` + +The in-box `TaskExecutor` runs each step as a child agent locally — in-process, +on Tokio — inheriting the session's agent registry, LLM client, workspace, MCP +tools, and subagent tracker. A host such as a cluster runtime substitutes its own executor +to place steps across a cluster; the combinators are unaffected. + +`concurrency_hint()` is **advisory, not a hard local bound**. The local default +returns the session's `max_parallel_tasks`; a scheduler-backed host may return +its cluster-wide target. Because it is a hint rather than a ceiling, +orchestration scales past a single process. + +A session exposes the in-box seam directly: + +- `AgentSession::agent_executor()` returns a session-backed `AgentExecutor`. +- `AgentSession::session_store()` returns the session's store (when one is + configured), which the resumable combinator needs to journal progress. + +The SDK grammar below calls `agent_executor()` for you; you only reach for these +when implementing or substituting a custom executor. + +## Parallel: barrier fan-out + +`execute_steps_parallel` fans `specs` out across the executor and awaits all of +them (a barrier). Results preserve input order, a panicked branch becomes a +failed `StepOutcome` instead of dropping the batch, and concurrency is bounded +by the executor's concurrency hint. + +```ts +const outcomes = await session.parallel([ + { + taskId: 'a', + agent: 'explore', + description: 'survey', + prompt: 'Map the auth module', + }, + { + taskId: 'b', + agent: 'review', + description: 'audit', + prompt: 'Review error handling', + }, +]); + +for (const o of outcomes) { + console.log(o.taskId, o.success, o.output); +} +``` + +```python +outcomes = session.parallel([ + {"task_id": "a", "agent": "explore", "description": "survey", "prompt": "Map the auth module"}, + {"task_id": "b", "agent": "review", "description": "audit", "prompt": "Review error handling"}, +]) + +for o in outcomes: + print(o["task_id"], o["success"], o["output"]) +``` + +## Pipeline: per-item chains, no inter-stage barrier + +`execute_pipeline` runs each item through a chain of `PipelineStage`s +independently. There is **no barrier between stages** — item A can be in stage 3 +while item B is still in stage 1 — so wall-clock is the slowest single chain, +not the sum of the slowest step per stage. + +A stage is a `(ctx) => spec | null` callback where `ctx` carries the previous +outcome and the original item. Return a spec to run the next step, or `null` to +stop that item's chain early; a chain also stops when a step fails (later stages +would only build on a failed result). The bridges **fail closed**: a stage that +hangs, returns `null`, or raises stops only its own chain. + +> A Node stage callback **must not throw** — wrap your logic in `try`/`catch` +> and return `null` on error (the same constraint as `setBudgetGuard`). A stage +> that hangs past the timeout fails closed for that chain. A Python stage that +> raises is caught and treated as `null`. + +```ts +const outcomes = await session.pipeline( + ['src/auth', 'src/api'], + [ + (ctx) => ({ + taskId: 's1', + agent: 'explore', + description: 'survey', + prompt: `Survey ${ctx.item}`, + }), + (ctx) => + ctx.previous?.success + ? { + taskId: 's2', + agent: 'review', + description: 'review', + prompt: `Review: ${ctx.previous.output}`, + } + : null, + ], +); +``` + +```python +def survey(ctx): + return {"task_id": "s1", "agent": "explore", "description": "survey", + "prompt": f"Survey {ctx['item']}"} + +def review(ctx): + prev = ctx["previous"] + if prev and prev["success"]: + return {"task_id": "s2", "agent": "review", "description": "review", + "prompt": f"Review: {prev['output']}"} + return None + +outcomes = session.pipeline(["src/auth", "src/api"], [survey, review]) +``` + +## Resumable: cross-node resume + +`execute_steps_parallel_resumable` is `parallel` plus a journal. At each step +boundary it writes a `WorkflowCheckpoint` to the `SessionStore` under a +`workflowId`. On resume it skips already-completed steps and re-dispatches the +rest. It records **only successful steps**, so a failed step retries on resume — +its effect did not complete. + +A `SessionStore` is required. The Node bridge rejects with +`"parallelResumable requires a sessionStore"` when none is configured; Python +raises the equivalent. + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +const outcomes = await session.parallelResumable( + [ + { + taskId: 'a', + agent: 'explore', + description: 'survey', + prompt: 'Map the auth module', + }, + { + taskId: 'b', + agent: 'review', + description: 'audit', + prompt: 'Review error handling', + }, + ], + 'release-audit', +); +``` + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session("/repo", opts) + +outcomes = session.parallel_resumable([ + {"task_id": "a", "agent": "explore", "description": "survey", "prompt": "Map the auth module"}, + {"task_id": "b", "agent": "review", "description": "audit", "prompt": "Review error handling"}, +], "release-audit") +``` + +Because the checkpoint is serializable and the executor is a parameter, a host +can resume an interrupted workflow on a **different node** by passing that node's +executor — the framework migrates the _what_, the host supplies the _where_. + +## Schema-forced step output + +A spec carrying an `output_schema` (`outputSchema` in Node) must return a value +conforming to that JSON Schema; the validated object lands in +`StepOutcome.structured`. The executor reuses the structured-output coercion and +repair machinery. A coercion failure **demotes the step to unsuccessful**, so a +caller never treats unvalidated text as the promised object. + +```ts +const [finding] = await session.parallel([ + { + taskId: 'classify', + agent: 'review', + description: 'classify', + prompt: 'Classify this defect', + outputSchema: { + type: 'object', + properties: { severity: { type: 'string' }, summary: { type: 'string' } }, + required: ['severity', 'summary'], + }, + }, +]); + +if (finding.success && finding.structured) { + console.log(finding.structured.severity); +} +``` + +## Placing orchestration steps across machines + +Lane queues remain a valid transport — but they are now **one option behind the +`AgentExecutor` seam, not the sole integration point**. To distribute work, a +host implements `AgentExecutor::execute_step` over whatever transport fits its +platform (HTTP, message queues, a job system, lane queues, internal RPC), wires +`concurrency_hint()` to its cluster target, and hands that executor to the +combinators. The grammar — parallel, pipeline, resumable — stays identical; only +placement moves. + +The result: a coordinator session owns the conversation, final synthesis, and +release decision, while its orchestration steps execute wherever the host places +them, with cross-node resume carried by the serializable checkpoint. + +See [Orchestration](/guide/orchestration) for the combinator grammar in +depth and [Persistence](/guide/persistence) for the checkpoint store. diff --git a/website/docs/v8.5.1/en/guide/orchestration.mdx b/website/docs/v8.5.1/en/guide/orchestration.mdx new file mode 100644 index 00000000..42130512 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/orchestration.mdx @@ -0,0 +1,470 @@ +--- +title: 'Orchestration' +description: 'Programmable, deterministic multi-agent orchestration: parallel fan-out, barrier-free pipelines, and resumable workflows' +--- + +# Orchestration + +Orchestration is an **Advanced** host-authored surface — the _programmable_ sibling of model-driven delegation. For ordinary coding-agent fan-out, prefer unified [`task`](/guide/tasks). With +[Tasks](/guide/tasks) and [Teams](/guide/teams) the LLM decides, at +run time, to call `task` with one or more `tasks[]` items — the shape of the +fan-out is whatever the model chose. Orchestration moves that decision into your code: you +express fan-out, pipelines, and verification panels as a grammar, so the shape +is reproducible, testable, budget-bounded, and resumable — independent of what +the model picks. + +Reach for orchestration when the _structure_ of the work is known to the host +ahead of time (run these three reviewers in parallel; flow each candidate +through explore → verify → review; resume this batch after a crash). Reach for +Tasks/Teams when you want the model to decide whether and how to delegate. + +## The framework / host boundary (the seam) + +Everything in this layer is written against a single seam, `AgentExecutor`: +"run this step, give me the result." That seam splits responsibilities cleanly: + +- The **framework** owns the _grammar_ — which steps exist, how they compose, + the concurrency _hint_, and the serializable contracts `AgentStepSpec` / + `StepOutcome`. +- The **host** owns _placement_ — transport, scheduling, and where a + step actually runs. + +The in-box default executor (`TaskExecutor`) runs every step locally, +in-process, on tokio. A host substitutes its own `AgentExecutor` to place steps +across a cluster; the combinators never observe where a step ran, so the same +orchestration scales from one process to a cluster without change. + +`concurrency_hint()` is **advisory**, not a hard local bound — it is the lever +that lets orchestration scale past a single process (a scheduler-backed host +returns its cluster-wide target instead of a local cap). + +The SDK wires this up for you: `AgentSession::agent_executor()` returns the +session-backed executor (it runs each step as a child agent on this node, +inheriting the session's agent registry, LLM client, workspace, and MCP tools), +and `session_store()` returns the session's store. The `parallel` / +`pipeline` / `parallelResumable` methods call these for you. + +## Step contracts + +A step is described by an `AgentStepSpec` and resolves to a `StepOutcome`. Both +are serializable on purpose: a host may ship a spec to another node, and the +resumable combinator persists outcomes into checkpoints. + +`AgentStepSpec` fields: + +- `task_id` — stable id for the step (you assign it); flows into lifecycle + events and checkpoints. +- `agent` — registry key of the agent to run (e.g. `explore`, `review`). +- `description` — short human label for display/tracking. +- `prompt` — the instruction handed to the child agent. +- `max_steps` _(optional)_ — per-step tool-round cap. +- `parent_session_id` _(optional)_ — parent session id for event correlation. +- `output_schema` _(optional)_ — when set, the step must return a value + conforming to this JSON Schema (see + [Schema-forced step output](#schema-forced-step-output)). + +`StepOutcome` fields: + +- `task_id` — the originating step's id. +- `session_id` — the child run's session id (failed steps remain addressable). +- `agent` — the agent that ran. +- `output` — the step's text output. +- `success` — `false` for a failed or panicked step (never a dropped sibling). +- `structured` _(optional)_ — schema-validated object, present only when the + spec carried an `output_schema`. + +The key casing differs by SDK: + +| Concept | Node (camelCase) | Python (snake_case) | +| -------------- | ----------------- | ------------------- | +| step id | `taskId` | `task_id` | +| tool-round cap | `maxSteps` | `max_steps` | +| parent session | `parentSessionId` | `parent_session_id` | +| forced schema | `outputSchema` | `output_schema` | + +`agent`, `description`, `prompt`, `output`, `success`, and `structured` are +spelled the same in both SDKs. + +## parallel — barrier fan-out + +`session.parallel(specs)` runs every spec as a fan-out and resolves with one +`StepOutcome` per spec, **in input order**. It maps to the core +`execute_steps_parallel` combinator. It is a **barrier**: it awaits every step +before returning. + +Each branch is isolated — a step that fails _or panics_ becomes +`success: false`; it never drops a sibling. Concurrency is bounded by the +executor's concurrency hint (the session's configured parallelism by default). + +```ts +const outcomes = await session.parallel([ + { + taskId: 'explore', + agent: 'explore', + description: 'Risky changes', + prompt: 'Find risky changed files in this diff.', + }, + { + taskId: 'verify', + agent: 'verification', + description: 'Test gaps', + prompt: 'Identify missing or weak verification.', + }, + { + taskId: 'review', + agent: 'review', + description: 'Correctness', + prompt: 'Review the diff for correctness risks.', + }, +]); + +for (const outcome of outcomes) { + if (outcome.success) { + console.log(outcome.taskId, outcome.output); + } else { + console.warn('failed:', outcome.taskId, outcome.output); + } +} +``` + +```python +outcomes = session.parallel([ + {"task_id": "explore", "agent": "explore", "description": "Risky changes", + "prompt": "Find risky changed files in this diff."}, + {"task_id": "verify", "agent": "verification", "description": "Test gaps", + "prompt": "Identify missing or weak verification."}, + {"task_id": "review", "agent": "review", "description": "Correctness", + "prompt": "Review the diff for correctness risks."}, +]) + +for outcome in outcomes: + if outcome["success"]: + print(outcome["task_id"], outcome["output"]) + else: + print("failed:", outcome["task_id"], outcome["output"]) +``` + +## pipeline — no barrier between stages + +`session.pipeline(items, stages)` flows each item through a chain of stages +**independently** — there is no barrier between stages, so item A can be in +stage 3 while item B is still in stage 1. Wall-clock time is the slowest _single +chain_, not the sum-of-slowest-per-stage that a per-stage barrier would incur. + +Stages are spec-builders, not specs: each stage receives the prior outcome and +the original item and returns the next step to run, or `null` / `None` to stop +that item's chain early. A failed step also stops the chain (a later stage would +only build on a failed result). The callback shapes: + +- Node: `(ctx) => spec | null` where `ctx = { previous: StepOutcome | null, item }` +- Python: `stage(ctx) -> spec | None` where `ctx = {"previous": , "item": }` + +A stage can branch on the prior outcome — e.g. "verify the finding the review +stage produced". + +Constraints (from the source): + +- Per-stage `output_schema` is **not supported** on pipeline stages — use + [`parallel`](#parallel--barrier-fan-out) for schema-validated steps. +- **Node:** a stage callback **must not throw** — a throw aborts the process + (the same constraint as `setBudgetGuard`). Wrap your logic in `try/catch` and + return `null` on error. +- **Node:** a stage that hangs past `timeoutMs` (the 3rd argument, default + `30000`) fails closed — it is treated as `null`, stopping only that chain. +- **Python:** a stage callable that raises is caught and treated as `None` + (stops only that chain). + +> A Node pipeline stage callback **must not throw.** In this napi version a JS +> throw at return-conversion aborts the process (the same fail-closed constraint +> as `setBudgetGuard`). Always wrap stage logic in `try/catch` and `return null` +> on error. + +```ts +const outcomes = await session.pipeline( + ['src/auth.ts', 'src/payments.ts'], + [ + (ctx) => ({ + taskId: `explore-${ctx.item}`, + agent: 'explore', + description: 'Inspect file', + prompt: `Summarize the responsibilities and risks of ${ctx.item}.`, + }), + (ctx) => { + try { + if (!ctx.previous) return null; + return { + taskId: `review-${ctx.item}`, + agent: 'review', + description: 'Review of prior finding', + prompt: `Review this summary for correctness risks:\n${ctx.previous.output}`, + }; + } catch { + return null; // stages must not throw + } + }, + ], +); +``` + +```python +def explore_stage(ctx): + item = ctx["item"] + return { + "task_id": f"explore-{item}", + "agent": "explore", + "description": "Inspect file", + "prompt": f"Summarize the responsibilities and risks of {item}.", + } + +def review_stage(ctx): + prev = ctx["previous"] + if prev is None: + return None + item = ctx["item"] + return { + "task_id": f"review-{item}", + "agent": "review", + "description": "Review of prior finding", + "prompt": f"Review this summary for correctness risks:\n{prev['output']}", + } + +outcomes = session.pipeline( + ["src/auth.ts", "src/payments.ts"], + [explore_stage, review_stage], +) +``` + +## Resumable / migratable workflows + +`session.parallelResumable(specs, workflowId)` (Node) / +`session.parallel_resumable(specs, workflow_id)` (Python) is `parallel` plus a +journal. It maps to `execute_steps_parallel_resumable`. + +At each step boundary it writes a `WorkflowCheckpoint` to the session store. On +resume it skips already-completed steps (reusing their cached outcomes) and +re-dispatches only the rest. It records **only successful steps** — a failed +step is not journaled, so it retries on resume. On full success the checkpoint +is deleted; only a crash leaves one behind for resume. + +Because the checkpoint is serializable and the executor is a parameter, a host +can resume an interrupted workflow on a **different node** (migration) by +passing that node's executor. + +This combinator **requires a configured session store** — both SDK methods +reject/raise without one (the Node error message is +`parallelResumable requires a sessionStore`). + +The `WorkflowCheckpoint` schema is `schema_version` / `workflow_id` / `steps` / +`checkpoint_ms`. A checkpoint written by a _future_, incompatible +`schema_version` is rejected on load (`ensure_loadable`). That failure is +fail-safe, not fatal: an unreadable checkpoint logs a warning and the workflow +re-runs from scratch rather than resuming from state it can't interpret. + +See [Persistence](/guide/persistence) for the store and +[Multi-Machine](/guide/multi-machine) for the migration path. + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +// First attempt — may be interrupted partway through. +let outcomes = await session.parallelResumable(specs, 'release-batch-42'); + +// After a crash/restart: same workflowId resumes, skipping completed steps. +outcomes = await session.parallelResumable(specs, 'release-batch-42'); +``` + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session("/repo", opts) + +# First attempt — may be interrupted partway through. +outcomes = session.parallel_resumable(specs, "release-batch-42") + +# After a crash/restart: same workflow_id resumes, skipping completed steps. +outcomes = session.parallel_resumable(specs, "release-batch-42") +``` + +## Shared budget across a fan-out + +By default each child agent counts its own LLM cost. Pass a token budget to +`parallel` and every child instead feeds **one shared ledger** — a single cap +for the whole fan-out. It maps to the core `WorkflowBudget`, an aggregating +`BudgetGuard` installed onto each child run. + +The budget is an **optional argument**, so it is backward compatible: + +- Without a budget, `parallel(specs)` returns the plain outcomes array, exactly + as before. +- With a budget, `parallel(specs, budgetTokens)` resolves to `{ outcomes, budget }`, + where `budget` is the ledger snapshot (`consumedTokens` / `limitTokens`). + +Once the cap is reached, a step that _starts_ afterwards is denied — its outcome +is `success: false` with a budget-exhausted message. It is a **soft cap**: +because usage is recorded _after_ each LLM call, a wide fan-out can race a few +in-flight turns past the cap before the ledger catches up. The framework never +force-kills an in-flight fan-out; an exhausted budget simply denies the _next_ +LLM call. + +```ts +// No budget → plain outcomes array (unchanged). +const outcomes = await session.parallel(specs); + +// With a budget → { outcomes, budget }. All children share one ledger. +const { outcomes: out, budget } = await session.parallel(specs, 500_000); +console.log(budget.consumedTokens, budget.limitTokens); // e.g. 48213, 500000 +// TS return type is a union: Array | { outcomes, budget }. +``` + +```python +# No budget → plain list (unchanged). +outcomes = session.parallel(specs) + +# With a budget → {"outcomes": [...], "budget": {"consumed_tokens", "limit_tokens"}}. +res = session.parallel(specs, budget_tokens=500_000) +print(res["budget"]["consumed_tokens"], res["budget"]["limit_tokens"]) +``` + +## Looping until done (`execute_loop`) + +For unknown-length, iterate-until-converge work (loop-until-dry, refine-until-good), +the core grammar adds `execute_loop`. Each round is a barrier (`execute_steps_parallel`); +a host-supplied predicate sees the round's outcomes and returns +`LoopDecision::Continue(next_specs)` or `LoopDecision::Stop`. A required +`max_iterations` is a **hard cap** — once reached the loop stops even if the +predicate would continue, so an LLM-driven loop can never run away. + +```rust +use a3s_code_core::orchestration::{execute_loop, AgentStepSpec, LoopDecision}; + +let outcomes = execute_loop(executor, initial_specs, /* max_iterations */ 5, None, |round| { + // Stop when the round produced no new findings; otherwise fan out follow-ups. + let follow_ups = derive_follow_ups(round); + if follow_ups.is_empty() { + LoopDecision::Stop + } else { + LoopDecision::Continue(follow_ups) + } +}) +.await; +``` + +> From the host SDKs you don't need a dedicated `loop` verb — write the loop in +> your own language (`while`/`for`) around `parallel`, deciding the next round +> from the outcomes. `execute_loop` exists for the Rust grammar and to give the +> loop a single, enforced termination guard. + +## The Workflow facade (Rust / embedding) + +`session.workflow()` returns a cheaply-clonable `Workflow` that pre-wires the +session's executor, persistence store, per-step event stream, and a stable, +session-derived root id. It is the programmable handle that bundles everything +above; control flow is ordinary Rust — `await` a verb, inspect the outcomes, +decide what runs next. + +- **Verbs** — `agent` (one step), `parallel` (barrier fan-out), `phase` (a + _named_, resumable barrier that emits milestones), `pipeline` (per-item + chains), and the non-failing `log`. Each delegates to exactly one combinator. +- **Phases & events** — `phase(name, specs)` derives a deterministic checkpoint + id (`{root}/{index}:{name}`), runs the resumable barrier when a store is + present, and emits `WorkflowEvent::PhaseStart` / `PhaseEnd` on a broadcast you + read with `subscribe()`. `log()` emits `WorkflowEvent::Log`. +- **Budget** — `session.workflow_with_token_budget(Some(limit))` installs a + shared `WorkflowBudget`; `budget_snapshot()` reads the ledger and a + `WorkflowEvent::BudgetExhausted` fires once the cap is hit. + +```rust +let wf = session.workflow(); // or session.workflow_with_token_budget(Some(500_000)) +let mut events = wf.subscribe(); + +// One step, then a *variable* fan-out computed from its result — the "dynamic" +// part: the shape is decided at run time, not declared up front. +let plan = wf.agent(AgentStepSpec::new("plan", "plan", "plan", goal)).await; +let specs = derive_specs(&plan); // your code +let done = wf.phase("implement", specs).await; // resumable barrier + milestones +let reviews = wf.phase("review", to_review(&done)).await; // budget shared across phases + +if let Some(b) = wf.budget_snapshot() { + println!("spent {} / {:?} tokens", b.consumed_tokens, b.limit_tokens); +} +``` + +The SDKs expose the flat `parallel` / `pipeline` / `parallelResumable` verbs (and +the `parallel` budget overload above); the full `Workflow` handle — phases, +event subscription, the loop combinator — is a Rust/embedding API. + +## Schema-forced step output + +A spec carrying `output_schema` (`outputSchema` in Node) forces the step to +return a value conforming to that JSON Schema; the validated object lands in +`StepOutcome.structured`. This reuses the same structured-output coercion + +repair machinery as the rest of A3S Code. A coercion failure **demotes the step +to unsuccessful** (`success: false`), so callers never treat unvalidated text as +the promised object. + +Forced schema applies to `parallel` / `parallelResumable` specs only — **not** +pipeline stages. + +```ts +const [outcome] = await session.parallel([ + { + taskId: 'triage', + agent: 'review', + description: 'Structured triage', + prompt: 'Triage this diff.', + outputSchema: { + type: 'object', + properties: { + severity: { type: 'string', enum: ['low', 'medium', 'high'] }, + summary: { type: 'string' }, + }, + required: ['severity', 'summary'], + }, + }, +]); + +if (outcome.success) { + console.log(outcome.structured.severity, outcome.structured.summary); +} +``` + +```python +outcomes = session.parallel([ + { + "task_id": "triage", + "agent": "review", + "description": "Structured triage", + "prompt": "Triage this diff.", + "output_schema": { + "type": "object", + "properties": { + "severity": {"type": "string", "enum": ["low", "medium", "high"]}, + "summary": {"type": "string"}, + }, + "required": ["severity", "summary"], + }, + }, +]) + +outcome = outcomes[0] +if outcome["success"]: + print(outcome["structured"]["severity"], outcome["structured"]["summary"]) +``` + +## Cost governance & lifecycle + +Orchestrated steps run through the same session, so the session's controls apply +to them directly. `setBudgetGuard` (Node) / `budget_guard` (Python) bounds the +LLM cost of every step; `close()` cancels in-flight steps along with the rest of +the session's work; and the host-provided identity labels (`tenant_id`, +`principal`, `agent_template_id`, `correlation_id`) flow through each step for +host-side aggregation and billing. See [Sessions](/guide/sessions) and +[Limits](/guide/limits) for the details of those controls. diff --git a/website/docs/v8.5.1/en/guide/persistence.mdx b/website/docs/v8.5.1/en/guide/persistence.mdx new file mode 100644 index 00000000..d739817b --- /dev/null +++ b/website/docs/v8.5.1/en/guide/persistence.mdx @@ -0,0 +1,297 @@ +--- +title: 'Persistence' +description: 'Saving and resuming sessions' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Persistence + +Persistence lets a session survive process restarts and gives product surfaces a stable session ID. + +A resumed session can repopulate task lists, execution history, artifacts, and +delivery summaries without replaying the completed run. + +A3S Code persists three related but different things: + +| Object | Written by | Resumed by | Purpose | +| ------------------- | ------------------------------------- | -------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | +| `SessionSnapshotV1` | `session.save()` or `autoSave` | `agent.resumeSession(id, options)` | Rehydrate one versioned generation containing the conversation, artifacts, traces, run records, verification reports, and subagent task snapshots. | +| Loop checkpoint | Agent loop while a run is in progress | `session.resumeRun(runId)` | Continue an interrupted run from the last completed tool-round boundary. Completed in-process runs delete this checkpoint. | +| Workflow checkpoint | `parallelResumable` / workflow phases | `parallelResumable(specs, workflowId)` | Skip already-completed orchestration steps after a process restart. | + +## File Session Store + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +let options = SessionOptions::new() + .with_session_id("release-review") + .with_file_session_store("./.a3s/sessions") + .with_auto_save(true); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; + +session.send("Review release readiness", None).await?; +session.save().await?; +``` + + + + +```ts title=persistence.ts +import { FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + // !mark(1:3) Snapshot config + sessionId: 'release-review', + sessionStore: new FileSessionStore('./.a3s/sessions'), + autoSave: true, +}); + +await session.send('Review release readiness'); +await session.save(); +``` + + + + +```python +from a3s_code import FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_id = 'release-review' +opts.session_store = FileSessionStore('./.a3s/sessions') +opts.auto_save = True + +session = agent.session('/repo', opts) +session.send('Review release readiness') +session.save() +``` + + + + +```go +options := &code.SessionOptions{ + SessionID: "release-review", + FileSessionStoreDir: ".a3s/sessions", + AutoSave: code.Ptr(true), +} +session, err := agent.Session(ctx, "/repo", options) +if err != nil { + return err +} +if _, err := session.Run(ctx, "Review release readiness"); err != nil { + return err +} +if err := session.Save(ctx); err != nil { + return err +} +``` + + + + +Go selects the built-in file store with `FileSessionStoreDir`; custom +`SessionStore` trait implementations remain a Rust embedding capability. + +## Atomic Snapshot Generations + +`session.save()` gathers the current persisted state into one +`SessionSnapshotV1` and calls `SessionStore::save_snapshot` once. The envelope +contains: + +- `schema_version` and `SessionData` +- tool artifacts +- trace events and run records +- verification reports +- delegated subagent task snapshots + +The file store writes one complete JSON envelope to a synced temporary file and +atomically replaces `.json`. Readers therefore observe the previous +generation or the next one, not a new conversation paired with old run or trace +fragments. The memory store publishes the same aggregate under one lock. Both +report `SessionStoreCapabilities { atomic_session_snapshots: true }`. + +Historical files remain readable. A bare `SessionData` file is combined with +the legacy artifact/trace/run/verification/subagent fragment locations during +load, then restored through the v1 in-memory shape. Once a new aggregate is +saved, the single envelope is authoritative. A document that already looks like +an aggregate but has a malformed or unsupported schema is rejected rather than +reinterpreted as legacy data. + +Custom stores must implement `save_snapshot` explicitly. Its default returns an +error; it does not fan an aggregate out into independent writes or silently +acknowledge a no-op. The default `load_snapshot` exists only as best-effort +legacy assembly, and `capabilities()` lets a host distinguish that behavior +from an atomic backend. + +## Negotiated durability (v8.5.1) + +`SessionStoreCapabilities` now also advertises optional KRN-6 guarantees. +Built-in adapters only set a flag when they prove the semantic; hosts must +negotiate before relying on it: + +| Capability | Meaning | +| ----------------------------- | ------------------------------------------------------------------------------------------------- | +| `aggregate_cas` | `save_snapshot_cas` commits only when the expected digest matches (or is omitted). | +| `append_only_event_log` | File store appends digest-only Intent/Committed WAL records around each atomic replace. | +| `lease_fencing` | `acquire_writer_lease` publishes a durable epoch; stale holders fail closed after takeover. | +| `encrypted_at_rest` | `FileSessionStore::with_encryption_key` seals documents with AES-256-GCM; wrong keys fail closed. | +| `watch` | `watch_commits` delivers digest-only `SessionStoreCommitEventV1` values after durable commits. | +| `reference_aware_artifact_gc` | Artifact retention pins host URI roots so limit eviction cannot drop reachable content. | + +The digest-only WAL stays unencrypted even when encryption at rest is enabled. +Memory and file adapters continue to advertise the atomic snapshot they already +proved; the new flags default to unadvertised until configured. + +## Resume + + + + +```rust +use a3s_code_core::SessionOptions; + +let resumed = agent + .resume_session_async( + "release-review", + SessionOptions::new().with_file_session_store("./.a3s/sessions"), + ) + .await?; +``` + + + + +```ts +const resumed = agent.resumeSession('release-review', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + + + + +```python +opts = SessionOptions() +opts.session_store = FileSessionStore('./.a3s/sessions') +resumed = agent.resume_session('release-review', opts) +``` + + + + +```go +resumed, err := agent.ResumeSession(ctx, "release-review", &code.SessionOptions{ + FileSessionStoreDir: ".a3s/sessions", +}) +``` + + + + +`resumeSession` restores a saved session snapshot. It is not the same as +`resumeRun`: use `resumeSession` when the user is continuing a saved +conversation, and use `resumeRun(runId)` only when a run checkpoint exists for +an interrupted run. Resume validates the snapshot schema before restoring any +history or runtime evidence. + +## Memory And Sessions + +Session persistence stores conversation and replay evidence. Memory stores +reusable task facts. Use both when you want resumable workflows that also learn +from repeated tasks. + +## Loop Checkpoints and Run Resumption + +When a `SessionStore` is configured, the agent loop persists a `LoopCheckpoint` +after each completed tool round. The boundary policy is strict: checkpoints are +taken **only** between tool rounds, never mid-tool. If a process dies while a +tool is executing, that round's work is lost on resume and the LLM +re-deliberates from the previous checkpoint — re-running a non-idempotent tool +(write, bash) on the wrong side of the boundary is worse than re-asking the LLM. + +`session.resumeRun(runId)` (Node) / `session.resume_run(run_id)` (Python) / +`session.ResumeRun(ctx, runID)` (Go) — core +`AgentSession::resume_run(checkpoint_run_id)` — loads the latest checkpoint +stored under that run ID and replays the loop from the last boundary. Because +the checkpoint lives in the shared store, resume can happen on **any** node that +shares it. Cumulative accounting continues rather than restarting at zero: +`total_usage` and `tool_calls_count` carry forward from the checkpoint. A new run +ID is allocated for the resumed work; the old-to-new relationship is host +metadata, not interpreted by the framework. + +Completed runs are not resumed through `resumeRun`; their final state is +available through `runs()`, `runEvents(runId)`, artifacts, verification reports, +and session snapshots. + +```ts +const result = await session.resumeRun('run-abc123'); +console.log(result.totalTokens); +``` + +```python +result = session.resume_run('run-abc123') +print(result.total_tokens) +``` + +```go +result, err := session.ResumeRun(ctx, "run-abc123") +if err != nil { + return err +} +fmt.Println(result.Usage.TotalTokens) +``` + +Go also exposes workflow checkpoint orchestration through +`ParallelResumable(ctx, specs, workflowID)`. + +`resume_run` rejects when the session has no `sessionStore` configured (or when +no checkpoint exists for the given ID). `SessionStore` gains +`save_loop_checkpoint` / `load_loop_checkpoint` / `delete_loop_checkpoint`; the +file store's writes are crash-atomic. `LoopCheckpoint::ensure_loadable()` is +called right after deserialization and rejects checkpoints from a future, +incompatible schema version, so neither `resume_run` nor the live-run sink acts +on an unreadable checkpoint. + +See CHANGELOG `[3.3.0]` — "Loop checkpoints + run resumption" — and `[3.4.0]` +"LoopCheckpoint::ensure_loadable()". + +## Workflow Checkpoints + +`WorkflowCheckpoint` is the step-boundary analogue of the tool-round +`LoopCheckpoint`, one level up: it journals completed orchestration steps so an +interrupted workflow resumes from the longest completed prefix. Its fields are +`schema_version`, `workflow_id`, `steps`, and `checkpoint_ms`, with the schema +pinned by the `WORKFLOW_CHECKPOINT_SCHEMA_VERSION` constant. A resuming run skips +the recorded steps and re-dispatches only the rest. + +`SessionStore` gains `save_workflow_checkpoint` / `load_workflow_checkpoint` / +`delete_workflow_checkpoint` (default no-ops; the file store writes +crash-atomically). Loads from a future, incompatible schema version are rejected +via `WorkflowCheckpoint::ensure_loadable()`. + +This pairs with the orchestration grammar — see +[Orchestration](/guide/orchestration) and +[Multi-Machine](/guide/multi-machine). + +See CHANGELOG `[3.4.0]` — "WorkflowCheckpoint". + +## Resuming on a Different Node + +Both checkpoint types are serializable. Combined with a shared `SessionStore` +and a pluggable executor, this lets a host resume an interrupted run or workflow +on a **different** node from the one that started it — the framework owns the +serializable contract, the host owns placement and transport. + +## Operational Notes + +Keep session stores out of public commits when they may contain prompts, tool output, or private file paths. diff --git a/website/docs/v8.5.1/en/guide/providers.mdx b/website/docs/v8.5.1/en/guide/providers.mdx new file mode 100644 index 00000000..4d9b48de --- /dev/null +++ b/website/docs/v8.5.1/en/guide/providers.mdx @@ -0,0 +1,131 @@ +--- +title: 'Providers' +description: 'ACL provider configuration and environment injection' +--- + +# Providers + +A3S Code reads runtime configuration from ACL. A config source can be an +`.acl` file path or an inline ACL string. JSON and legacy HCL configs are not +part of the current config surface. + +## Basic Shape + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + name = "Human readable model name" + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +`apiKey` and `api_key` are accepted aliases. `baseUrl` and `base_url` are +accepted aliases. Commit templates and non-secret defaults only; keep real API +keys, private endpoints, and account-specific model names in the environment or +in a secret manager. + +## Provider Families + +The built-in factory covers three paths: + +| Provider name | Client path | Notes | +| ---------------------------- | -------------------------- | ------------------------------------------------------------- | +| `anthropic` / `claude` | Anthropic client | Uses the configured model id and optional provider base URL. | +| `openai` / `gpt` | OpenAI-compatible client | Use this for OpenAI-compatible Chat Completions endpoints. | +| `glm` / `zhipu` / `bigmodel` | Zhipu-compatible client | Use environment variables for the key and base URL. | +| Any other provider name | OpenAI-compatible fallback | Useful for private or self-hosted OpenAI-compatible services. | + +The runtime does not hard-code model names. `default_model` and per-session +`model` values are identifiers in the `provider/model-id` form that must match +the provider blocks you define. + +## Delegation Controls + +```acl +agent_dirs = ["./.a3s/agents"] + +auto_delegation { + enabled = true + auto_parallel = false + allow_manual_delegation = true + min_confidence = 0.72 + max_tasks = 4 +} +``` + +`max_parallel_tasks` limits bounded sibling fan-out. `auto_delegation.enabled` +turns on automatic subagent delegation. The top-level `auto_parallel = false` +overrides `auto_delegation.auto_parallel` and disables only automatic parallel +child-agent fan-out; manual `task` fan-out remains available. When +`allow_manual_delegation = false`, the model-visible `task` tool is not +registered. + +## Storage + +```acl +storage_backend = "memory" +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +Use memory storage for short-lived tests. Use `storage_backend = "file"` plus +`sessions_dir` for resumable local sessions loaded from ACL. `storage_url` is +parsed for custom storage metadata, but it does not create a file-backed +session store by itself. SDK hosts can always pass +`sessionStore: new FileSessionStore(...)` / +`opts.session_store = FileSessionStore(...)` directly. + +## Private Provider Checks + +Real-provider smoke tests should point at a local, git-ignored ACL file through +`A3S_CONFIG_FILE`. Do not copy provider values into commands, logs, docs, pull +requests, or committed fixtures. + +```bash +A3S_CONFIG_FILE=/path/to/local/config.acl \ + scripts/real_config_env_integration.sh +``` + +SDK parity is covered by a separate real-provider check: + +```bash +A3S_CONFIG_FILE=/path/to/local/config.acl \ + scripts/sdk_real_config_env_integration.sh +``` + +These two wrappers deliberately rewrite credentials only inside a +`providers "openai"` block. For a native OpenAI-compatible provider name such +as `providers "deepseek"`, load the ACL directly through the provider-specific +runner instead of using the wrappers: + +```bash +A3S_CONFIG_FILE=/absolute/path/to/.a3s/config.acl \ + cargo test -p a3s-code-core --test test_deepseek_adversarial_e2e -- \ + --ignored --test-threads=1 --nocapture +``` + +The complete Node.js, Python, and Go DeepSeek retrieval matrix uses +`A3S_REAL_EVAL_ROOT=/absolute/path/to/a3s`; see the +[cross-SDK evaluation contract](https://github.com/A3S-Lab/Code/blob/main/sdk/evaluation/README.md#real-deepseek-matrix) +for platform-specific commands and gates. + +Keep secret-bearing raw provider evidence with the release or CI artifact, not +in public documentation. Aggregate redacted metrics may be published when they +contain no endpoint, header, environment-variable name, credential, prompt, or +source text. For the complete local API surface, see +[API Contract](/guide/api-contract). diff --git a/website/docs/v8.5.1/en/guide/rfcs/_meta.json b/website/docs/v8.5.1/en/guide/rfcs/_meta.json new file mode 100644 index 00000000..04371f2a --- /dev/null +++ b/website/docs/v8.5.1/en/guide/rfcs/_meta.json @@ -0,0 +1 @@ +["workspace-remote-git"] diff --git a/website/docs/v8.5.1/en/guide/rfcs/workspace-remote-git.mdx b/website/docs/v8.5.1/en/guide/rfcs/workspace-remote-git.mdx new file mode 100644 index 00000000..da264a21 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/rfcs/workspace-remote-git.mdx @@ -0,0 +1,537 @@ +--- +title: 'RFC: Remote Workspace Git Backend' +description: 'HTTP/JSON protocol for serving git operations to non-local workspace backends' +--- + +# RFC: Remote Workspace Git Backend + +| Field | Value | +| ------- | -------------------------------------------------------------- | +| Status | **Implemented** in v2.6.x (`core/src/workspace/remote_git.rs`) | +| Owner | `crates/code` workspace subsystem | +| Tracks | Phase 4.1 of S3 / non-local workspace hardening | +| Related | `S3WorkspaceBackend`, `WorkspaceGit*` trait family | + +This document is retained as the protocol specification — clients and +servers speaking this surface should match the wire shapes defined +below. Subsequent revisions should land as separate amendment documents +rather than editing this one in place, so the "what was shipped" history +stays auditable. + +This RFC proposes a protocol and a Rust client that let `a3s-code` run the +built-in `git` tool against workspaces backed by something other than a local +filesystem — most importantly S3 and future container / DFS backends, which +cannot host a `.git` directory directly. + +## 1. Motivation + +The workspace abstraction already lets the built-in tools talk to any +filesystem implementation via `WorkspaceFileSystem`. For `bash`, `grep`, +`glob`, and `git` we use **capability gating**: if the backend cannot +service the operation, the corresponding tool is not registered and the +model never sees it. + +That works for `bash` (object storage really cannot run a shell) and is +acceptable for `grep` (we added a degraded `LIST + GET + regex` path). It is +**painful for `git`**: many a3s-code workflows expect to query branch / +commit state, diff working tree, create branches, and stash changes. Hiding +`git` everywhere except local sessions splits the user experience and +prevents cloud workspaces from being first-class. + +The proposal is to introduce a **remote `WorkspaceGit` backend** that +delegates these operations over the network to an external service the host +operates. The service owns a real working tree (or a libgit2-backed +implementation) and exposes a small HTTP API. The a3s-code client speaks +that API and presents it to tools as just another `WorkspaceGit` provider. + +```text + ┌──────────────────┐ + │ a3s-code │ + model │ │ + ──────► │ git tool │ HTTP/JSON + │ │ │ ┌────────────────┐ + │ ▼ │ │ │ + │ WorkspaceGit ───┼──►│ gitserver │ + │ (RemoteGit…) │ │ (libgit2 / sh)│ + │ │ │ │ + │ WorkspaceFs ────┼──►│ (S3, etc.) │ + └──────────────────┘ └────────────────┘ +``` + +## 2. Non-Goals + +- **Hosting a gitserver.** This RFC defines a client and a protocol the + client speaks. Implementations of the server side (libgit2 wrapper, + shell-out, Gitea-API adapter, ...) are out of scope. +- **Replacing local git.** When `WorkspaceFs` is `LocalWorkspaceBackend`, + the existing `LocalWorkspaceBackend` `WorkspaceGit` impl remains the + default. Hosts mix and match per session. +- **Push / pull.** The `WorkspaceGit` trait surface is read-only against + remotes (`list_remotes` returns the configured set; no `git push` / + `git fetch`). A future RFC may add those if a tool needs them. +- **Worktrees.** Worktrees are a local-filesystem concept; the remote + backend will **not** implement `WorkspaceGitWorktreeProvider`. See §8. + +## 3. Protocol Choice — HTTP/JSON + +**Decision:** HTTP/JSON, not gRPC. + +| Criterion | HTTP/JSON | gRPC | +| ---------------------- | ------------------------------------- | ------------------------------ | +| Existing deps | `reqwest` already in tree | `tonic` + proto toolchain new | +| Schema strictness | manual `serde` types | `.proto` typed | +| Mocking in tests | trivial (`wiremock`) | needs gRPC mock infra | +| Operator debuggability | `curl` works | `grpcurl` (less ubiquitous) | +| Streaming | chunked / SSE | native bidi streams | +| Operation count | ~12 ops, all req/resp | small; streaming rarely needed | +| Server impl freedom | trivial wrapper around `git`, `gitea` | requires a gRPC server | + +The operation set is small (~12 RPCs) and the request/response shapes are +flat. Streaming would only meaningfully help `log` and `diff`, both of +which are bounded by client-side limits already. The mocking and operator +debuggability wins of HTTP/JSON outweigh the schema benefits of gRPC at +this scale. If the operation surface grows substantially or streaming +becomes load-bearing, gRPC can be revisited. + +We use **`POST` for every operation**. Git operations are imperative, not +resource-CRUD, and request bodies make versioning trivial (add optional +fields, ignore unknown). `GET` would force everything into query strings. + +## 4. Repository Identity + +The client knows it is talking about a specific repository; the server +needs to route. We encode the repo identity in the **URL path**: + +``` +POST /v1/repos/{repo_id}/git/ +``` + +`repo_id` is an opaque, URL-safe string negotiated out of band between the +host that configures the client and the gitserver operator. Typical values: + +- `users/{user_id}/sessions/{session_id}` (1:1 with an S3 workspace prefix) +- A UUID +- A path to a working tree on the gitserver + +The client treats `repo_id` as opaque; the server is responsible for +mapping it to an actual working tree / bare repo. + +## 5. Endpoint Reference + +All endpoints are `POST`. All requests and responses are `application/json`. +Field naming is `snake_case` to match Rust serde defaults. + +### 5.1 Status — `WorkspaceGit::status` + +``` +POST /v1/repos/{repo_id}/git/status +→ 200 { + "branch": "main", + "commit": "abc123...", + "is_worktree": false, + "is_dirty": true, + "dirty_count": 3 + } +→ 404 {"error":{"code":"REPO_NOT_FOUND", ...}} +``` + +### 5.2 Log — `WorkspaceGit::log` + +``` +POST /v1/repos/{repo_id}/git/log +{"max_count": 10} +→ 200 { + "commits": [ + {"id":"abc...", "message":"feat: ...", "author":"Alice ", "date":"2026-05-19T..."} + ] + } +``` + +### 5.3 List Branches — `WorkspaceGit::list_branches` + +``` +POST /v1/repos/{repo_id}/git/branches +→ 200 {"branches":[{"name":"main", "is_current":true}, ...]} +``` + +### 5.4 Create Branch — `WorkspaceGit::create_branch` + +``` +POST /v1/repos/{repo_id}/git/branches/create +{"name":"feat/x", "base":"main"} +→ 201 {} +→ 409 {"error":{"code":"BRANCH_EXISTS", ...}} +→ 404 {"error":{"code":"BASE_NOT_FOUND", ...}} +``` + +### 5.5 Checkout — `WorkspaceGit::checkout` + +``` +POST /v1/repos/{repo_id}/git/checkout +{"refspec":"feat/x", "force":false} +→ 200 {"stdout":"Switched to branch 'feat/x'"} +→ 409 {"error":{"code":"WORKING_TREE_DIRTY", ...}} +``` + +### 5.6 Diff — `WorkspaceGit::diff` + +``` +POST /v1/repos/{repo_id}/git/diff +{"target": null} // null = working tree vs index +{"target": "main"} // diff against ref +→ 200 {"diff":"", "truncated": false} +``` + +`truncated` is `true` when the server clipped the body — see §9. The +client surfaces this in the `git diff` tool result. + +### 5.7 List Remotes — `WorkspaceGit::list_remotes` + +``` +POST /v1/repos/{repo_id}/git/remotes +→ 200 {"remotes":[{"name":"origin", "url":"git@github.com:...", "direction":"fetch"}]} +``` + +### 5.8 Is Repository — `WorkspaceGit::is_repository` + +``` +POST /v1/repos/{repo_id}/git/exists +→ 200 {"is_repository": true} +``` + +Distinct from "repo not found" (404): the server may be configured to +serve repo IDs that map to non-git directories; `is_repository` lets the +client probe without inferring intent from a 404. + +### 5.9 List Stashes — `WorkspaceGitStashProvider::list_stashes` + +``` +POST /v1/repos/{repo_id}/git/stashes +→ 200 {"stashes":[{"index":0, "message":"WIP on main: ..."}]} +``` + +### 5.10 Stash — `WorkspaceGitStashProvider::stash` + +``` +POST /v1/repos/{repo_id}/git/stashes/create +{"message":"wip", "include_untracked":true} +→ 201 {} +→ 409 {"error":{"code":"NOTHING_TO_STASH", ...}} +``` + +## 6. Authentication + +The client supports two transport modes, configurable per-session: + +1. **Bearer token (default).** `Authorization: Bearer `. Token + provisioning is the host's responsibility (e.g. short-lived JWT minted + by the same identity layer that gates S3 access). +2. **mTLS.** Pass both `client_cert_pem` and `client_key_pem` paths via + the backend config. The client reads both files at construction, + concatenates them, and hands the result to `reqwest::Identity::from_pem`. + The `rustls-tls` backend expects the key in PKCS#8 PEM format. + Setting only one of the pair fails at construction with a clear error. + +Both are mutually compatible — a deployment that uses both for defence in +depth simply sets both. + +No-auth mode (for localhost development) is supported by configuring an +empty bearer token (and no mTLS); the client emits a `tracing::warn!` +on construction to make this visible. + +## 7. Error Model + +HTTP status code is the **transport** signal. The error **kind** is in +the JSON body: + +```json +{ + "error": { + "code": "BRANCH_EXISTS", + "message": "branch 'feat/x' already exists" + } +} +``` + +Client mapping: + +| HTTP | Default outcome | +| ------- | ------------------------------------------- | +| 200/201 | `Ok(...)` | +| 400 | `Err(anyhow!("bad request: {message}"))` | +| 401/403 | `Err(anyhow!("auth failed: {message}"))` | +| 404 | `Err(anyhow!("not found: {message}"))` | +| 409 | typed conflict — see below | +| 5xx | `Err(anyhow!("gitserver internal: {...}"))` | + +The client introduces one typed error for backwards-compatible recovery: + +```rust +#[derive(Debug, Clone, thiserror::Error)] +#[error("remote git conflict: {code}: {message}")] +pub struct RemoteGitConflict { + pub code: String, + pub message: String, +} +``` + +Tools that want to recover from `BRANCH_EXISTS` / `WORKING_TREE_DIRTY` / +`NOTHING_TO_STASH` downcast `anyhow::Error::downcast_ref::()`, +the same pattern `edit` / `patch` use for `WorkspaceVersionConflict` on +S3 compare-and-swap writes. + +Error codes documented in the public schema (extensible): + +| Code | Origin | +| -------------------- | ---------------------------------------------------- | +| `REPO_NOT_FOUND` | `repo_id` not served | +| `NOT_A_REPOSITORY` | path exists but is not a git repo | +| `BRANCH_EXISTS` | `create_branch` with existing name | +| `BRANCH_NOT_FOUND` | checkout / diff target missing | +| `BASE_NOT_FOUND` | `create_branch` base ref missing | +| `WORKING_TREE_DIRTY` | `checkout` would lose changes (and `force` is false) | +| `NOTHING_TO_STASH` | stash on a clean tree | +| `RATE_LIMITED` | per-tenant throttle hit (server-defined) | + +## 8. Optional Traits + +`WorkspaceGit` is implemented in full. +`WorkspaceGitStashProvider` is implemented. +`WorkspaceGitWorktreeProvider` is **deliberately not implemented** by the +remote backend. Worktrees are a local-filesystem concept that map poorly +onto a remote service: + +- "Create a worktree at path X" — the client has no path concept; the + server's path layout is opaque. +- The local tool flow that uses worktrees (parallel agent runs against + isolated copies) is better served on cloud workspaces by spinning up + **separate sessions with separate `repo_id`s**, not by emulating a + filesystem feature. + +Tools that depend on `WorkspaceGitWorktreeProvider` see `None` from +`services.git_worktree()` and emit a clear "worktrees unavailable on +remote git workspaces" error. + +## 9. Size and Cost Bounds + +Following the S3 backend's pattern, the remote git client enforces +client-side ceilings so the model cannot trigger unbounded responses: + +| Setting | Default | Applies to | +| ------------------------ | ------- | ---------------------------------------------------- | +| `max_diff_bytes` | 1 MiB | `diff` response body | +| `max_log_entries` | 200 | `log` `max_count` cap | +| `request_timeout` | 30 s | every HTTP call | +| `operation_timeout` (WS) | 60 s | the `WorkspaceServices`-level wrapper applies on top | + +The server is expected to honour these too — the client passes the relevant +caps in the request where applicable (`max_log_entries`), and surfaces +`truncated: bool` returned by the server on `diff` so the tool can hint +"large diff truncated; narrow your target". + +## 10. Rust Client Design + +```rust +// core/src/workspace/remote_git.rs (feature = "remote-git") + +#[derive(Debug, Clone)] +pub struct RemoteGitBackendConfig { + pub base_url: String, // https://git.example.invalid + pub repo_id: String, // path-segment safe + pub bearer_token: Option, + pub client_cert_pem: Option, // mTLS + pub client_key_pem: Option, + pub request_timeout: Option, // default 30s + pub max_diff_bytes: Option, // default 1 MiB + pub max_log_entries: Option, // default 200 +} + +#[derive(Debug, Clone)] +pub struct RemoteGitBackend { + http: reqwest::Client, + base_url: String, + repo_id: String, + max_diff_bytes: u64, + max_log_entries: usize, +} + +#[async_trait] +impl WorkspaceGit for RemoteGitBackend { /* see §5 */ } + +#[async_trait] +impl WorkspaceGitStashProvider for RemoteGitBackend { /* see §5 */ } +``` + +Composition factory mirrors the S3 path: + +```rust +impl WorkspaceServices { + /// Attach a remote git provider to an existing filesystem backend. + pub fn with_remote_git( + self: Arc, + cfg: RemoteGitBackendConfig, + ) -> Arc { ... } +} +``` + +Or as a top-level convenience for an S3 + remote-git workspace: + +```rust +pub fn s3_with_remote_git( + s3: S3BackendConfig, + git: RemoteGitBackendConfig, +) -> Arc { ... } +``` + +Wiring follows the existing builder pattern: + +```rust +let backend = Arc::new(RemoteGitBackend::new(cfg)); +let git: Arc = backend.clone(); +let stash: Arc = backend; + +WorkspaceServices::builder(workspace_ref, fs) + .file_system_ext(fs_ext) // S3 ETag CAS + .git(git) + .git_stash(stash) + // no git_worktree — see §8 + .operation_timeout(Duration::from_secs(60)) + .build() +``` + +Capability gating then registers the `git` tool automatically. + +## 11. Per-Call Observability + +Every HTTP call emits a `tracing::debug!` event with the same field +shape `S3WorkspaceBackend` uses for its per-call metering (see +`emit_s3_call_event` in `core/src/workspace/s3.rs`): + +| Field | Example | +| ------------- | ----------------------------- | +| `op` | `git.status`, `git.diff`, ... | +| `repo_id` | `sessions/example` | +| `outcome` | `ok` \| `error` | +| `status` | HTTP status code | +| `bytes` | response body length | +| `duration_ms` | wall-clock | + +Hosts that already meter S3 cost can extend the same subscriber to meter +gitserver cost; no new dependency or hook surface is needed. + +## 12. Composition Examples + +### S3 workspace + remote git + +```rust +let ws = WorkspaceServices::s3_with_remote_git( + S3BackendConfig::new("workspace", "sessions/example", access_key, secret_key) + .endpoint("https://s3.example.invalid") + .force_path_style(true) + .enable_search(true), + RemoteGitBackendConfig::new( + "https://git.example.invalid", + "sessions/example", + ) + .bearer_token(token), +); + +let session = agent + .session_builder("s3://workspace/sessions/example") + .options(SessionOptions::new().with_workspace_backend(ws)) + .build() + .await?; +``` + +Tools registered for this session: `read`, `write`, `edit`, `patch`, `ls`, +`grep`, `glob`, `git`. (`bash` remains hidden — object storage cannot +host a shell.) + +### Local filesystem + remote git (mixed) + +Useful when CI runs against a local checkout but the host wants to route +git operations through a sandboxed service (e.g. for audit / rate limiting). + +```rust +let local = WorkspaceServices::local("/workspaces/repo"); +let ws = local.with_remote_git(remote_cfg); // overrides local git provider +``` + +## 13. Open Questions + +These should be resolved before Phase 4.2 (implementation) starts. + +1. **Diff format pinning.** The local backend returns raw `git diff` + stdout. Should the remote API guarantee a specific diff dialect (POSIX + `diff -u`? libgit2's variant?) or pass through whatever the server + produces? Recommendation: pass through, document that the server + should produce unified diff. + +2. **Concurrent operations.** If two clients hit the same `repo_id` + simultaneously (one running `checkout`, one running `diff`), is the + server expected to serialise? Recommendation: server-side serialisation + per `repo_id`. Document the expectation; do not retry on conflict in + the client. + +3. **Long-running operations.** `checkout` against a large tree could + exceed `request_timeout`. Should the protocol support a polling / + async-job mode? Recommendation: defer. Set realistic timeouts; the + working-tree sizes we target (per-session sandboxes) are small. + +4. **Hooks.** The server may have pre-commit / pre-push hooks installed. + Do we surface a "hook output" channel in responses? Recommendation: + include hook stderr in the `checkout` / `stash` response when + non-empty; document that hook failures map to HTTP 422 with a + `HOOK_FAILED` code. + +5. **Schema versioning.** First-cut endpoints are under `/v1/`. When + should we bump to `/v2/`? Recommendation: only on incompatible + request/response shape changes; additive fields stay under `/v1/` + (clients ignore unknown fields). + +6. **Reference RFC.** Should a reference implementation (e.g. a minimal + libgit2-backed gitserver written in Rust) live in this repo, or stay + out-of-tree? Recommendation: out-of-tree. The client and protocol are + sufficient; reference servers belong with operators. + +## 14. Out of Scope (Future RFCs) + +- **Push / fetch from upstreams.** Needed for "agent commits and pushes + changes back to origin" workflows. Will need credential delegation. +- **Sparse / partial checkout.** Useful for huge monorepos; the current + surface assumes the entire repo is materialised on the server. +- **Streaming `log` and `diff`.** Required if response bodies routinely + exceed `max_diff_bytes`. Would motivate revisiting the gRPC decision. +- **Hooks management.** Listing / configuring server-side hooks from the + client. + +## 15. Implementation Notes (shipped in v2.6.x) + +The original draft sketched an 8-step implementation plan; what +actually shipped (Phase 4.2 plus Phase 5.x follow-ups) is: + +1. `RemoteGitBackend` / `RemoteGitBackendConfig` / `RemoteGitConflict` + live in `core/src/workspace/remote_git.rs`. No + `remote-git` feature was added — `reqwest` was already a hard + dependency for the LLM clients, so the module compiles + unconditionally. +2. `WorkspaceGit` and `WorkspaceGitStashProvider` implemented in full. + `WorkspaceGitWorktreeProvider` deliberately omitted (see §8). +3. `WorkspaceServices::with_remote_git` is the public attachment + point. The internal helper `with_git_provider` (added in v2.6.x + follow-up) handles the actual struct-literal copy so future fields + on `WorkspaceServices` cannot be silently dropped by the decorator. +4. mTLS is supported in addition to bearer token (`client_cert_pem` + - `client_key_pem`), shipped in a Phase 5.2 follow-up that also + updated §6 above. +5. `diff` is hardened against unbounded responses: HTTP body is + streamed with a hard cap of `max_diff_bytes * 4` (Phase 6.2). The + soft `max_diff_bytes` display truncation continues to apply + post-decode. +6. Test surface: 25+ wiremock-backed unit tests in `remote_git.rs`, + one end-to-end test driving the `git` built-in tool through the + remote backend, plus the workspace-wide conformance suite added + in Phase 6.3. +7. README and CHANGELOG updated; rustls-tls backend selected to match + the AWS SDK. +8. SDK exposure landed in Phase 5.1 (Node + Python). diff --git a/website/docs/v8.5.1/en/guide/security.mdx b/website/docs/v8.5.1/en/guide/security.mdx new file mode 100644 index 00000000..9d44a836 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/security.mdx @@ -0,0 +1,239 @@ +--- +title: 'Security' +description: 'Permission policy, HITL, hooks, and verification gates' +--- + +# Security + +A3S Code exposes security controls at session creation time and through +session lifecycle hooks. Treat direct host calls such as `session.tool()` and +`session.bash()` as privileged host operations: the application that calls them +is responsible for deciding whether to expose that power to a user. + +Delegated child runs intersect their local permissions with the parent +permission checker and inherit the parent process sandbox. A child can narrow +its capabilities but cannot replace the host boundary. Child-local approval +behavior applies only to an `Ask` introduced by the child policy; a parent +`Ask`, or a tool-owned escalation after both policies allow, remains under the +parent confirmation provider. Missing providers fail closed before a HITL +request is emitted. Keep high-risk release or publish commands behind explicit +policy. + +A permission prompt should explain the operation, reason, scope, and risk—not +just present two buttons. + +## Enforced Execution Boundary + +Permission policy decides whether an operation is allowed, denied, or needs +approval. It does not, by itself, confine a process. Every local A3S Code +session binds a concrete native `BashSandbox` before capability construction. +The same handle reaches direct tools, model-governed tools, workflows, Skills, +and delegated runs. Hosts use `SessionOptions::with_sandbox_handle` only to +replace this default with an equivalent boundary; non-local workspace backends +retain their explicit command-runner contract. + +Core and the Code TUI use the A3S-owned `a3s-sandbox` Rust library through its +`NativeBashSandbox` adapter. It selects the native boundary for the host +platform—Seatbelt on macOS, Bubblewrap namespaces plus seccomp on Linux, and +AppContainer plus a kill-on-close Job Object on Windows—and probes that +boundary before enabling Bash. No Node.js, npm package, or global runtime +installation is required or selected. + +Capability probing is necessary before readiness. macOS requires the system +`/usr/bin/sandbox-exec`; Linux requires `/usr/bin/bwrap` and permitted +unprivileged user namespaces; Windows uses PowerShell 7 from the system +Program Files directory and the native AppContainer APIs. The TUI runs a +bounded command through the actual OS boundary before enabling its deferred +handle. Failure marks that handle unavailable, so Default can request one exact +escalated host invocation and Auto denies Bash. Embedded Core sessions also +fail closed: initialization failure installs an error-only handle instead of +falling back to the local workspace runner. + +The adapter fails closed and never silently retries on the host. It denies +network egress, local binding, and Unix sockets; limits writes to the workspace +and a private scratch directory; protects `.git`, `.a3s`, `.agents`, `.codex`, +`.claude`, `.vscode`, `.idea`, and control files; masks common credential +stores; and scrubs ambient secrets. Delegated tasks, Skills, and workflow steps +retain the same handle. + +Because built-in file tools execute in-process, the TUI separately enables +Core's local workspace credential policy. It covers direct and range reads, +writes, edits, patches, and both manifest-backed and fallback grep. Explicit +sensitive paths fail closed, broad grep omits protected candidates, and +source-tree hardlink aliases are denied before mutation. Ordinary +package-store hardlinks remain usable unless they alias a discovered +credential inode. Read-only Git diff regenerates output only for allowed +changed paths, option-like revisions cannot become Git flags, and displayed +remotes omit embedded HTTP credentials and query tokens. + +The release gate exercises the native backend on macOS, Linux, and Windows. It +proves normal workspace writes and offline toolchain commands remain usable, +while outside and symlink writes, protected metadata mutations, credential +reads, network egress, local listeners, and Unix sockets remain blocked. + +The terminal execution modes apply this boundary as follows: + +| Mode | Process behavior | Approval behavior | +| ------- | ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| Default | Ordinary Bash runs in the installed sandbox. | Only an explicit host escape, missing sandbox, protected metadata mutation, mutating Git call, or annotated external side effect asks. | +| Plan | Bash and mutations are unavailable. | Boundary crossings are denied, not queued for later execution. | +| Auto | Ordinary Bash runs only in the installed sandbox. | HITL is disabled; any operation that cannot stay inside the boundary is denied. | + +## Threat Model + +The protected assets are host files outside the active workspace, repository +and agent-control metadata, credential stores, host network and listener +capabilities, process lifetime, and the user's authority to approve a specific +operation. Model output, repository content, command output, fetched content, +Skill instructions, and MCP results are untrusted inputs. A malicious +dependency or child process is assumed able to close output pipes early, fork +descendants, create symlinks, inspect its environment, and attempt filesystem, +socket, or network escapes. + +The trusted computing base is the running CLI/Core binary, the verified +release support tree, the exact resolved Node executable, the selected sandbox +provider and operating-system enforcement, and host code that invokes direct +SDK helpers. A configured MCP server or native integration is a separately +trusted extension; missing or unsafe behavior annotations cause confirmation +rather than granting read-only status. + +Trust does not waive process ownership. Each local stdio MCP server leads a +dedicated Unix process group. Closing or dropping its transport stops pipe +tasks, clears pending requests, reaps the leader, and terminates descendants. +The stderr pipe is drained independently so diagnostics cannot stall protocol +traffic. + +This local boundary does not claim kernel-level protection against a compromised +sandbox provider, Node runtime, CLI binary, or operating system. It also does +not turn the privileged host-direct SDK into end-user authorization. Use a +stronger workload provider for hostile multi-tenant code, kernel attack +resistance, or OCI-level isolation. + +## Invocation Ingress + +Every session-owned path enters one scoped invocation kernel: + +| Ingress | Effective origin | Trust handling | +| --------------------------------------------------------------------- | -------------------------------------------- | --------------------------------------------------------------------------------------------------------- | +| Model tool call | Agent | Mode policy, permission, tool-owned escalation, and HITL apply. | +| Model-owned `batch`, `program`, workflow, or public custom nesting | Governed nested | Re-enters the same policy for every child. | +| Direct host call to an ordinary custom tool | Host direct at the top level | The selected top-level call is trusted; public nested calls become governed nested calls. | +| Direct host call to built-in `batch`, `program`, or dynamic workflow | Host direct, then trusted host-direct nested | Only host-selected built-in children retain control-plane authority. | +| Skill, Task, worker, or other model sub-run created from host context | Agent in the child | Ambient host-direct authority is removed before model dispatch. | +| Delegated child permission or confirmation | Governed nested/Agent | Child policy intersects the parent; parent `Ask` and tool-owned escalation stay with the parent provider. | +| Tool behavior metadata | Escalation on the current origin | It can require confirmation but cannot weaken `Ask` or `Deny`. | + +Low-level `ToolRegistry` and standalone `ProgramExecutor` APIs are deliberately +ungoverned building blocks for hosts that own the registry. AgentSession and +the TUI do not use those APIs as a fallback. + +## Complete TUI Decision Matrix + +The cells below are the terminal outcomes for model and ordinary governed +nested invocations. “Deny” means no confirmation event is created. “Confirm +once” means one exact invocation ID is pending; cancelling or expiring it +cannot settle another prompt. + +| Mode | Sandbox | Ordinary Bash | Explicit host Bash | Protected metadata mutation | Tool-owned confirmation | Annotated external side effect | +| ------- | ----------- | --------------------------------------------------------- | --------------------------------------------------------- | --------------------------- | ----------------------- | ------------------------------ | +| Default | Ready | Execute in sandbox | Confirm exact call once; execute on host only if approved | Confirm exact mutation once | Confirm once | Confirm once | +| Default | Unavailable | Confirm exact call once; execute on host only if approved | Confirm exact call once; execute on host only if approved | Confirm exact mutation once | Confirm once | Confirm once | +| Plan | Ready | Deny | Deny | Deny | Deny | Deny | +| Plan | Unavailable | Deny | Deny | Deny | Deny | Deny | +| Auto | Ready | Execute in sandbox | Deny | Deny | Deny | Deny | +| Auto | Unavailable | Deny | Deny | Deny | Deny | Deny | + +Origin modifies that matrix only as follows: + +| Origin/call shape | Matrix relationship | +| ---------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| Agent or ordinary governed nested | Uses the matrix exactly. | +| Direct host call to a custom tool | The selected top-level call skips model-facing permission/HITL; every public nested call returns to the matrix. | +| Trusted built-in host-direct nested | Skips model-facing permission/HITL because the host supplied the built-in orchestration and its children. Hooks, budget, timeout, cancellation, recursion checks, sandbox selection, and sanitization still apply. | +| Model sub-run born inside any host-direct call | Returns to the Agent row and uses the matrix exactly. | + +Decision precedence is fail-closed: active-Skill restrictions and hard +guardrails first, then hook blocks, origin authority, mode/permission policy, +tool-owned escalation, confirmation availability, and finally execution. +Terminal policies always use `TimeoutAction::Reject`; Core's generic +`AutoApprove` option is not accepted by the TUI. + +Treat the native sandbox as a local enforcement provider, not as the stack-wide +execution contract. A3S Code owns agent policy and its `BashSandbox` interface. +A3S Runtime owns durable, provider-neutral Task and Service lifecycle and +placement. A3S Box owns OCI and stronger isolation. A3S Observer and A3S Sentry +can add execution evidence and adaptive runtime enforcement as defense in +depth; they do not replace the deterministic per-process sandbox boundary. + +## Secure Downloads + +`download` is a bounded workspace mutation, not an unrestricted network or +filesystem primitive. It is registered only when the session has a writable +local workspace. Model-selected calls pass through the same permission policy, +HITL confirmation, hooks, timeout, cancellation, and workspace checks as other +mutations. Direct `session.tool('download', ...)` calls remain privileged host +operations and require authorization in the embedding application. + +Its network boundary accepts only HTTP(S), rejects user information and +non-public targets, and revalidates every bounded redirect hop. Direct +connections reject DNS answers containing any private or otherwise reserved +address and pin the validated public addresses for that hop. Cross-origin +redirects drop credentials and resource validators. Explicit proxy mode leaves +hostname resolution to the configured proxy but retains literal-host and +redirect checks. + +Signed query parameters are preserved because object stores and release systems +need them to authorize the request. They are removed from diagnostics and +`source_anchors`, so successful and failed tool results do not expose those +secrets in metadata. + +The destination must remain below the local workspace and may not cross +symlinks. Content is streamed to an adjacent temporary file under byte and time +limits. Strict Range validation, optional `expected_sha256`, sync-before-promote, +and atomic replacement keep incomplete or unverified data away from the final +path. Cancellation and failure remove the temporary file; `overwrite` defaults +to `false`. + +See [Tools](/guide/tools#binary-safe-local-downloads) for the complete parameter +contract. + +## Permission Policy + +```ts title=permissions.ts +const session = agent.session('/repo', { + // !callout(1:7) Applied before execution + permissionPolicy: { + deny: ['bash(rm -rf*)', 'write(**/.env*)'], + ask: ['bash(git push*)', 'bash(npm publish*)'], + allow: ['read(*)', 'search(*)', 'bash(cargo test*)'], + defaultDecision: 'ask', + enabled: true, + }, +}); +``` + +Avoid permissive defaults for release or production sessions. Make dangerous commands explicit and auditable. + +## Hooks + +Hooks are registered on a session with an event type, matcher, optional config, +and handler: + +```ts +session.registerHook( + 'observe-secret-read', + 'pre_tool_use', + { pathPattern: '**/.env*' }, + { priority: 100 }, + () => ({ action: 'continue' }), +); + +console.log(session.hookCount()); +session.unregisterHook('observe-secret-read'); +``` + +## Verification + +Verification reports and summaries are available from the session and selected +result fields. Release workflows should require tests, package checks, CI +checks, and provider evidence. diff --git a/website/docs/v8.5.1/en/guide/sessions.mdx b/website/docs/v8.5.1/en/guide/sessions.mdx new file mode 100644 index 00000000..c477063c --- /dev/null +++ b/website/docs/v8.5.1/en/guide/sessions.mdx @@ -0,0 +1,593 @@ +--- +title: 'Sessions' +description: 'Creating, streaming, resuming, and saving workspace-bound sessions' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Sessions + +`Agent` owns configuration and provider state. `Session` binds that agent to one workspace and one conversation lifecycle. + +This is where UI integration usually begins: subscribe to `AgentEvent` values +from the session and map them to progress, tool calls, confirmations, and results. + +```ts title=session.ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +// !focus(1:7) +const session = agent.session('/repo', { + model: 'provider/model-id', + planningMode: 'enabled', + goalTracking: true, + autoDelegation: { enabled: true, maxTasks: 4 }, + autoParallel: false, +}); +``` + +## Rust Construction + +Rust session construction is async-first because default memory, file-backed +stores, queues, trajectory recording, and MCP discovery can require I/O: + +```rust +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +`session_async`, `resume_session_async`, `session_for_agent_async`, and +`session_for_worker_async` are direct async alternatives. Node and Python keep +their existing factory names and delegate to this async construction kernel +inside the native binding. + +The synchronous Rust `Agent::session` factory is a strict compatibility path. +It accepts explicitly pre-initialized resources and never starts or blocks an +async runtime. A default/file memory store, file session store, queue, +trajectory recorder, or any host MCP manager supplied in `SessionOptions` +returns `CodeError::AsyncSessionBuildRequired`; session-option MCP capability +discovery is always async. Use the builder instead of catching that error and +silently changing backends. The sync path can only inherit agent-global MCP +tools already cached during agent initialization. + +`planningMode` is explicit: use `'auto'` for the default structured pre-analysis +path, `'enabled'` to force planning, and `'disabled'` to turn planning off for +latency-sensitive calls. The legacy boolean `planning` option remains available +for compatibility. + +Planning emits run-scoped state. Hosts can render that state as a TaskList and +update each item as the runtime records progress instead of trying to infer +progress from text tokens. + +## Single-Flight Operations + +A session admits only one transcript-affecting operation at a time. `send`, +`stream`, their attachment variants, slash commands, and `resumeRun` share the +same fail-fast admission gate. An overlapping call returns +`CodeError::SessionBusy` before reading history or dispatching a command; it is +not queued behind the active operation. + +Wait for the active result, consume the stream to completion, or cancel it +before starting another conversation operation. The runtime retains a stream's +lease until its producer has actually stopped, even if the public stream handle +is dropped or aborted. Direct host tool helpers do not mutate the transcript and +do not claim this lease. + +Node and Python stream iterators wait for that lifecycle cleanup at the terminal +boundary. Once a fully consumed iterator reports completion, an immediate next +conversation operation will not inherit a stale busy state from that stream. + +## Send + + + + +```rust +let result = session + .send( + "Review this repository and list release blockers", + None, + ) + .await?; + +println!("{}", result.text); +println!("{}", result.usage.total_tokens); +println!("{:?}", result.verification_summary().status); +``` + + + + +```ts +const result = await session.send( + 'Review this repository and list release blockers', +); + +console.log(result.text); +console.log(result.totalTokens); +console.log(result.verificationStatus); +``` + + + + +```python +result = session.send( + "Review this repository and list release blockers", +) + +print(result.text) +print(result.total_tokens) +print(result.verification_status) +``` + + + + +```go +result, err := session.Run( + ctx, + "Review this repository and list release blockers", +) +if err != nil { + return err +} + +fmt.Println(result.Text) +fmt.Println(result.Usage.TotalTokens) +fmt.Println(result.VerificationSummary.Status) +``` + + + + +## Stream + + + + +```rust +use a3s_code_core::{AgentEvent, CodeError}; + +let (mut events, lifecycle) = session + .stream("Run the focused tests and explain failures", None) + .await?; + +while let Some(event) = events.recv().await { + match event { + AgentEvent::TextDelta { text } => print!("{text}"), + AgentEvent::ToolStart { name, .. } => println!("\ntool: {name}"), + AgentEvent::End { .. } => break, + AgentEvent::Error { message } => return Err(CodeError::Llm(message)), + _ => {} + } +} +lifecycle + .await + .map_err(|error| CodeError::Internal(error.into()))??; +``` + + + + +```ts +const stream = await session.stream( + 'Run the focused tests and explain failures', +); + +while (true) { + const { value: event, done } = await stream.next(); + if (done) break; + if (!event) continue; + + if (event.text) process.stdout.write(event.text); + if (event.toolName) console.log('tool:', event.toolName); +} +``` + + + + +```python +for event in session.stream("Run the focused tests and explain failures"): + if event.type == "text_delta" and event.text: + print(event.text, end="", flush=True) + elif event.type == "tool_start": + print(f"\ntool: {event.tool_name or 'unknown'}") + elif event.type == "error": + raise RuntimeError(event.error or "stream error") +``` + + + + +```go +stream, err := session.Stream( + ctx, + "Run the focused tests and explain failures", + nil, +) +if err != nil { + return err +} + +for event := range stream.Events { + if event.Type != code.EventTextDelta { + continue + } + var payload struct { + Text string `json:"text"` + } + if err := event.DecodePayload(&payload); err != nil { + return err + } + fmt.Print(payload.Text) +} +if err := <-stream.Done; err != nil { + return err +} +``` + + + + +Every SDK event is an `EventEnvelopeV1` projection with `version === 1`, an open +`type` string, complete `payload`, and optional `metadata`. Convenience fields +such as `text` and `toolName` are derived from that envelope. Keep a default +branch and retain the payload for future event types. + +## Steer and Interrupt an Active Run + +Run control changes an operation already in progress; it does not start a +second transcript turn. `steer` queues a newer user direction for the next +safe point. `interrupt` cancels current provider and Tool work cooperatively, +then lets supervised cleanup settle before the Run becomes `cancelled`. + + + + +```rust +use a3s_code_core::{InterruptRequest, SteerRequest}; + +let state = session.run_control_snapshot().await; +let receipt = session + .steer(SteerRequest::new("Prioritize the failing test")) + .await?; +println!("{:?}", receipt.state); + +session + .interrupt(InterruptRequest::new().with_reason("User stopped the run")) + .await?; +``` + + + + +```ts +const state = await session.runControlSnapshot(); +const receipt = await session.steer('Prioritize the failing test', { + runId: state?.runId, + expectedTurnId: state?.turnId, + expectedTurnRevision: state?.turnRevision, +}); +console.log(receipt.state); + +await session.interrupt({ reason: 'User stopped the run' }); +``` + + + + +```python +state = await session.run_control_snapshot_async() +options = {} +if state: + options["run_id"] = state["run_id"] + if state.get("turn_id") is not None: + options["expected_turn_id"] = state["turn_id"] + options["expected_turn_revision"] = state["turn_revision"] +receipt = await session.steer_async( + "Prioritize the failing test", + options, +) +print(receipt["state"]) + +await session.interrupt_async({"reason": "User stopped the run"}) +``` + + + + +```go +state, err := session.RunControlSnapshot(ctx) +if err != nil { + return err +} +options := &code.SteerOptions{} +if state != nil { + options.RunID = &state.RunID + options.ExpectedTurnID = state.TurnID + options.ExpectedTurnRevision = &state.TurnRevision +} +receipt, err := session.Steer(ctx, "Prioritize the failing test", options) +if err != nil { + return err +} +fmt.Println(receipt.State) +_, err = session.Interrupt(ctx, &code.InterruptOptions{}) +``` + + + + +Requests are idempotent when a caller reuses the same request ID with the same +payload. Optional Run and expected-Turn fields make stale UI actions fail +closed. Accepted control is not yet applied control: observe +`run_control_applied` or inspect persisted Run events when the distinction +matters. Neither operation changes the selected model, permissions, sandbox, +budgets, or approval policy. + +## Side Questions + +The SDK does not expose a dedicated ephemeral-question helper. Snapshot the +current history and pass it explicitly to `send` or `stream`. Explicit history +is used for that call only; it does not append the answer back into the session +history. + +```ts +const snapshot = session.history(); +const answer = await session.send( + 'What files has this session already inspected?', + snapshot, +); + +console.log(answer.text); +console.log(session.history().length === snapshot.length); +``` + +## Resume + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +let session = agent + .resume_session_async( + "release-review", + SessionOptions::new().with_file_session_store("./.a3s/sessions"), + ) + .await?; +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.resumeSession('release-review', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + + + + +```python +from a3s_code import FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.resume_session("release-review", opts) +``` + + + + +```go +options := &code.SessionOptions{ + FileSessionStoreDir: ".a3s/sessions", +} +session, err := agent.ResumeSession(ctx, "release-review", options) +``` + + + + +Set `autoSave: true` or call `await session.save()` when using a session store. +This resumes a saved session snapshot; interrupted run checkpoints use +`session.resumeRun(runId)` instead. Go uses `Save`, `ResumeSession`, and +`ResumeRun(ctx, runID)` for the same two persistence levels. +See [Persistence](/guide/persistence). + +## Lifecycle and Close + +`session.close()` is a full graceful stop. The first call flips the session +into the **closed** state — further `send`/`stream` calls fast-fail with +`CodeError::SessionClosed` instead of starting a new run — then cancels the +active run, every in-flight delegated subagent task, and all pending +human-in-the-loop confirmations. Subsequent calls are no-ops and never panic. +Check the closed state with `session.isClosed()` (Node) / `session.is_closed()` +(Python) / `session.IsClosed(ctx)` (Go). + +```ts +session.close(); +if (session.isClosed()) { + // send/stream now reject with CodeError::SessionClosed +} +``` + +```python +session.close() +if session.is_closed(): + # send/stream now reject with CodeError::SessionClosed + pass +``` + +```go +if err := session.Close(ctx); err != nil { + return err +} +closed, err := session.IsClosed(ctx) +``` + +### Cancellation token + +Every run derives its per-operation cancellation token from a single +session-level parent via `child_token()`, so `close()` cascades to all +in-flight work in one shot. Embedders that need the raw token — for example to +wire it into a host-side `select!` or to abort the session without the +run-store and hook side effects of `close()` — can clone it through +`AgentSession::session_cancel_token()`. + +### Agent-side registry + +The owning `Agent` tracks its live sessions by `Weak` reference (pruned +lazily) so a control plane can drive lifecycle without holding a session +handle: + +- `Agent::list_sessions()` returns the live session IDs (sorted, stable). +- `Agent::close_session(id)` closes one session by ID — the same cleanup as + `AgentSession::close()`, invoked out-of-band. +- `Agent::close()` closes every live session and tears down agent-owned + background resources (it also disconnects the global MCP connections). + After it returns, new `session` / `resumeSession` calls fail fast with + `CodeError::SessionClosed`. +- `Agent::is_closed()` reports whether the agent itself has been closed. + +```ts +const ids = await agent.listSessions(); +await agent.closeSession(ids[0]); +await agent.close(); // closes all remaining sessions + global MCP +console.log(agent.isClosed()); +``` + +```python +ids = agent.list_sessions() +agent.close_session(ids[0]) +agent.close() # closes all remaining sessions + global MCP +print(agent.is_closed()) +``` + +```go +ids, err := agent.ListSessions(ctx) +if err == nil && len(ids) > 0 { + _, err = agent.CloseSession(ctx, ids[0]) +} +err = agent.Close(ctx) +``` + +See CHANGELOG `[3.3.0]` — "Session / Agent lifecycle control". + +## Host Identity Labels + +`SessionOptions` carries four opaque identity fields that the host can attach +at session creation. The framework only transports them — it never interprets +or enforces them. They are propagated into `SessionData`, hooks, and traces, +and restored on resume, so the host can drive multi-tenant aggregation, +billing, and distributed tracing: + +| Node (camelCase) | Python (snake_case) | Go | Meaning | +| ----------------- | ------------------- | ----------------- | ------------------------------------------------------------- | +| `tenantId` | `tenant_id` | `TenantID` | Multi-tenant label | +| `principal` | `principal` | `Principal` | User / service that triggered the session | +| `agentTemplateId` | `agent_template_id` | `AgentTemplateID` | Agent template / definition the session was instantiated from | +| `correlationId` | `correlation_id` | `CorrelationID` | Distributed-trace correlation id | + +```ts +const session = agent.session('/repo', { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', +}); +``` + +```python +opts = SessionOptions() +opts.tenant_id = 'tenant-example' +opts.principal = 'principal-example' +opts.agent_template_id = 'agent-template-example' +opts.correlation_id = 'trace-example' +session = agent.session('/repo', opts) +print(session.tenant_id, session.principal) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + TenantID: "tenant-example", + Principal: "principal-example", + AgentTemplateID: "agent-template-example", + CorrelationID: "trace-example", +}) +``` + +See CHANGELOG `[3.3.0]` — "Host-provided identity labels". + +## Run Replay + +Each `send()` or `stream()` creates a run record. Applications can inspect those +records for UI state, audit, replay, cancellation, and tests: + +```ts +const runs = await session.runs(); +const latest = runs.at(-1); + +if (latest) { + console.log(await session.runSnapshot(latest.id)); + console.log(await session.runEvents(latest.id)); +} +``` + +```go +runs, err := session.Runs(ctx) +if err == nil && len(runs) > 0 { + latest := runs[len(runs)-1] + snapshot, snapshotErr := session.RunSnapshot(ctx, latest.ID) + events, eventsErr := session.RunEvents(ctx, latest.ID) + _, _, _ = snapshot, snapshotErr, eventsErr + _ = events +} +``` + +`currentRun()` is for the operation that is current at the time of the call. +During an active `send()` or `stream()`, pass its `id` to `cancelRun(id)` to +request cancellation. When idle, `currentRun()` may return `null` or a retained run +snapshot depending on the preceding control flow, so use `runs()` for completed +history and inspect `status` before treating a snapshot as cancellable: + +```ts +const current = await session.currentRun(); +if (current?.id && current.status === 'running') { + await session.cancelRun(current.id); +} +``` + +## Agent Definitions + +`sessionForAgent()` applies a named agent definition from built-in agents, `.a3s/agents`, or configured `agentDirs`. + +```ts +const session = agent.sessionForAgent('/repo', 'explore', ['./agents'], { + planningMode: 'auto', +}); +``` + +```go +session, err := agent.SessionForAgent( + ctx, + "/repo", + "explore", + []string{"./agents"}, + &code.SessionOptions{PlanningMode: code.PlanningAuto}, +) +``` + +For disposable value-defined workers, use `SessionForWorker`; both methods +return the regular Go `Session` API. diff --git a/website/docs/v8.5.1/en/guide/skills.mdx b/website/docs/v8.5.1/en/guide/skills.mdx new file mode 100644 index 00000000..82cc7ee0 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/skills.mdx @@ -0,0 +1,82 @@ +--- +title: 'Skills' +description: 'Prompt-time skills, inline skills, and the Skill tool' +--- + +# Skills + +Skills are reusable instructions that the harness can discover and rank. +File-backed skill directories, inline skills, explicit registries, and +`search_skills` share the same discovery path. A3S Code no longer ships default +embedded skills; `builtinSkills: true` is accepted by the SDKs as a compatibility +no-op. + +## Compatibility Flag + +```ts +const session = agent.session('/repo', { + builtinSkills: true, +}); +``` + +This flag does not add default skills. Put reusable behavior in project skill +files, inline skills, or agent definitions. + +## Skill Files + +Store Markdown files with frontmatter in a skill directory: + +```md +--- +name: release-review +description: Review release blockers and verification evidence +allowed-tools: 'read(*), search(*), bash(cargo test*)' +--- + +Check package metadata, changelog, release scripts, and CI status. +Return blockers first. +``` + +```ts +const session = agent.session('/repo', { + skillDirs: ['/repo/.a3s/skills'], +}); +``` + +Use `skillDirs` for skill files. Use `agentDirs` for worker/subagent +definitions. + +## Inline Skills + +```ts +const session = agent.session('/repo', { + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: 'Always separate blockers from nice-to-have improvements.', + }, + ], +}); +``` + +## Skill Tool + +`search_skills` finds relevant skills: + +```ts +await session.tool('search_skills', { + query: 'release blockers', + limit: 5, +}); +``` + +Markdown skill files with `allowed-tools` frontmatter and inline skills are both +discoverable. Skill administration is an SDK or filesystem concern rather than +a model-visible management tool. + +When a skill is invoked through the `Skill` tool, `allowed-tools` is +fail-secure: omitted frontmatter grants no tools to that skill invocation. +Declare the smallest needed allow-list. Ordinary session tool calls are not +restricted by active skills unless `enforceActiveSkillToolRestrictions` is +explicitly enabled for legacy behavior. diff --git a/website/docs/v8.5.1/en/guide/tasks.mdx b/website/docs/v8.5.1/en/guide/tasks.mdx new file mode 100644 index 00000000..760faec9 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/tasks.mdx @@ -0,0 +1,352 @@ +--- +title: 'Tasks' +description: 'Manual and automatic delegation with the unified task tool and subagents' +--- + +# Tasks + +The routine multi-agent path is one model-visible `task` tool. Its `tasks` array +accepts one focused child or several independent children for concurrent fan-out. +Child context stays isolated from the parent conversation, and only compact +results return instead of full transcripts. + +The same delegation core also powers automatic subagent delegation. Enable it +when the runtime should proactively start specialist child agents for +high-confidence work, and disable only automatic parallel fan-out with +`autoParallel: false` when you want serial automatic delegation. + +Web clients can present these states as a plan and a separate list of child-agent +runs, keyed by task ID. + +## Built-in Subagents + +| Agent | Use it for | +| ----------------------------- | --------------------------------------------------------------------------------------- | +| `explore` | Read-only codebase search, file inspection, and structure discovery. | +| `plan` | Read-only implementation plans and architecture analysis. | +| `general` / `general-purpose` | Multi-step implementation work with read/write/command access. | +| `verification` | Focused checks, reproductions, regression validation, and adversarial testing. | +| `review` | Findings-first code review for correctness, regressions, security, and maintainability. | + +You can mention them explicitly, for example `@review`, `@agent-plan`, `use the verification subagent`, or `delegate to general-purpose`. + +## Manual Delegation + +Ask the parent agent to delegate a bounded job: + +```text +Use task to ask an explore agent to inspect the auth module. +Return files inspected, findings, risks, and confidence. +``` + +If the host already knows the task boundary, call the same core tool directly: + +```ts +const task = await session.task({ + agent: 'explore', + description: 'Inspect auth module', + prompt: 'Return files inspected, findings, risks, and confidence.', +}); +if (task.exitCode !== 0) throw new Error(task.output); +console.log(task.output); +``` + +A child agent should return a compact contract: + +- summary +- files inspected or changed +- evidence references +- risks and unknowns +- confidence + +The parent should not ingest the full child transcript. + +## Parallel Delegation + +Use `task` with several `tasks` items, or `session.tasks(...)`, when independent +work can run concurrently: + +```text +Run one task call with three independent tasks: +1. inspect provider config parsing +2. inspect Node SDK declarations +3. inspect release scripts + +Merge the results into one release-readiness report. +``` + +```ts +const batch = await session.tasks([ + { + agent: 'explore', + description: 'Inspect config', + prompt: 'Check provider parsing.', + }, + { + agent: 'verification', + description: 'Verify SDK', + prompt: 'Check SDK declarations.', + }, +]); +if (batch.exitCode !== 0) throw new Error(batch.output); +console.log(batch.output); +``` + +The unified `task` call accepts 1-32 items. A single item may request +`background`; a multi-item call collects every branch and therefore rejects +`background: true`. By default every branch must succeed. Set +`allow_partial_failure` only for evidence-gathering work that can use incomplete +results; `min_success_count` is available only in that mode and must not exceed +the submitted task count. + +`session.task(...)` and `session.tasks(...)` return `ToolResult` values from the +same `task` tool. Read `output` for the compact child summary and check +`exitCode` before treating the result as successful. `maxParallelTasks` in +session options and `max_parallel_tasks` in ACL bound sibling fan-out. + +Prefer `task` / `session.tasks` for all fan-out. Model-visible and SDK +`parallel_task` / `parallelTask` helpers are **removed** (`HARNESS-CONV4`). Use +multi-item `task` only. + + + +## Agent-Wide Priority Scheduler + +Every `Agent` owns one scheduler shared by all sessions created from it. The +scheduler limits how many independent operations may execute at once and +chooses which queued operation receives the next slot. It is backed by the +`a3s-lane` priority queue and is enabled without extra setup. + +This is an admission boundary, not a preemptive executor: work that already +owns a slot continues until it completes or is cancelled. Priority determines +which pending operation starts when a slot becomes available. + +### What shares the boundary + +The same `max_active` capacity covers: + +- conversation runs started with send, run, or stream +- trusted or governed direct-tool calls made by the host +- detached background children +- workflows started by the host + +This prevents several sessions from each consuming an independent concurrency +budget. A busy background session cannot bypass interactive work by entering +through a different execution API. + +Three nearby controls solve different problems: + +| Control | Scope | +| ------------------------------- | ---------------------------------------------------------- | +| `task_scheduler.max_active` | Global admission across every session owned by one `Agent` | +| `max_parallel_tasks` | Sibling fan-out inside one delegated task or workflow | +| [Lane queue](/guide/lane-queue) | Optional external or hybrid worker dispatch | + +A session's single-flight rule is separate too: two transcript-changing calls +on the same session fail fast instead of waiting in this scheduler. + +### Configure capacity and aging + +```acl +task_scheduler { + max_active = 4 + aging_interval_ms = 30000 +} +``` + +Both values must be greater than zero. Defaults are four active operations and +a 30-second aging interval. + +### Choose a priority + +| Priority | Intended use | Aging | +| ------------- | ---------------------------------------------------- | --------------------------------- | +| `urgent` | Explicit host control work that must run next | Never ages | +| `interactive` | User-facing turns | Default and maximum aged priority | +| `foreground` | Visible work that is not blocking direct interaction | Ages toward interactive | +| `background` | Detached or asynchronous work | Ages toward interactive | +| `maintenance` | Lowest-priority housekeeping | Ages toward interactive | + +Lower classes run after higher classes. Equal effective priorities remain +FIFO. Every full `aging_interval_ms` promotes waiting non-urgent work by one +level, capped at `interactive`, so continuous user traffic cannot permanently +starve background or maintenance work. `urgent` remains reserved above aged +work. + +Set the priority when creating a session: + +```rust +use a3s_code_core::{SessionOptions, TaskPriority}; + +let options = SessionOptions::new() + .with_task_priority(TaskPriority::Background); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +```ts +const session = await agent.sessionAsync('/repo', { + taskPriority: 'background', +}); +``` + +```python +from a3s_code import SessionOptions + +options = SessionOptions() +options.task_priority = "background" +session = agent.session("/repo", options) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + TaskPriority: code.TaskPriorityBackground, +}) +``` + +Accepted names are `urgent`, `interactive`, `foreground`, `background`, and +`maintenance`. Invalid names fail session-option validation. + +### Observe occupancy + +Hosts can read the same point-in-time snapshot through either the `Agent` or +one of its sessions: + +```rust +let stats = agent.task_scheduler_stats().await?; +let same_scheduler = session.task_scheduler_stats().await?; +println!("active={} pending={}", stats.active, stats.pending); +``` + +```ts +const stats = await agent.taskSchedulerStats(); +const sameScheduler = await session.taskSchedulerStats(); +console.log(stats.active, stats.pendingByPriority.background); +``` + +```python +stats = agent.task_scheduler_stats() +same_scheduler = session.task_scheduler_stats() +print(stats["active"], stats["pendingByPriority"]["background"]) +``` + +```go +stats, err := agent.TaskSchedulerStats(ctx) +sameScheduler, err := session.TaskSchedulerStats(ctx) +fmt.Println(stats.Active, stats.PendingByPriority.Background) +``` + +| Value | Meaning | +| ------------------- | ------------------------------------------------------ | +| `maxActive` | Configured global capacity | +| `active` | Operations currently holding a slot | +| `pending` | Operations waiting for admission | +| `activeByPriority` | Active operations grouped by their requested priority | +| `pendingByPriority` | Waiting operations grouped by their requested priority | +| `closed` | The scheduler is shutting down | + +Rust exposes snake-case struct fields; Node.js and the Python dictionaries use +the camel-case wire names; Go exposes exported struct fields. The snapshot is +diagnostic state, not a reservation—values can change immediately after it is +read. + +### Cancellation and shutdown + +Cancellation removes pending work before it can acquire a slot. Cancelling an +active operation releases its slot when that operation settles. Closing the +`Agent` rejects queued and new admissions, then waits for already-admitted work +to finish before scheduler shutdown completes. + +## Automatic Delegation + +Automatic delegation is opt-in. The runtime scores the current request against built-in and custom agent descriptions, then launches up to `maxTasks` child runs when confidence is high enough. + +```ts +const session = agent.session('/repo', { + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, +}); +``` + +```acl +auto_delegation { + enabled = true + auto_parallel = false + min_confidence = 0.72 + max_tasks = 4 +} +``` + +`autoParallel: false` / `auto_parallel = false` is the global kill switch for automatic parallel child-agent fan-out. Manual `task` fan-out and `session.tasks(...)` remain available. + +## Agent Directories + +Load custom agent definitions through `agentDirs`, `agent_dirs`, or the built-in A3S directories: + +```ts +const session = agent.session('/repo', { agentDirs: ['./.a3s/agents'] }); +const loaded = session.registerAgentDir('./more-agents'); +``` + +A3S scans configured `agent_dirs`, project/user `.a3s/agents`, and Claude-compatible `.claude/agents` migration paths. Prefer `.a3s/agents` for new projects. + +Markdown agent files support frontmatter: + +```markdown +--- +name: docs-auditor +description: Use proactively after documentation changes +tools: Read, Grep, Glob +disallowedTools: + - Write + - Bash(rm:*) +--- + +Audit docs for drift, broken examples, and unclear migration notes. +``` + +The `tools` field is an allowlist. `disallowedTools` is a denylist and wins over allowed tools. Model routing fields are intentionally outside this compatibility layer. + +## Worker Agents + +Register disposable worker agents with `workerAgents` or `registerWorkerAgent()`: + +```ts +const session = agent.session('/repo', { + workerAgents: [ + { + name: 'frontend-worker', + description: 'Small verified frontend fixes', + kind: 'implementer', + model: 'provider/model-id', + maxSteps: 24, + confirmationInheritance: 'auto_approve', + }, + ], +}); +``` + +### Confirmation Inheritance + +Control how child runs resolve Ask decisions with `confirmationInheritance`: + +- `'auto_approve'` (default): child runs auto-approve all Ask decisions +- `'deny_on_ask'`: child runs fail immediately when encountering an Ask +- `'inherit_parent'`: child runs inherit the parent's confirmation policy + +Legacy lifecycle control-plane APIs are removed. Applications that need UI state should consume streaming events, run replay, and Node `cancelRun(runId)`. + +## Programmable orchestration + +Everything on this page is model-driven: `task`, `session.task(...)` / +`session.tasks(...)`, and auto-delegation let the LLM decide when and how to fan +out. When the host already knows the shape of the work and wants it to be +deterministic and reproducible, express it programmatically instead with +`session.parallel(...)`, `session.pipeline(...)`, and +`session.parallelResumable(...)`. See [Orchestration](/guide/orchestration) for +developer-expressed fan-out, barrier-free pipelines, and resumable/migratable +workflows. diff --git a/website/docs/v8.5.1/en/guide/teams.mdx b/website/docs/v8.5.1/en/guide/teams.mdx new file mode 100644 index 00000000..93f19d80 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/teams.mdx @@ -0,0 +1,128 @@ +--- +title: 'Teams' +description: 'Building teams with agent definitions, task tools, and automatic delegation' +--- + +# Teams + +Teams are a harness pattern built from named agent definitions plus the unified +`task` delegation core. A3S Code does not expose a separate team runner API; the +parent session stays responsible for synthesis, policy, and final verification. + +## Recommended Shape + +1. Put durable roles in `.a3s/agents` or configure additional `agentDirs`. +2. Keep each role focused: description, prompt, allowed tools, denied tools. +3. Enable `autoDelegation` when the runtime should select high-confidence subagents. +4. Use `session.task(...)` or `session.tasks(...)` when the host already knows the lanes. +5. Merge child summaries, evidence references, and risks in the parent. + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + planningMode: 'auto', + autoDelegation: { enabled: true, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, +}); + +await session.send(` +Build a release-readiness team: +- explorer: find risky changed areas +- tester: identify missing verification +- security: review side-effect paths + +Use independent subagents where useful and return a single prioritized report. +`); +``` + +## Custom Agent Files + +Markdown agent files use Claude-compatible frontmatter, with A3S-native placement under `.a3s/agents`: + +```markdown +--- +name: release-reviewer +description: Use proactively after release or CI changes +tools: Read, Search, Bash(cargo test*) +disallowedTools: + - Write + - Bash(git push*) +--- + +Review release blockers first, then risks, then follow-up work. +``` + +A3S also reads `.claude/agents` as a migration source. Prefer `.a3s/agents` for new projects. + +## Built-in Team Roles + +Use these without creating files: + +- `explore`: read-only repository exploration +- `plan`: read-only implementation planning +- `general` / `general-purpose`: multi-step implementation +- `verification`: checks, repros, and regression validation +- `review`: findings-first code review + +## Manual Lanes + +SDK callers can call direct helpers when the host already knows the lanes: + +```ts +const result = await session.tasks([ + { + agent: 'explore', + description: 'Changed files', + prompt: 'Find risky changed files.', + }, + { + agent: 'verification', + description: 'Test gaps', + prompt: 'Find missing verification.', + }, + { + agent: 'review', + description: 'Regression review', + prompt: 'Review correctness risks.', + }, +]); + +if (result.exitCode !== 0) throw new Error(result.output); +console.log(result.output); +``` + +`session.tasks(...)` is the host-driven multi-item wrapper around `task`; it +returns a `ToolResult`, not a `StepOutcome[]`. Use `session.parallel(...)` when +you need one structured outcome per lane. + +When the team's lane structure is fixed and should be reproducible and resumable rather than model-chosen, use the programmable combinators in [Orchestration](/guide/orchestration) (`session.parallel` / `session.pipeline` / `session.parallelResumable`). + +## Worker Agents + +Register disposable worker agents with `workerAgents` or `registerWorkerAgent()` when the role is constructed by the host instead of stored on disk: + +```ts +const session = agent.session('/repo', { + workerAgents: [ + { + name: 'frontend-worker', + description: 'Small verified frontend fixes', + kind: 'implementer', + model: 'provider/model-id', + maxSteps: 24, + confirmationInheritance: 'auto_approve', + }, + ], +}); +``` + +Control how child runs resolve Ask decisions with `confirmationInheritance`: + +- `'auto_approve'` (default): child runs auto-approve all Ask decisions +- `'deny_on_ask'`: child runs fail immediately when encountering an Ask +- `'inherit_parent'`: child runs inherit the parent's confirmation policy + +## Runtime State + +Applications should observe streaming events and run replay snapshots, then cancel active Node work with `cancelRun(runId)` when needed. diff --git a/website/docs/v8.5.1/en/guide/telemetry.mdx b/website/docs/v8.5.1/en/guide/telemetry.mdx new file mode 100644 index 00000000..61c0e850 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/telemetry.mdx @@ -0,0 +1,136 @@ +--- +title: 'Telemetry' +description: 'Trace events, verification summaries, and runtime observability' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Telemetry + +A3S Code exposes runtime evidence through events, traces, verification reports, +and result metadata. + +## Result metadata + + + + +```rust +let result = session.send("Run release checks", None).await?; + +println!("{}", result.usage.prompt_tokens); +println!("{}", result.usage.completion_tokens); +println!("{}", result.usage.total_tokens); +println!("{}", result.tool_calls_count); +println!("{:?}", result.verification_summary().status); +println!("{}", result.verification_summary_text()); +``` + + + + +```ts +const result = await session.send('Run release checks'); + +console.log(result.promptTokens); +console.log(result.completionTokens); +console.log(result.totalTokens); +console.log(result.toolCallsCount); +console.log(result.verificationStatus); +console.log(result.verificationSummaryText); +``` + + + + +```python +result = session.send('Run release checks') + +print(result.prompt_tokens) +print(result.completion_tokens) +print(result.total_tokens) +print(result.tool_calls_count) +print(result.verification_status) +print(result.verification_summary_text) +``` + + + + +```go +result, err := session.Run(ctx, "Run release checks") +if err != nil { + return err +} + +fmt.Println(result.Usage.PromptTokens) +fmt.Println(result.Usage.CompletionTokens) +fmt.Println(result.Usage.TotalTokens) +fmt.Println(result.ToolCallsCount) +fmt.Println(result.VerificationSummary.Status) +fmt.Println(result.VerificationSummaryText) +``` + + + + +## Traces and verification + + + + +```rust +let trace = session.trace_events(); +let reports = session.verification_reports(); +let summary = session.verification_summary(); +let text = session.verification_summary_text(); +``` + + + + +```ts +const trace = session.traceEvents(); +const reports = session.verificationReports(); +const summary = session.verificationSummary(); +const text = session.verificationSummaryText(); +``` + + + + +```python +trace = session.trace_events() +reports = session.verification_reports() +summary = session.verification_summary() +text = session.verification_summary_text() +``` + + + + +```go +trace, traceErr := session.TraceEvents(ctx) +reports, reportsErr := session.VerificationReports(ctx) +summary, summaryErr := session.VerificationSummary(ctx) +text, textErr := session.VerificationSummaryText(ctx) +``` + + + + +The Go methods cross the native bridge and therefore return an `error`; Rust, +Node.js, and Python expose these in-process observation getters synchronously. + +## Streaming events + +Streaming returns versioned `AgentEvent` envelopes. `version`, `type`, +`payload`, and optional `metadata` are the stable lossless fields; text, tool +names, tool output, errors, token totals, and verification summaries are +convenience projections. Unknown future `type` values and their payloads remain +available to hosts. + +## Logging + +For product telemetry, prefer structured trace events and verification reports +over scraping console output. diff --git a/website/docs/v8.5.1/en/guide/tools.mdx b/website/docs/v8.5.1/en/guide/tools.mdx new file mode 100644 index 00000000..96992b73 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/tools.mdx @@ -0,0 +1,652 @@ +--- +title: 'Tools' +description: 'Tool selection, direct tool calls, and verification evidence' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Tools + +A3S Code keeps a registry of available tools. `toolNames()` returns the current +session tool surface. That surface is assembled from the workspace capability +set plus session-level integrations, so a non-local workspace can intentionally +hide tools it cannot service. + +Tool activity reaches clients through the session event stream, so a UI can +render starts, output, errors, and completion without parsing terminal text. + +## Tool Surface + +| Layer | Tools | Registration rule | +| ------------------------ | --------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Workspace-gated builtins | `read`, `write`, `edit`, `patch`, `download`, `search`, `ls`, `bash`, `git` | Registered only when `WorkspaceServices` advertises the required capability. `search` exposes grep/glob modes and adds BM25 when reads are available; `download` additionally requires a writable local workspace. | +| Runtime builtins | `web_fetch`, `web_search`, `batch`, `program` | Registered by the core tool executor. `batch` and `program` receive the current scoped invoker, so their usable inner tools depend on the session surface and keep the caller's governance scope. | +| Session bootstrap tools | `task`, `generate_object`, `search_skills`, `Skill` | Added while building an `AgentSession`. `task` is the single model-visible delegation schema (including multi-item fan-out). The `parallel_task` alias is removed (`HARNESS-CONV4`). Delegation disappears when manual delegation is disabled. | +| TUI workflow tools | `dynamic_workflow` and, after `/login`, optional host runtime tools | Registered by the `a3s code` host. `dynamic_workflow` powers `ultracode` and DeepResearch with A3S Flow replay; host runtime tools are login-gated integrations. | +| Dynamic integrations | `mcp____` and host-registered tools | Added from MCP managers or host code after discovery. | + +Use `toolNames()` / `toolDefinitions()` in tests or application diagnostics when +a workflow depends on a specific model-visible tool. Tool visibility is not a +security grant. Model-selected tool calls inside `send`, `run`, and `stream` +pass through active-skill restrictions, permission policy, confirmation, hooks, +budget, queue/timeouts, cancellation, recursive-call protection, output +sanitization, artifact limits, and workspace path checks. Direct SDK calls such +as `session.tool(...)` are host control-plane calls with a different explicit +policy, described below. +Use `activeTools()` for a different question: which tool calls are currently +running in an active operation. + +## One Invocation Kernel + +The runtime attaches an origin to every invocation: + +| Origin | Examples | Permission / confirmation policy | +| -------------------------- | -------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Agent | A model emits a tool call during `send` or `stream` | Full model-facing policy and HITL apply. | +| Governed nested | Inner calls from a model-owned `batch`, `program`, workflow, or public custom `InvocationRuntime` | Re-enters the model-facing policy and inherits the invocation stack, cancellation, budget, hooks, and sandbox. Ambient host-direct context cannot change this origin. | +| Runtime internal | `dynamic_workflow` starts its private `program` execution engine | Skips only the duplicate permission/HITL decision already made for `dynamic_workflow`; hooks, evidence, budgets, cancellation, recursion checks, and all script Tool calls remain governed. | +| Host direct | `session.tool(...)`, typed read/write/git helpers, `session.program(...)`, and direct task helpers | Trusted control plane: model-facing permission/HITL are skipped because the host selected the call. | +| Trusted host-direct nested | A built-in `batch`, `program`, or dynamic workflow that the host called directly invokes a host-selected child | Retains trusted control-plane authority for that built-in nested operation. Third-party tools cannot construct this origin through the public API. | + +Model sub-runs created by a host-direct Skill, Task, or custom tool start again +with the Agent origin. A public custom tool can make nested calls only through +`InvocationRuntime`, which always creates the governed nested origin. This +prevents one direct call to arbitrary extension code from becoming a reusable +authority token. + +All origins use the same invocation kernel. Pre-hooks can block them; +budget checks run before side effects; queue/timeout, cancellation, recursive +invocation protection, post-hooks, and security-provider output sanitization +remain active. A nested call cannot fall back to a raw registry while a scoped +invoker is installed. + +The host-direct policy is not an end-user authorization system. The embedding +application must authenticate and authorize a user before translating their +request into a direct SDK helper. Closing the session cancels an in-flight +host-direct tool through the session cancellation scope. + +## Bounded Tool Contracts + +Governed agent, nested, and session calls validate arguments against each +tool's cached JSON Schema before confirmation or side effects. Tools also +publish per-invocation scheduling capabilities: read-only, idempotent, +resumable, cancellation-safe, pagination support, maximum parallelism, and +output kind. + +`read`, `ls`, paginated `search` modes, Git log/list/diff, and `web_fetch` return explicit +continuation cursors or offsets. Shell capture is byte-bounded across both +streams, retains the beginning and end with exact accounting, and still +exposes stdout and stderr separately to a sandbox host. A command deadline +includes both output draining and `child.wait()`, so closing both pipes cannot +evade it. Timeout and cancellation terminate the complete Unix process group. +Large file changes replace full before/after event payloads with bounded +previews and unified diff text plus hashes, sizes, and artifact references. + +Every Tool result includes `metadata.a3s_tool_result_evidence` with schema +`a3s.code.tool-result-evidence.v1`. It records original and projected byte +counts, deterministic token estimates, an exact SHA-256 repeat key, the loss +mode, and an immutable inline or artifact content reference. These values are +Harness observations rather than provider billing records, and this evidence +does not itself transform the Tool result. + +### Deterministic Tool-result projection + +Each session pins one `a3s.code.tool-result-transform-policy.v1` policy. The +default conservative policy retains the first 100 KiB and does not fold or +sample content. The context-efficient profile keeps a UTF-8-safe 64 KiB head +and 32 KiB tail, folds three or more exact repeated lines, and samples up to 32 +items from an oversized top-level JSON array. + + + + +```rust +use a3s_code_core::{tools::ToolResultTransformPolicyV1, SessionOptions}; + +let options = SessionOptions::new().with_tool_result_transform_policy( + ToolResultTransformPolicyV1::context_efficient(), +); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + + + + +```ts +const session = await agent.sessionAsync('/repo', { + toolResultTransformPolicy: { + schema: 'a3s.code.tool-result-transform-policy.v1', + maxOutputBytes: 100 * 1024, + headBytes: 64 * 1024, + tailBytes: 32 * 1024, + foldRepeatedLines: true, + repeatedLineThreshold: 3, + structuredSampleItems: 32, + }, +}); +``` + + + + +```python +from a3s_code import SessionOptions, ToolResultTransformPolicy + +options = SessionOptions() +options.tool_result_transform_policy = ToolResultTransformPolicy.context_efficient() +session = agent.session("/repo", options) +``` + + + + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + ToolResultTransformPolicy: &code.ToolResultTransformPolicy{ + Schema: "a3s.code.tool-result-transform-policy.v1", + MaxOutputBytes: 100 * 1024, + HeadBytes: 64 * 1024, + TailBytes: 32 * 1024, + FoldRepeatedLines: true, + RepeatedLineThreshold: 3, + StructuredSampleItems: 32, + }, +}) +``` + + + + +Projection is deterministic: Core samples an oversized JSON array first, +folds exact repeated lines next, and only then applies the UTF-8-safe head/tail +bound if the result is still too large. `max_output_bytes` may be 1-100 KiB; +non-compatibility profiles must leave 512 bytes inside that bound for the +transformation marker. + +The policy is stored in the session snapshot. Resume inherits it when omitted +and rejects an explicitly different policy, so replay cannot silently change +what the model observed. + +Every result records `original_bytes`, `projected_bytes`, original/projected +token estimates using `utf8-bytes-ceil-div-4/v1`, source and projection +digests, byte/token deltas, `repeat_key`, `content_ref`, and +`transform_algorithm`. `loss_mode` is one of `none`, `bounded_preview`, +`head_tail`, `deterministic_transform`, or `composite`. Lossless results use an +inline SHA-256 reference; lossy results retain the complete original under an +immutable `a3s://tool-output/...` artifact URI. + +### Binary-safe local downloads + +The `download` tool writes an HTTP(S) resource into a writable local workspace. +It is not registered for S3, browser, or other non-local backends. Model-selected +calls are workspace mutations, so the normal permission policy and HITL path +apply; a direct `session.tool(...)` call is a privileged host decision. + +```ts +const result = await session.tool('download', { + url: 'https://downloads.example.com/model.bin?signature=...', + file_path: 'artifacts/model.bin', + overwrite: false, + connections: 4, + max_bytes: 536870912, + timeout: 300, + expected_sha256: + '0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef', +}); +``` + +| Parameter | Required | Contract | +| ----------------- | -------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `url` | yes | Public `http://` or `https://` URL. User information is rejected; fragments are removed. Signed query parameters are preserved for the actual requests. | +| `file_path` | no | Workspace-relative destination. If omitted, a sanitized filename is inferred from `Content-Disposition`, URL `file` / `filename`, the path, or `download.bin`. | +| `overwrite` | no | Replace an existing regular file only after the replacement is complete and verified. Defaults to `false`. | +| `connections` | no | Requested Range concurrency from 1 through 4. If omitted, concurrency is selected by size; small or unvalidated resources use one coherent response. | +| `max_bytes` | no | Declared and streamed-byte limit. Defaults to 512 MiB (`536870912`); hard maximum 8 GiB (`8589934592`). | +| `timeout` | no | Total deadline in seconds, including retries, hashing, and promotion. Defaults to 300; hard maximum 3600. | +| `expected_sha256` | no | Exactly 64 hexadecimal characters. A mismatch leaves the destination unchanged. | + +Before each redirect hop, the shared safe HTTP transport validates the target +against SSRF-sensitive addresses. Direct connections reject mixed +public/private DNS answers and pin the validated addresses for that hop. +Redirects are bounded and revalidated; cross-origin hops do not inherit +credentials or `If-Range`. An explicitly configured proxy resolves hostnames, +while URL-level literal-address checks still run. + +The downloader probes Range support, requires exact `Content-Range` and body +boundaries, and uses a stable validator before parallelizing independent ranges. +It retries a bounded set of transport, rate-limit, and server failures, then +falls back to one sequential response when a parallel protocol or network +attempt is unsafe. Data streams into an adjacent temporary file. Cancellation, +timeouts, size failures, and checksum failures remove that partial file; only a +fully synced and optionally verified file is atomically promoted. + +Result metadata contains the workspace path, byte count, content type, strategy, +connection count, Range support, overwrite status, safe source anchors, and the +verified digest when requested. Signed query parameters are used for requests +but removed from source anchors and diagnostics, so they are not leaked through +tool metadata. + +### Repository-context modes + +Use `read.files` when the relevant paths are already known and one bounded +response is cheaper than several tool turns: + +```json +{ + "files": [ + { "path": "src/lib.rs" }, + { "path": "src/config.rs", "offset": 40, "limit": 80 } + ], + "max_output_bytes": 65536 +} +``` + +The shared byte budget includes headers and the continuation text. Results +stay in request order, and one unreadable member does not discard successful +members. When `metadata.batch.truncated` is true, copy +`metadata.batch.continuation` into the next call's `files` value; its offsets +and remaining limits resume without repeating completed lines. + +`search` is the single model-facing workspace search tool. Always pass `mode` +and `query`: use `grep` for regular-expression content search, `glob` for path +discovery, and `bm25` for native lexical relevance ranking. These modes are +separate planes: `grep` never opens durable zvec FTS, and `bm25` never becomes +the exact-match authority. When the host +explicitly enables Workspace Retrieval, the same schema also advertises +`semantic` for exact cosine ranking over the session-owned, Memory-authoritative +index and +`hybrid` for reciprocal-rank fusion across exact, lexical, symbol, and semantic +evidence. Disabled sessions omit those two modes. The shared `path` field +scopes all modes; `include` filters candidate files for grep, BM25, semantic, +and hybrid search. + +In `mode: "grep"`, `output_mode` selects the smallest useful result shape: + +| Mode | Result | +| -------------------- | ------------------------------------------------------------- | +| `content` | Matching lines and optional context (default) | +| `files_with_matches` | Lexically cursor-paginated matching paths | +| `count` | Lexically cursor-paginated matching-line counts per file | +| `summary` | Full-scan line and file totals without rendered match content | + +Built-in workspace backends perform non-content scans without constructing +discarded match text. With default `local-code`, manifest-backed workspaces +also build a lazy in-tree trigram candidate cache under `.a3s-code/grep-trigram` +so literal needles open fewer files before the exact regex scan. Non-literal +patterns and index failures fail open to today's full scan. S3 results set +`metadata.search.truncated` and warn when +the backend's object scan limit makes totals or paths incomplete. + +In `mode: "glob"`, the `query` is the glob pattern. Backend relevance or +recency order is preserved by default; set `sort: "path"` before cursor +pagination when stable lexical pages are required. + +In `mode: "bm25"`, the `query` is plain text. A bounded native Rust scorer +splits code identifiers and CJK text, ranks 80-line chunks, and returns only +top-k snippets plus source anchors. It uses workspace search to narrow +candidates and keeps at most 256 files, 512 KiB per file, and 16 MiB in memory; +no database, embedding model, or external reranker is required. + +```ts +const ranked = await session.tool('search', { + mode: 'bm25', + query: 'workspace permission policy', + path: 'core/src', + include: '*.rs', + limit: 8, + context: 2, +}); +``` + +Semantic and hybrid calls use the same tool rather than a separate vector +database tool: + +```json +{ + "mode": "hybrid", + "query": "where session shutdown releases temporary indexes", + "path": "core/src", + "include": "*.rs", + "limit": 8 +} +``` + +Index construction is asynchronous and file-atomic. Results are reread and +digest-verified against current source before rendering, and closing the +session releases every vector. See +[Workspace Retrieval](/guide/context#workspace-retrieval) for activation, +partial-readiness behavior, embedding routes, resource bounds, and lifecycle. + +For exact-string changes, call `edit` with `dry_run: true` to return the same +before/after diff metadata without writing. Then apply the edit with +`expected_replacements` set to the previewed count, and optionally add +`max_replacements` as an independent upper bound. Dry runs advertise read-only +capability and are safe for `batch` parallelization. + +`batch` accepts at most 32 calls and applies at most 16-way concurrency. It +fans out only when every child declares safe read-only, idempotent behavior; +mutating and unknown tools are serialized. A partial batch identifies failed +indices while treating the orchestration as completed, so callers retry only +failed items. Multi-item `task` fan-out has the same 32-task bound and settles +cancelled children before publishing terminal state. + +The model-facing `task` schema always uses `tasks` with 1-32 items. One item runs +a focused child and may use `background`; multiple independent items run +concurrently and cannot set `background: true`. Each item accepts `agent`, +`description`, `prompt`, optional `max_steps`, and optional `output_schema`. +`min_success_count` is valid only with `allow_partial_failure: true` and must be +between 1 and the submitted item count. Provider and child runtimes own typed +retry policy; the fan-out layer never replays a branch based on error text. + +### Structurally gated web search + +`web_search` reports `complete`, `partial`, or `failed` in metadata. Its default +path executes headless engines first, conventional HTTP/RSS engines only when +the combined structural retrieval requirements are not met, and native APIs +only when both earlier tiers remain insufficient. Browser discovery and pool creation are +therefore lazy. A3S Code v8.5.1 pins `a3s-search` v3.1.0 and uses the packaged +or shared-cache Moli runtime by default. Minimal Rust embeddings can disable +default features; Chrome and Lightpanda remain explicit configured backends. + +Moli resolution checks an explicit executable, the package sidecar, the verified +per-user cache, and a discoverable system installation before attempting an +HTTPS download of the pinned release. The cache is protected by a cross-process +install lock and atomic receipts, so multiple `a3s-code` processes reuse one +installation. Set `auto_download_moli = false` for an offline/strict deployment. +On Linux musl, the release emits `MOLI_UNAVAILABLE` because upstream Moli has no +musl asset; provide a system/explicit executable or choose another backend. + +```acl +search { + headless { + backend = "moli" + auto_download_moli = true + max_tabs = 4 + } +} +``` + +The final cascade fails closed when it still does not satisfy the structural +retrieval requirements. Successful JSON output keeps the result-array contract. +Insufficient JSON output is an error with a typed +`retrieval_requirements_not_met` envelope, the candidate rows, observed +retrieval health, and the required structural thresholds. Search does not use +an external semantic verifier or reranker. + +Session-scoped closed/open/half-open circuit state skips known quota, +permission, rate-limit, transport, repeated-empty, and timeout failures without +retrying them on every request; `Retry-After` is retained. Delegated research +contexts share search bulkheads, bounded browser retry budgets, and +identical-request coalescing. Request-scoped proxies reach the lazy browser +tier, and `search_coalescing` metadata reports leader, shared, bypassed, and +abandoned requests. An explicit `engines` argument runs only the requested +tiers. Empty results with engine errors are failures rather than successful +empty searches. Timeout, cancellation, invalid-argument, partial-failure, and +rate-limit errors carry structured error kinds, while tier decisions, retrieval +health, engine outcomes, attempt duration, and retry context remain available +in metadata. + +`web_fetch` also preserves failure semantics instead of deriving retry advice +from rendered messages. Request and response-body I/O failures, HTTP 408, and +HTTP 5xx responses use the typed `transport` kind. HTTP 429 uses +`rate_limited` and carries a parsed `Retry-After` delay when the server supplies +one; the outer tool deadline uses `timeout`. Other HTTP status failures remain +ordinary status errors unless the runtime has typed evidence that retrying is +safe. + +## Structured Output with `generate_object` + +The `generate_object` tool asks the configured LLM for a JSON value, validates +the response against a JSON Schema, and returns the validated value only on a +zero-exit result. Root objects, arrays, enums, constants, composition keywords, +and local `$ref` definitions are supported. The active timeout starts after +model-generation admission, is forwarded to clients that accept an active +transport budget, and remains in force across bounded schema repairs. It works +in two ways: + +1. **Agent-driven**: The LLM sees `generate_object` in its tool list and calls it autonomously when structured output is needed. +2. **Direct call**: Your application calls `session.tool('generate_object', ...)` to bypass model-driven tool selection. The tool itself still calls the configured LLM. + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['sentiment', 'confidence'], + properties: { + sentiment: { type: 'string', enum: ['positive', 'negative', 'neutral'] }, + confidence: { type: 'number', minimum: 0, maximum: 1 }, + }, + }, + prompt: 'Classify: "This product is amazing!"', + schema_name: 'sentiment', + mode: 'tool', + max_repair_attempts: 2, // retries if schema validation fails +}); + +if (result.exitCode !== 0) { + throw new Error(result.output); +} + +const { object } = JSON.parse(result.output); +// object = { sentiment: "positive", confidence: 0.95 } +``` + +### Parameters + +| Parameter | Type | Required | Description | +| --------------------- | ------- | -------- | -------------------------------------------------------------------------------- | +| `schema` | object | yes | JSON Schema used to validate the output value | +| `prompt` | string | yes | Non-whitespace generation or extraction instruction | +| `schema_name` | string | no | 1-59 ASCII letters, digits, `_`, or `-` (default: `result`) | +| `schema_description` | string | no | Synthetic-tool description, at most 4,096 bytes | +| `system` | string | no | Optional system prompt, at most 32,768 bytes | +| `mode` | string | no | `"auto"` / `"strict"` / `"json"` / `"tool"` / `"prompt"` (default: auto) | +| `max_repair_attempts` | integer | no | 0-5 (default: 2) | +| `include_raw_text` | boolean | no | Include the provider text or tool arguments used for extraction (default: false) | +| `timeout_ms` | integer | no | Active generation deadline, 1,000-600,000 ms (default: 120,000) | + +### Modes + +- **tool**: Forces a synthetic tool whose parameters are the schema when the provider supports forced tool calls. +- **prompt**: Appends schema instructions to the prompt. It is useful as a prompt-only fallback, but depends more on model compliance. +- **auto**: Selects forced-tool mode when available, otherwise prompt mode. +- **strict**: Uses provider-native strict JSON Schema when supported, otherwise safely falls back to forced-tool or prompt mode. +- **json**: Uses provider-native JSON-object mode when supported, otherwise safely falls back to forced-tool or prompt mode. + +Every resolved mode retains the provider-facing response schema as host-only +validation metadata; that metadata is never serialized as an extra provider +request field. Rust composite clients can use +`structured::is_complete_streamed_value(...)` to accept only a complete JSON +value that validates against this schema, including endpoints that omit a +terminal stream event. They can also inspect +`LlmClient::has_distinct_non_streaming_transport()` before treating a blocking +call as an independent fallback instead of replaying the same streaming +failure mode under another method name. + +### Streaming + +When called through `session.stream()`, partial objects are emitted as +`tool_output_delta` events. Snapshots are rate-limited to one event per 100 ms; +objects over the event budget emit byte accounting instead of duplicating the +full value in the stream: + +```ts +const stream = await session.stream('Extract all invoices...'); + +while (true) { + const { value: ev, done } = await stream.next(); + if (done) break; + if (!ev) continue; + + if (ev.type === 'tool_output_delta' && ev.toolName === 'generate_object') { + const { object_partial } = JSON.parse(ev.text); + renderProgress(object_partial); + } +} +``` + +### Repair Loop + +If the LLM output fails schema validation, the tool automatically retries by +feeding the validation errors back to the model. This handles edge cases like +missing required fields or wrong enum values without application-level retry logic. + +## Direct Tool Calls + +SDK callers can call deterministic tools directly: + +```ts +const files = await session.glob('src/**/*.rs'); +const hits = await session.grep('PermissionPolicy'); +const status = await session.git('status'); +const output = await session.bash('cargo test -p a3s-code-core'); +const raw = await session.tool('read', { file_path: 'README.md' }); +const schemas = session.toolDefinitions(); +const hitLines = hits.split('\n').filter(Boolean).length; + +console.log(files.length, hitLines, output.length, schemas.length); +console.log(status.output); +console.log(raw.output); +``` + +Direct calls execute inside the session workspace and should be treated as +privileged host operations. They do not claim the session's single-flight +conversation lease because they do not update transcript history. + +For delegated child work, use SDK helpers over the same core tools: + +```ts +await session.task({ + agent: 'explore', + description: 'Find auth files', + prompt: 'Inspect auth-related files and return a compact evidence list.', +}); + +await session.tasks([ + { agent: 'explore', description: 'Find tests', prompt: 'Locate auth tests.' }, + { + agent: 'verification', + description: 'Check risk', + prompt: 'Review auth edge cases.', + }, +]); +``` + +Automatic subagent delegation also uses these core tools. `autoParallel: false` +disables only automatic parallel fan-out; it does not remove manual `task` +fan-out or `session.tasks(...)`. + +## Programmatic Tool Calling + +PTC is the next step beyond a single direct call. The `program` tool runs a sandboxed JavaScript script in an embedded QuickJS VM. The script defines `async function run(ctx, inputs)` and replaces repeated model-tool turns with one bounded program. + +Instead of spending LLM turns on: + +```text +grep -> read -> grep -> read -> summarize +``` + +the model can ask `program` to run a script: + +```js +// search-auth.js +export default async function run(ctx, inputs) { + const hits = await ctx.grep(inputs.query, { glob: '*.rs' }); + const files = await ctx.glob('crates/**/*.rs'); + const snippets = []; + + for (const file of files.slice(0, 20)) { + const content = await ctx.readFile(file); + if (content.includes(inputs.query)) { + snippets.push({ file, preview: content.slice(0, 1200) }); + } + } + + return { + summary: `Found ${snippets.length} candidate files for ${inputs.query}`, + evidence: snippets, + rawSearch: hits, + }; +} +``` + +Run it through the SDK helper with either inline `source` or a workspace-relative `.js` or `.mjs` file path: + +```js +await session.program({ + path: 'scripts/ptc/search-auth.js', + inputs: { query: 'PermissionPolicy' }, + allowedTools: ['grep', 'glob', 'read'], + limits: { + timeoutMs: 30000, + maxToolCalls: 30, + maxOutputBytes: 65536, + }, +}); +``` + +`session.program(...)` is equivalent to `session.tool('program', { type: 'script', language: 'javascript', ... })` but uses SDK-native naming. When `allowedTools` / `allowed_tools` is omitted, the script can call every registered tool except `program`. Provide an allow-list when a workflow should run with a smaller capability surface. + +`ctx.readFile(path, options)` returns the selected text without the `read` +Tool's line anchors or continuation footer, which makes it suitable for string +processing. Use `ctx.read(path, options)` when the script needs the complete +Tool result, including line-numbered `output`, exit code, metadata, and +continuation evidence. Both forms invoke the same governed `read` Tool and +consume the same script call budget. + +The QuickJS VM receives no filesystem, network, subprocess, or environment permissions. The only useful capabilities are the `ctx` methods wired back to A3S Code tools. PTC returns a readable `ToolResult.output` and structured data in `ToolResult.metadataJson`. Keep large raw output out of the prompt; summarize findings, evidence refs, risks, and suggested next actions. + +In `a3s code`, PTC is also used by `DynamicWorkflowRuntime`. Recursive +`program`, `dynamic_workflow`, and the removed `parallel_task` alias are kept out +of the default TUI PTC allow-list. QuickJS may call `task` with one item, but +direct multi-item fan-out is blocked. A dynamic workflow schedules a Flow step +named `task` for local fan-out; the TUI host executes it outside QuickJS. After +`/login`, +dynamic workflow PTC steps may call `ctx.tool("runtime", ...)` when a host +runtime tool is registered in the session. + +Authorizing a model-selected `dynamic_workflow` authorizes its private QuickJS +execution engine; callers do not need a second `program(*)` rule for that +implementation detail. Every Tool named in `allowed_tools` still re-enters the +normal permission, confirmation, Hook, budget, sandbox, and cancellation path. + +Dynamic workflows default structured generation to single-flight. Set +`limits.maxConcurrentGenerations` to 2-4 only when independent +`generate_object` steps should fan out; each admitted step receives a client +fork bound to its exact run and step identity. Providers that cannot fork a +session remain single-flight. The Rust helper +`dynamic_workflow::recover_dynamic_workflow_step_output(...)` can recover a +completed durable step only when the run id, original query, and step id all +match. It does not act as a cross-run query cache and never promotes an +incomplete step. + +## Tool Results + +`ToolResult` includes `name`, `output`, `exitCode`, and optional `metadataJson`. +Direct `session.tool(...)`, `session.program(...)`, `session.git(...)`, +`session.writeFile(...)`, `session.ls(...)`, `session.editFile(...)`, and +`session.patchFile(...)` calls return `ToolResult`. The typed read/search/shell +helpers return simpler values: `readFile`, `grep`, and `bash` return strings, +while `glob` returns a string array. Long outputs should be summarized before +they are fed back to the model. + +## Verification + +Use verification commands to turn "done" into evidence: + +```ts +const report = await session.verifyCommands('release readiness', [ + { + id: 'unit', + kind: 'test', + description: 'Run core tests', + command: 'cargo test -p a3s-code-core', + required: true, + timeoutMs: 120000, + }, +]); +``` diff --git a/website/docs/v8.5.1/en/guide/tui.mdx b/website/docs/v8.5.1/en/guide/tui.mdx new file mode 100644 index 00000000..cfe0ef06 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/tui.mdx @@ -0,0 +1,719 @@ +--- +title: 'A3S Code TUI' +description: 'Install, configure, and operate the a3s code terminal workspace' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# A3S Code TUI + +`a3s code` is the interactive terminal workspace for A3S Code. It is shipped by +the [`a3s` CLI](https://github.com/A3S-Lab/a3s), drives the Rust +`a3s-code-core` runtime, and renders the runtime event stream with +[`a3s-tui`](https://github.com/A3S-Lab/TUI). + +Use the TUI when you want the ready coding-agent product in a terminal. Use an +A3S Code SDK when you are embedding the same runtime in another host, runner, +IDE bridge, or controlled product surface. + +## Surface Map + +| Surface | Repository | Responsibility | +| ------------- | ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| A3S Code SDKs | [A3S-Lab/Code](https://github.com/A3S-Lab/Code) | Rust runtime crate plus Node.js, Python, and Go SDKs for sessions, tools, permissions, persistence, verification, and event replay. | +| `a3s code` | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | Terminal coding-agent application that drives `a3s-code-core` sessions. | +| A3S Use | [A3S-Lab/Use](https://github.com/A3S-Lab/Use) | Independently released Browser, native Office, built-in OCR, optional Office compatibility, and signed external application capabilities projected into Code through standard MCP and Skills. | +| `a3s-tui` | [A3S-Lab/TUI](https://github.com/A3S-Lab/TUI) | Terminal UI framework used by the CLI. It is the renderer and component library, not the agent runtime. | +| A3S Flow | [A3S-Lab/Flow](https://github.com/A3S-Lab/Flow) | Durable workflow engine used by `DynamicWorkflowRuntime` for replayable per-turn dynamic workflows. | +| A3S monorepo | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | Product docs, release orchestration, submodule pins, and related crates. | + +## Install And Start + +Run the installer for your platform: + + + + +```bash +curl --proto '=https' --tlsv1.2 -LsSf \ + https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.sh | sh +``` + + + + +```powershell +[Net.ServicePointManager]::SecurityProtocol = [Net.ServicePointManager]::SecurityProtocol -bor [Net.SecurityProtocolType]::Tls12 +irm https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.ps1 | iex +``` + + + + +The scripts select the release archive for the current system and architecture +and verify its SHA-256 digest. You can also use +`brew install a3s-lab/tap/a3s` or `cargo install a3s`. + +After installation, run the CLI from the workspace the agent should inspect: + +```bash +a3s code +a3s code resume +a3s code resume +a3s code update +``` + +The top-level `a3s update` command reaches the same updater as TUI `/update`. + +## First Launch And Config + +The TUI discovers config in this order: + +1. `A3S_CONFIG_FILE` +2. `.a3s/config.acl` while walking upward from the current directory +3. `~/.a3s/config.acl` + +On first launch, when no config exists, the CLI creates a starter +`~/.a3s/config.acl` and opens it in the built-in editor. Use `/config` later to +edit the active ACL file from inside the TUI. + +Project-local config can set models, providers, an optional A3S OS endpoint +with `os = "https://..."`, `flow_dir`, `agent_dir`, `mcp_dir`, `skill_dir`, +storage, memory, delegation, and asset paths. Keep real keys, private provider +URLs, tenant identifiers, and user-specific paths out of committed examples. +Resolve credentials through environment variables: + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 8 +os = env("A3S_OS_URL") + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} +``` + +## Capability Overview + +`a3s code` combines the coding chat loop, file and config editing, durable +context, memory, local asset development, optional host integrations, runtime +fan-out, trusted runtime views, and engineered automation loops. + +| Area | What the TUI provides | +| ------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Coding loop | Chat with the agent, stream reasoning/text/tool events, approve or deny gated tools, switch `/auto`, run direct shell turns with `!`, set a persistent `/goal`, clear context, and fork sessions. | +| Workspace UI | `/ide` opens a file tree and editor, while `/config` edits the active config in the same surface. Both use terminal-safe type-aware file/folder sigils, semantic icon colors, icon-bearing breadcrumbs, and a ruled line-number gutter. `Ctrl+T` opens the complete semantic transcript, and file edits render bounded diffs through shared TUI components. | +| Models | `/model` switches configured ACL providers and signed-in account-backed model tabs when available. | +| Effort | `/effort` changes thinking budget, tool-round budget, continuation count, and rigor guidance from `low` through `max` and `ultracode`. | +| Tools and safety | File, binary-safe `download`, search, shell, git, web, structured-output, MCP, PTC `program`, and model-visible `task` tools pass through workspace boundaries, permission policy, HITL approval, timeouts, hooks, and traces. | +| Application capabilities | A3S Use is prepared on first TUI use when policy allows, then projects built-in Browser, native Office, OCR, optional OfficeCLI compatibility, and installed signed external domains into a dedicated restricted `use` worker. | +| Context and memory | The footer tracks context fill and auto-compaction. `/ctx` searches past sessions, `/ctx ` attaches a transcript window, `/ctx save ` promotes it to memory, `/sleep` consolidates the day, and `/memory` browses durable memories as an event/entity graph. | +| Dynamic workflows | `ultracode` and `?` DeepResearch can use `DynamicWorkflowRuntime`, a local A3S Flow-backed runtime that records workflow and step history while sandboxed PTC scripts perform tool work. | +| Parallel work | Local fan-out uses `task` with multiple independent `tasks[]` items. Dynamic workflows schedule a host Flow step named `task`; QuickJS/PTC may call one item directly, while direct multi-item fan-out is blocked. | +| Optional runtime tools | A3S OS can register runtime tools after `/login`; local tools and `task` remain available without an account. | +| Deep research | Prefix a prompt with `?` to start the standalone evidence-first engine. Exact-query bootstrap and bounded semantic planning run concurrently, fetched evidence is published before optional synthesis, and generic fallback never depends on a topic or named entity. | +| Asset development | `/agent`, `/mcp`, `/skill`, and `/okf` enter local development modes with an active asset, review commands, clone/draft flows, and publish/deploy/status surfaces. | +| Workflow assets | `/flow` selects or drafts workflow DAG files for local review and optional host publication. | +| Knowledge | `/kb` manages a local personal knowledge vault. `/okf` manages shareable OKF knowledge-package assets. | +| Engineered loops | `/loop init`, `/loop run`, `/loop audit`, and `/loop logs` manage durable maker/checker loops under `.a3s/loops` with reports, budgets, state files, and optional runtime/view evidence. | +| Operations | `/help` opens the command guide, `/theme` changes syntax themes, `/plugin` and `/reload` refresh skills/plugins, `Open view` reopens the latest validated local report or trusted runtime view when shown, and `/update` upgrades and restarts the CLI. | + +DeepResearch writes progressively publishable artifacts to +`.a3s/research/artifacts//report.md` and +`.a3s/research/artifacts//index.html`. The independent +`a3s-deep-research` crate owns orchestration, evidence admission, typed +claim-graph compilation, quality gates, and rendering; CLI and TUI +share one typed runner and supply only product adapters. + +The request language is pinned across planning, admission, and publication. +Comprehensive publication requires report-wide volume and source diversity plus +a multi-step analytical chain in every resolved material dimension: multi-source +facts, comparison, mechanism or trade-off analysis, cross-source synthesis, and +a supported implication or applicability boundary. Repeated openings, +near-duplicate claims, and source-by-source summaries do not satisfy that gate. +A closed narrative plan selects natural headings and paragraph groups without +changing evidence; rendering uses continuous prose and keeps traceability in a +collapsed disclosure. + +The self-contained HTML follows A3S Code's design tokens with a sticky left +action menu, centered editable report, sticky right table of contents, save and +print actions, and a responsive stacked layout. Each run also records a bounded +`.a3s/research/runs//journal-v2.jsonl` projection without absolute +artifact paths. Complete material coverage publishes `synthesized`; supported +incomplete work may publish depth-gated `qualified`; failed synthesis preserves +`source_backed`; and an empty safe-source set produces `no_evidence`. + +## Browser, Office, And OCR Through A3S Use + +A3S Use is an independently released first-use component. Before terminal +takeover, `a3s code` reuses a healthy install or installs the verified release +when networking and automatic setup are allowed. `--offline`, +`A3S_OFFLINE=1`, and `A3S_NO_AUTO_INSTALL=1` remain strict no-mutation +boundaries. Setup failure is non-fatal and remains visible through `/use`. +Runtimes and model assets can also be prepared explicitly: + +```bash +a3s install use --source release +a3s install use/browser + +# Optional OfficeCLI compatibility provider. Native Office is built into Use. +a3s install use/office + +# Install or repair the pinned local PP-OCRv6 models. +a3s install use/ocr +a3s use ocr doctor --json +``` + +When the parent Use binary is ready, Code consumes its versioned +capability registry and watches both generation and content revision. Provider +changes and extension enable, disable, install, or upgrade events then hot-plug +their MCP and verified `SKILL.md` surfaces into the current session. If +first-use setup was disabled or failed, install the parent explicitly and +restart Code once; later capability changes do not require a restart. + +Use `/use` or `/use status` to inspect the binary path/version, registry +convergence, provider readiness, MCP connection/tool count, and verified/loaded +Skills. `/use repair` prints non-destructive repair guidance only. It never +runs an installer or changes extension state. + +The primary coding model intentionally does not receive raw +`mcp__use_*` definitions. It sees a dedicated `use` worker through `task` and +delegates application work to that worker. The worker can see only +`mcp__use_*`; it cannot use shell, workspace, unrelated MCP, or recursive +delegation tools. Managed Skills provide domain guidance but cannot widen this +permission boundary. Ask naturally, for example, “Use Browser to inspect this +page,” “Use Office to update this workbook,” or “Use OCR on this scan.” If a +capability is unavailable, the worker reports the typed failure instead of +falling back to another tool. The default Browser surface includes doctor and a +bounded installer; missing managed Browser files can therefore be prepared +inside the Use worker only after parent TUI confirmation. + +The projected routes are intentionally distinct: + +| Capability | Code surface | Provider behavior | +| ------------- | --------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Browser | `mcp__use_browser__*` | Uses a discovered browser or requests the managed provider through a mutation that requires parent HITL. Closed-world session inspection can run without an extra prompt; navigation, network/open-world reads, input, clicks, and submits require HITL. | +| Native Office | `mcp__use_office__*` | Uses the bounded native Office package kernel and its revision/conflict checks. Reads are closed-world; mutations and destructive saves require HITL. | +| Built-in OCR | `mcp__use_ocr__*` | Runs the pinned `PP-OCRv6_small` detection and recognition models locally through ONNX Runtime. Doctor and extraction are closed-world read-only tools; source bytes never leave the device. | + +MCP behavior metadata is escalation-only. An operation is frictionless only +when it declares `readOnlyHint=true`, `openWorldHint=false`, and is not +destructive. Missing annotations, open-world access, mutation, destruction, or +submit risk asks the parent TUI; a parent denial remains authoritative. Office +mutations are never retried automatically. In particular, +`use.office.outcome_unknown` means the change may already have been applied, so +the worker preserves the evidence and stops. + +Structured MCP output is not flattened away: Code retains output schemas, +`structuredContent`, images, text/blob resources, protocol metadata, and +SHA-256-addressed bounded artifacts for the normal transcript/tool-result path. + +## Browser, Office, And OCR Through A3S Use + +A3S Use is an independently released first-use component. Before terminal +takeover, `a3s code` reuses a healthy installation or installs the verified +release when networking and automatic setup are allowed. `--offline`, +`A3S_OFFLINE=1`, and `A3S_NO_AUTO_INSTALL=1` remain strict zero-network, +zero-receipt boundaries. Setup failure is non-fatal; install the parent +explicitly and restart Code once if first-use preparation was disabled or +failed. + +When Use is ready, Code consumes its versioned registry and gives the initial +MCP projection a separate bounded window. Browser, native Office, OCR, and +installed extension routes are exposed only through a dedicated `use` worker. +The primary model never sees raw `mcp__use_*` tools, and the worker cannot use +the workspace, shell, unrelated MCP servers, or recursive delegation. Managed +Skills provide domain guidance without widening that boundary. + +## Interaction Model + +The main screen is an event-driven transcript. User messages, model text, +reasoning deltas, tool starts, streamed tool output, approvals, subagent +progress, plans, memory events, and final summaries arrive as structured +runtime events and are rendered incrementally. + +Assistant Markdown is source-backed and newline-gated. Stable rows are paced +through an adaptive commit queue, deep or stale queues catch up automatically, +and active tables remain in a replaceable tail until completion. Tail updates +reuse the wrapped transcript prefix; terminal resize still reflows both regions +from raw source without truncating headings, code, URLs, graphemes, or cells. + +| Input | Behavior | +| ---------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Normal prompt | Sends a coding-agent turn. If a turn is busy, the input is queued. | +| `/queue` | Opens the pending follow-up queue with stable selection, Send now, remove, and explicitly confirmed clear actions. | +| `/history` | Fuzzy-searches prompts from the current session and restores the selected text. | +| `/copy` / `/copy transcript` | Copies the latest assistant source Markdown or the complete semantic session. | +| `/export [path]` | Atomically creates a no-clobber Markdown session file inside the workspace. | +| `/tasks` | Opens live delegated-task control with search, progress/output inspection, refresh, and confirmed cancellation. | +| `/permissions` | Searches exact session/project grants, opens canonical arguments, and revokes only after a second matching confirmation. | +| `! ` | Runs a direct shell turn through the same workspace output surface. | +| `? ` | Starts exact-query bootstrap and bounded semantic planning, stages a source-backed report, then optionally compiles a typed claim graph into synthesized or qualified output. | +| `@` | Attaches a workspace file through the file picker path. | +| `/` | Opens the command palette and slash command registry. | +| `Shift+Enter` | Inserts a newline in the input. | +| `Ctrl+O` | Sends the current draft now by cancelling and settling the active turn before consuming it. | +| `Shift+Tab` | Cycles default, plan, and auto modes. | +| `PgUp` / `PgDn` | Scrolls the transcript or active full-screen panel. | +| `Shift+End` | Jumps to the latest transcript output. | +| `Ctrl+T` | Opens the complete live semantic session transcript, including user and assistant messages, plans, every tool lifecycle and full tool output, subagent state, and the current streaming tail. | +| `Ctrl+R` | Opens prompt-history search; repeated presses cycle matching prompts. | +| `Ctrl+B` | Opens or closes delegated-task control without interrupting the parent turn. | +| `Esc` | Interrupts the running turn or closes the active panel. | +| `Ctrl+C` twice | Quits after session persistence runs. | + +The UI keeps long-running work observable. The transcript can show streamed +model text, tool input/output, progress deltas, approvals, runtime view +buttons, dynamic-workflow artifacts, subagent activity, queue entries, memory +events, and context-fill warnings. `Ctrl+T` opens the complete live semantic +transcript, including the current Markdown tail and full tool output. The +`/tasks` panel reads the current session's authoritative task snapshots while +the parent turn keeps streaming. It keeps every running task plus bounded +recent history, refreshes every second, preserves selection across reordering, +searches task metadata, progress, and output, opens full details, and requires +a second matching cancellation action. The standalone `a3s top` command shows +local process activity outside the TUI. + +`/permissions` remains available while the parent turn streams. It separates +ephemeral session grants from project grants stored in +`.a3s/permissions.acl`, preserves selection while filtering, and opens the +authoritative canonical arguments. `X` or Delete must be repeated on the same +row before revocation. Both scopes update the live checker immediately; project +removal then performs an atomic `a3s-acl` rewrite and restores the grant if +persistence fails. Revocation affects future checks only and does not cancel +tools already running. + +Pending follow-ups use Lane priority and FIFO metadata, while each row retains +its submission-time execution mode, attachments, and Plan state. `/queue` +selects rows by stable sequence: Up/Down or the wheel moves selection, Enter or +`S` sends the selected row now, Delete or `D` removes one row, and `C` opens a +separate clear confirmation. Esc closes the modal without changing the +composer draft. During a live turn, Enter on an empty composer sends the queue +head now. + +Prompt recall remains available through Up/Down. For longer sessions, +`/history` or `Ctrl+R` opens a fuzzy-searchable catalog capped at 100 matches. +Results rank by relevance and recency, repeated prompts retain distinct +positions, repeated Ctrl+R cycles matches, and Enter or Tab restores the +selected prompt. Esc closes the panel without changing the existing draft. + +`/copy` selects the current live assistant tail when one exists, then falls +back to the latest committed assistant response. `/copy transcript` uses the +same stable semantic Markdown projection as `/export`: user and assistant +messages, visible tool lifecycle/output, and visible delegated-task results. +Private reasoning, transient UI notices, terminal-width-dependent rows, and +hidden duplicate cells are excluded. Native clipboard delivery is reported +only when verified; otherwise the TUI describes the OSC 52 request and its +64,000-byte UTF-8 payload bound. + +`/export` generates a unique session-and-time filename in the workspace root. +An optional path is interpreted relative to the workspace, may target an +existing directory, and is created atomically with private permissions. Parent +traversal, escaping symlinks, missing parent directories, and existing targets +are rejected instead of overwriting or writing outside the workspace. + +## Code Intelligence + +The agent and TUI `/ide` share one read-only Code +Intelligence runtime. The current language profiles cover Rust and +TypeScript/JavaScript. Install the matching `rust-analyzer` or +`typescript-language-server` executable before starting `a3s code`. A missing +server degrades only that language; it does not disable the editor, ordinary +file tools, or another working language. + +Semantic results always describe files saved on disk. Unsaved editor buffers +are never published to the shared runtime. The UI labels those results as the +"saved version," and saving refreshes the workspace manifest and subsequent +semantic queries. + +Press `:` inside `/ide` to use: + +| Command | Result | +| ------------------------ | -------------------------------------------------------- | +| `:status` | Language state and negotiated capabilities. | +| `:symbols [query]` | Current-file symbols or bounded workspace symbol search. | +| `:definition` | Definitions at the saved-file cursor position. | +| `:declaration` | Declarations at the saved-file cursor position. | +| `:references` | References at the saved-file cursor position. | +| `:implementations` | Implementations at the saved-file cursor position. | +| `:diagnostics` | Diagnostics for the current saved file. | +| `:diagnostics workspace` | Bounded diagnostics across saved workspace files. | + +The agent receives the same capability through `code_symbols`, +`code_navigation`, and `code_diagnostics`. Source reads, text search, and +mutations still go only through `read`, unified `search` in grep mode, `edit`, +and `patch`; Code Intelligence does not create a second mutation path. + +## Cross-Session Context Recall + +When the local `ctx` index is installed and initialized, startup enables two +recall tiers: long-term Memory for curated durable information, and `/ctx` for +raw history across tools and sessions. Use it to recover prior decisions, +commands, failures, and test evidence instead of deriving the same work again. + +| Command | Behavior | +| --------------- | ------------------------------------------------------------------------------------------ | +| `/ctx ` | Search the local session index and display up to eight selectable hits. | +| `/ctx ` | Pull a bounded window around hit n and attach it once to the next message. | +| `/ctx save ` | Save the hit as episodic memory with `ctx_event_id` and `ctx_session_id` provenance links. | + +An attached transcript is control-character stripped, size-bounded, and quoted +as untrusted historical reference. Instructions inside it do not become +current user instructions. A promoted memory carries `source=ctx`, so the +memory view can jump back to its originating event or session. + +## Sessions And Safety + +Core session snapshots auto-save under +`/.a3s/tui/sessions/v1/sessions`; TUI-owned per-session state is +stored under `/.a3s/tui/session-state/v1`. Exiting prints the complete +`a3s code resume ` command and highlights it when color output is +enabled. `a3s code resume` without an id resumes the newest saved session in the +current workspace. Resume restores the selected model and credential source, +effort profile, execution mode (`default`, `plan`, or `auto`), and syntax theme. +`/fork` copies the current transcript into a new session id while keeping the +original. `/fork worktree` also creates an `a3s/fork-` branch in a sibling +`.a3s-worktrees` directory, transfers tracked and untracked workspace content +through a binary patch, and copies the complete session into that isolated +workspace. It excludes TUI persistence from the patch, never changes the real +Git index, and prints the exact command for opening the fork. + +Ordinary user turns retain bounded pre/post Git tree checkpoints. `/rewind` +forks the pre-turn conversation and reverses the last file patch only after a +whole-patch conflict check succeeds. A later edit to a touched file causes a +full refusal instead of an overwrite. `HEAD`, the real Git index, and the +original session remain unchanged. In a non-Git workspace, conversation rewind +still works and reports that file rewind was unavailable. `/clear` starts a +fresh conversation. + +If exit interrupts a durable `/goal`, the goal is saved as paused. The resumed +TUI presents `Resume goal` and `Leave paused`: resume continues the next goal +iteration without changing the restored execution mode, while leave enters the +session with the goal paused so `/goal resume` can continue it later. + +The TUI owns human-in-the-loop approval, but approval is a sandbox-boundary +event rather than a tool-category event: + +| Mode | Quiet execution | Boundary crossing | +| ------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| Default | Workspace file mutations, governed orchestration, and ordinary Bash inside the installed local process sandbox. | Explicit host Bash escalation, Bash when no process sandbox is available, protected control metadata, mutating Git operations, and annotated external side effects enter HITL. | +| Plan | Read-only workspace and web discovery only. | Mutations, Bash, and escalation requests are denied. A completed plan is reviewed before implementation starts as a separate Default turn. | +| Auto | The same bounded workspace mutations and process sandbox as Default. | Auto never opens HITL. Missing-sandbox Bash, host escalation, protected control metadata, and hard policy denials fail closed. | + +Shift+Tab cycles Default, Plan, and Auto. A running or queued turn keeps the +mode captured when it was submitted. Permission and confirmation routing are +snapshotted at run admission; delegated, parallel, Skill, and background +descendants keep that snapshot after the composer advances to another mode. +Tool execution timeouts and confirmation timeouts are tracked separately, so +waiting for a human does not consume the command runtime budget. + +`/permissions` makes remembered grants inspectable and revocable. Session and +project scopes remain distinct, exact arguments are available through Enter, +and a second matching `X` or Delete is required. Project changes use an atomic +`a3s-acl` rewrite; revocation affects future checks only. + +All local filesystem work stays under active workspace services and A3S Code +permission policy. Local chat, file edits, subagents, MCP, memory, asset +drafting, and `DynamicWorkflowRuntime` work without `/login`. Account-backed +runtime tools, hosted asset publishing, trusted view links, and hosted activity +panels are available only when a host explicitly configures and registers them. + +## Tool Runtime + +A3S Code TUI exposes tools through the session registry, not by letting the +model run arbitrary host APIs. Each tool call carries a name, JSON arguments, +streamed output, timeout policy, permission decision, and traceable event id. + +### Local downloads + +In a writable local workspace, the model can call `download` to stream a public +HTTP(S) resource into the repository. The call is shown as a normal modifying +tool event and follows the active mode: Default can pause for HITL, Plan denies +the mutation, and Auto permits only the bounded workspace-safe form. + +The card reports the final workspace path, bytes, sequential or parallel Range +strategy, and optional verified SHA-256 without displaying signed URL query +parameters. Downloads use adjacent temporary files, clean partial data on +cancellation or failure, and atomically expose the result only after all +limits and optional `expected_sha256` verification pass. The complete +parameters and defaults are documented under +[Tools](/guide/tools#binary-safe-local-downloads). + +### Local process sandbox + +Core installs the A3S-owned `a3s-sandbox` Rust backend by default for every +local session. Before terminal takeover, the TUI additionally probes that +backend through `NativeBashSandbox` and enables its deferred handle only after +the probe succeeds. Seatbelt, Bubblewrap namespaces plus seccomp, or +AppContainer plus a Job Object is selected by the host platform. The backend is +linked into the CLI and does not require Node.js, npm, a support payload, or a +global runtime installation. + +There is no silent host fallback. Offline mode and +`A3S_NO_AUTO_INSTALL=1` still leave the native backend untouched. If no +verified boundary is available, the deferred handle remains error-only, +Default asks before running an exact escalated Bash invocation on the host, and +Auto denies Bash. With the native backend, routine +commands inherit an OS-enforced boundary that: + +- denies outbound networking, local listeners, and Unix sockets; +- limits writes to the workspace and a private per-run scratch directory; +- keeps repository, A3S, agent, editor, and tool control metadata read-only; +- blocks reads from common credential stores and removes ambient secrets from + the command environment; +- denies existing `.env*` files at every governed source-tree depth and masks + pre-existing multi-link source files for both reads and writes; +- preserves bounded streaming output, timeout, cancellation, and process-group + cleanup. + +The TUI also enables Core's local workspace credential policy for built-in +file operations. Direct and range reads, writes, edits, patches, and both +manifest-backed and fallback grep enforce the same sensitive-path and +source-hardlink rules. Explicit sensitive targets fail closed and broad grep +filters them without exposing matching content. Ordinary package-store +hardlinks remain readable unless they alias a discovered credential inode. +Read-only Git diff is regenerated only for allowed changed paths; +option-like revisions cannot become Git flags, and displayed remotes omit +embedded HTTP credentials and query tokens. + +Delegated tasks, Skills, and dynamic workflow steps inherit the same sandbox, +permission checker, and parent confirmation boundary. Child-local `Ask` +decisions retain the worker's configured confirmation behavior, but a parent +`Ask` or tool-owned escalation remains governed by the parent provider. A +child-local allow rule or automatic confirmation provider cannot weaken the TUI +boundary. + +The CLI release, Homebrew formula, and standalone self-update preserve the same +verified support tree. npm is a development fallback, not the production +release supply path. + +This local process sandbox is the low-latency boundary for routine repository +commands. A3S Code owns permission and escalation policy; the sandbox provider +owns OS enforcement. A3S Runtime owns provider-neutral Task and Service +lifecycle and placement, not per-command approval. Dependency-heavy, untrusted, +OCI, or stronger-isolation workloads belong on A3S Box or a provider selected +through A3S Runtime. + +| Tool family | TUI behavior | +| ----------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Workspace tools | `read`, `write`, `edit`, `patch`, local `download`, `ls`, unified `search`, `bash`, `git`, `web_fetch`, and `web_search` run through workspace services, path boundaries, timeout handling, sandbox enforcement, and confirmation policy. | +| Structured output | `generate_object` uses `Generating/Generated object` cards while keeping schema-shaped JSON in the same bounded event stream as normal tools. | +| MCP tools | Configured MCP servers are registered as `mcp____` names and use the same approval and output rendering path. | +| PTC scripts | The `program` tool runs sandboxed JavaScript-compatible scripts with a host-provided `ctx` object. Recursive `program`, `dynamic_workflow`, and the removed `parallel_task` alias stay out of the default allow-list; one-item `task` calls are allowed, but direct fan-out is blocked. | +| Delegation | `task` launches one focused child for a single `tasks[]` item or fans out multiple independent items on the native host runtime, preserves input order, emits subagent progress events, and respects `max_parallel_tasks`. | +| Dynamic workflow | `dynamic_workflow` is registered in the TUI because `ultracode` and `?` DeepResearch use it. It records project-local A3S Flow history under `.a3s/workflow` and schedules host fan-out steps with the canonical name `task`. | +| Optional runtime | Host-provided runtime tools are registered only after `/login`. Once present, normal model turns and dynamic workflow PTC steps can call them for hosted batch execution. | +| Dynamic tools | Agent-directory and host-registered tools without a dedicated renderer use bounded `Calling/Called tool(args)` fallback cards. | + +## Effort Profiles + +`/effort` rebuilds the active session with a different depth profile. The +design scales work on three axes: + +- Thinking budget for providers that expose extended thinking. +- Tool-round budget and continuation count for all providers. +- Model-agnostic prompt guidance for rigor, verification, and decomposition. + +| Level | Thinking budget | Tool rounds | Continuations | Parallel tasks | Intended behavior | +| ----------- | --------------: | ----------: | ------------: | -------------: | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `low` | 2,048 | 240 | 4 | 4 | Fast, minimal changes with narrow verification. | +| `medium` | 8,192 | 800 | 8 | 8 | Balanced default behavior without extra depth steering. | +| `high` | 16,384 | 1,200 | 12 | 12 | More deliberate planning, relevant tests, and self-review. | +| `xhigh` | 32,768 | 1,800 | 16 | 16 | Compare alternatives, probe edge cases, and verify thoroughly. | +| `max` | 65,536 | 2,400 | 24 | 24 | Maximum rigor for correctness, adversarial checks, and completeness. | +| `ultracode` | 65,536 | 3,200 | 32 | 32 | Message-gated dynamic workflow mode. Trivial turns stay direct; complex turns may use `dynamic_workflow`, A3S Flow replay, native `task` fan-out, and signed-in `runtime`. | + +All effort levels keep local `task` available, with the TUI session limiting +explicit sibling fan-out through `max_parallel_tasks`. +Runtime-driven automatic delegation is disabled for `low` through `max`; signed-in +Codex models still receive the corresponding native `reasoning.effort` value. +`ultracode` enables automatic delegation, planning, goal tracking, and +dynamic-workflow guidance, but the pre-analysis gate still decides whether a +turn actually needs planning or fan-out. Synthesis-only continuations cannot +start another delegation wave. + +## Dynamic Workflows Vs `/flow` + +A3S Code has two workflow concepts and they are intentionally different: + +| Concept | Surface | Purpose | +| ------------------------ | ------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `DynamicWorkflowRuntime` | Model-visible `dynamic_workflow` tool, used by `ultracode` and `?` DeepResearch | Per-turn dynamic orchestration. A sandboxed JavaScript PTC function returns A3S Flow commands such as `complete`, `fail`, `schedule_step`, or `schedule_steps`; A3S Flow records replayable workflow and step history. | +| Workflow assets | `/flow`, `/flow publish`, `/flow run`, `/flow deploy`, `/flow open`, `/flow logs`, `/flow status` | Durable workflow asset lifecycle. Local DAG JSON files can be reviewed locally and, when a host integration is configured, published with runtime-binding metadata. | + +Dynamic workflow scripts are runtime artifacts, not a separate TypeScript SDK. +They run inside the existing `program` QuickJS sandbox and may call only the +tools that the host allows through `ctx`. + +```javascript +export default async function run(ctx, inputs) { + if (inputs.kind === 'workflow') { + return { + type: 'schedule_steps', + steps: [ + { + step_id: 'inspect', + step_name: 'inspect_workspace', + input: { query: inputs.input.query }, + }, + { + step_id: 'fanout', + step_name: 'task', + input: { + tasks: [ + { + agent: 'explore', + description: 'Find test coverage', + prompt: 'Inspect relevant tests and coverage gaps.', + }, + { + agent: 'review', + description: 'Review risk', + prompt: 'Review the approach for regressions.', + }, + ], + }, + }, + ], + }; + } + + if (inputs.step_name === 'inspect_workspace') { + const hits = await ctx.search(inputs.input.query, { + mode: 'grep', + include: '*.rs', + }); + return { hits }; + } + + return { ok: true }; +} +``` + +PTC is hosted by QuickJS. A script may call `task` with one item, but direct +multi-item fan-out is blocked. When a workflow needs parallel local subagents, +it schedules a Flow step named `task`; the TUI host executes the native +implementation outside QuickJS. If A3S OS registers a runtime tool after +`/login`, dynamic workflow PTC steps may also call `ctx.tool("runtime", ...)`. + +## Optional A3S OS Runtime And Views + +Add an A3S OS endpoint to `config.acl`, then sign in: + +```acl +os = "https://os.example.com" +``` + +```text +/login +``` + +Signed-out behavior is intentionally useful but local: chat, file editing, +workspace tools, local MCP, local asset drafting, memory, `/ctx`, `/kb`, +`task`, `dynamic_workflow`, DeepResearch with validated local +reports, and local loops keep working. + +Signed-in behavior adds A3S OS asset publication, runtime tools, trusted view +links, and asset activity panels without changing the local-first TUI contract. + +The signed-in progressive capability API uses one permission-filtered endpoint +and expands only as needed: + +```text +list → search → describe → execute +modules find op one schema run it +``` + +The model discovers modules, searches operations, and loads the complete schema +for only the operation it will call. An `execute` request with `shaped=true` +can return a trusted `.view` or `viewUrl`, which becomes an inline `Open view` +action in the TUI. + +Login also registers the `runtime` tool. It resolves a tool-kind worker by UUID +or name, submits independent inputs as an A3S OS Function-as-a-Service batch, +streams per-item progress, and returns aggregated results. When signed out, the +tool is absent from the model registry; local files, shell, MCP, `task`, +`dynamic_workflow`, Memory, and `/ctx` remain available. + +| Capability | Signed out | Signed in after `/login` | +| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------- | +| Coding chat and workspace tools | Available with local permission checks and HITL approval. | Available with the same local safety path. | +| Context, memory, and local knowledge | `/ctx`, `/memory`, `/sleep`, and `/kb` use local stores. | Local stores remain available; hosted reports may also return trusted views. | +| Dynamic workflows | `DynamicWorkflowRuntime` can run local Flow-backed orchestration and host-side `task` fan-out. | Normal workflow PTC steps may also call registered runtime tools; DeepResearch fan-out remains local for now. | +| Asset authoring | `/agent`, `/mcp`, `/skill`, `/flow `, and `/okf` can draft and review local assets. | Publish, deploy, run, open, logs, status, list, and activity commands can use A3S OS. | +| Runtime views | Validated local DeepResearch HTML is served through a loopback-only viewer and becomes an inline `Open view` action; OS `.view` and `viewUrl` responses are unavailable. | Local report views remain available, and OS `.view` and `viewUrl` responses also become inline `Open view` actions. | +| Runtime activity | The standalone `a3s top` command observes local processes. | Asset `activity` commands can inspect hosted jobs, runs, invocations, indexing, and workflow activity. | + +## Command Reference + +These commands are available outside the asset-specific flows: + +| Command | Capability | +| -------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `/help` | Open the full command guide with slash commands, command forms, input modes, keys, panels, and resume help. | +| `/model` | Switch among configured ACL models and signed-in account-backed model tabs when available. | +| `/effort` | Change the active effort profile from `low` to `ultracode`. | +| `/init` | Analyze the workspace and generate an `AGENTS.md` instruction file. | +| `/config` | Edit the active ACL config in the built-in editor. | +| `/queue` | Inspect pending follow-ups, send the exact selected row now, remove one row, or explicitly confirm clearing all pending rows. | +| `/history` | Fuzzy-search prompts from the current session and restore a selected prompt without changing the current draft on cancel. | +| `/copy` / `/copy transcript` | Copy the latest assistant source Markdown or the complete semantic session with explicit native/OSC 52 delivery feedback. | +| `/export [path]` | Atomically create a private, no-clobber Markdown session snapshot inside the workspace. | +| `/tasks` | Inspect running and recent delegated tasks, search progress/output, open full details, refresh, or safely cancel a running task. | +| `/permissions` | Search exact session/project grants, inspect canonical arguments, and revoke with a second matching confirmation; project changes atomically update `.a3s/permissions.acl`. | +| `/use` / `/use status` / `/use repair` | Inspect live A3S Use capabilities or print explicit, non-mutating repair guidance. | +| `/login` / `/logout` | Sign in or out of the configured account integration; login can register host capabilities and runtime tools. | +| `Open view` | Reopen the latest validated local report or captured trusted runtime view when the inline action is shown. | +| `/ide` | Open the workspace file browser and editor. | +| `/memory` | Browse durable memory as an event/entity graph with tiers, aliases, relations, conflicts, and forget candidates. | +| `/ctx ` | Search past ctx-indexed sessions. | +| `/ctx ` | Attach a previous search result to the next message. | +| `/ctx save ` | Promote a previous session hit into durable memory. | +| `/sleep` | Consolidate the day's work into memory. | +| `/kb` | Manage the local personal knowledge base. | +| `/goal ` | Set a persistent goal for the current session or active asset mode. | +| `/goal resume` | Continue a durable goal that was left paused during session resume. | +| `/compact` | Summarize and shrink the active conversation context. | +| `/clear` | Start a fresh conversation in the current session surface. | +| `/fork` / `/fork session` | Branch the current transcript into a new session id. | +| `/fork worktree` | Create an isolated branch/worktree and copy the current workspace plus complete session into it. | +| `/rewind` | Fork the conversation before the last completed user turn and reverse its file patch only when conflict checks pass. | +| `/relay` | Open the session/background-work dashboard. Press `/` to filter the bounded per-source catalog by task, status, model, session id, or source path; `Space` toggles a task peek. Stable identities preserve a still-present selected session across manual `R` refreshes and automatic 15-second refreshes. | +| `/auto` | Switch future turns into non-interactive Auto mode; hard denials fail directly without opening HITL. | +| `/plugin` / `/reload` | Manage and hot-reload skills/plugins. | +| `/theme` | Cycle syntax highlighting themes. | +| `/update` | Upgrade the CLI and restart back into the saved session. | +| `/exit` | Quit `a3s code` after session persistence runs. | + +Asset command families include `/agent`, `/mcp`, `/skill`, `/flow`, `/okf`, +and `/loop`. Their publish, deploy, run, open, logs, status, list, activity, +clone, review, and local development forms are scoped to the selected asset +family. + +## Headless Smoke Mode + +Set `A3S_CODE_TUI_SMOKE=1` to exercise the same `AgentSession::stream()` +integration without taking over the terminal. This is useful for release smoke +checks of the configured model/session path. + +```bash +A3S_CODE_TUI_SMOKE=1 a3s code +``` + +## Related Pages + +1. [A3S CLI Code TUI](https://a3s-lab.github.io/a3s/docs/cli/code-tui) +2. [Commands](/guide/commands) +3. [Tools](/guide/tools) +4. [Memory](/guide/memory) +5. [Sessions](/guide/sessions) +6. [Security](/guide/security) diff --git a/website/docs/v8.5.1/en/guide/verification.mdx b/website/docs/v8.5.1/en/guide/verification.mdx new file mode 100644 index 00000000..9ebfde9d --- /dev/null +++ b/website/docs/v8.5.1/en/guide/verification.mdx @@ -0,0 +1,390 @@ +--- +title: 'Verification' +description: "Prove a turn is done with verification commands and reports instead of trusting the model's claim" +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Verification + +The harness treats "done" as something that must be **proven**, not merely +claimed. When the model says a task is complete, that assertion is worth nothing +on its own. Verification turns the claim into evidence: you declare commands +that _must_ succeed, the runtime executes them, and the result carries a report +you can inspect, gate on, or surface to a user. + +Verification is session-scoped. The Rust core runs each command, records its +exit status and output, and rolls every report up into a single summary that +travels alongside the turn result. + +Product UIs can present a delivery summary first, followed by the command output +and file-level changes used to support it. + +## Running Verification Commands + +A verification command is a small, named check: an `id`, a `kind`, a +human-readable `description`, and the `command` to run. Mark a check `required` +when a failure should be treated as a hard failure rather than a warning. + + + + +```rust +use a3s_code_core::verification::VerificationCommand; + +let commands = vec![ + VerificationCommand::required( + "build", + "build", + "Project compiles", + "cargo build --all-features", + ) + .with_timeout_ms(120_000), + VerificationCommand::required( + "tests", + "test", + "Unit tests pass", + "cargo test", + ), +]; +let report = session + .verify_commands("release-readiness", &commands) + .await?; +println!("{report:#?}"); +``` + + + + +```ts +const report = await session.verifyCommands('release-readiness', [ + { + id: 'build', + kind: 'build', + description: 'Project compiles', + command: 'cargo build --all-features', + required: true, + timeoutMs: 120000, + }, + { + id: 'tests', + kind: 'test', + description: 'Unit tests pass', + command: 'cargo test', + required: true, + }, +]); + +console.log(report); +``` + + + + +```python +report = session.verify_commands('release-readiness', [ + { + "id": "build", + "kind": "build", + "description": "Project compiles", + "command": "cargo build --all-features", + "required": True, + "timeout_ms": 120000, + }, + { + "id": "tests", + "kind": "test", + "description": "Unit tests pass", + "command": "cargo test", + "required": True, + }, +]) + +print(report) +``` + + + + +```go +report, err := session.VerifyCommands(ctx, "release-readiness", []code.VerificationCommand{ + { + ID: "build", + Kind: "build", + Description: "Project compiles", + Command: "cargo build --all-features", + Required: true, + TimeoutMS: code.Ptr(uint64(120000)), + }, + { + ID: "tests", + Kind: "test", + Description: "Unit tests pass", + Command: "cargo test", + Required: true, + }, +}) +if err != nil { + return err +} +fmt.Println(report) +``` + + + + +The `subject` (here `release-readiness`) labels the batch so multiple +verification passes within one session stay distinct in the reports. + +## Reading The Post-Turn Summary + +Every turn's `send()` result also carries read-only verification fields, so you +can gate on the outcome without issuing a separate verification call. Use these +to decide whether the turn actually accomplished what it claimed. + + + + +```rust +let result = session + .send("Apply the fix and run the checks", None) + .await?; +let summary = result.verification_summary(); + +println!("{:?}", summary.status); +println!("{}", summary.pending_required_check_count); +println!("{}", summary.failed_check_count); +println!("{}", summary.report_count); +println!("{}", result.verification_summary_text()); + +if summary.failed_check_count > 0 { + return Err(a3s_code_core::CodeError::Session( + "turn reported done but verification failed".to_string(), + )); +} +``` + + + + +```ts +const result = await session.send('Apply the fix and run the checks'); + +console.log(result.verificationStatus); +console.log(result.pendingVerificationCount); +console.log(result.failedVerificationCount); +console.log(result.verificationReportCount); +console.log(result.verificationSummaryText); + +if (result.failedVerificationCount > 0) { + throw new Error('Turn reported done but verification failed'); +} +``` + + + + +```python +result = session.send('Apply the fix and run the checks') + +print(result.verification_status) +print(result.pending_verification_count) +print(result.failed_verification_count) +print(result.verification_report_count) +print(result.verification_summary_text) + +if result.failed_verification_count > 0: + raise RuntimeError('Turn reported done but verification failed') +``` + + + + +```go +result, err := session.Run(ctx, "Apply the fix and run the checks") +if err != nil { + return err +} + +summary := result.VerificationSummary +fmt.Println(summary.Status) +fmt.Println(summary.PendingRequiredCheckCount) +fmt.Println(summary.FailedCheckCount) +fmt.Println(summary.ReportCount) +fmt.Println(result.VerificationSummaryText) + +if summary.FailedCheckCount > 0 { + return errors.New("turn reported done but verification failed") +} +``` + + + + +## Inspecting Reports And Summaries + +Beyond the per-turn fields, the session exposes the full set of reports, a +structured summary, the available presets, and a human-readable digest. The +digest is the quickest way to show a person _why_ a turn passed or failed. + + + + +```rust +let reports = session.verification_reports(); +let summary = session.verification_summary(); +let presets = session.verification_presets(); +let text = session.verification_summary_text(); + +println!( + "{} reports, status {:?}, {} presets", + reports.len(), + summary.status, + presets.len() +); +println!("{text}"); +``` + + + + +```ts +import { formatVerificationSummary } from '@a3s-lab/code'; + +const reports = session.verificationReports(); +const summary = session.verificationSummary(); +const presets = session.verificationPresets(); + +// Either the session helper or the standalone formatter yields readable text. +console.log(session.verificationSummaryText()); +console.log(formatVerificationSummary(summary)); +``` + + + + +```python +reports = session.verification_reports() +summary = session.verification_summary() +presets = session.verification_presets() + +# The session helper returns a ready-to-print human-readable digest. +print(session.verification_summary_text()) +``` + + + + +```go +reports, err := session.VerificationReports(ctx) +if err != nil { + return err +} +summary, err := session.VerificationSummary(ctx) +if err != nil { + return err +} +presets, err := session.VerificationPresets(ctx) +if err != nil { + return err +} +text, err := session.VerificationSummaryText(ctx) +if err != nil { + return err +} + +fmt.Println(len(reports), summary.Status, len(presets)) +fmt.Println(text) +``` + + + + +`verificationPresets()` returns workspace-aware check templates inferred from +files such as `Cargo.toml`, `package.json`, `pyproject.toml`, and `go.mod`. +Treat them as starting points: review the commands, timeouts, and required +flags for the project before gating releases or user-visible automation. + +## How A3S Code Itself Is Qualified + +Turn verification answers whether one agent task produced its claimed result. +Repository qualification answers a different question: whether every public +A3S Code capability still satisfies its contract across Core, SDKs, resource +limits, and supported deployment surfaces. A green build alone cannot answer +that question. + +The repository therefore separates four evidence classes: + +| Evidence class | What it proves | What it does not prove | +| --------------------------------- | --------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------- | +| Deterministic correctness | Activation, successful behavior, invalid input, permissions, cancellation, lifecycle, ordering, and cleanup against fixed oracles | Real provider, browser, or object-store availability | +| Deterministic resource gates | Bounds on calls, retries, records, bytes, queues, candidates, tool rounds, and retained state | Wall-clock latency on every machine | +| Release performance qualification | Release-build p50, p95, maximum, resource accounting, workload parameters, and machine metadata for stable local work | Remote model or public-search latency | +| External qualification | Compatibility with a named live model, browser, collector, or storage service under recorded conditions | Hermetic reproducibility or a universal performance claim | + +The capability ledger maps all 27 advertised product areas to executable +evidence and keeps any unresolved gap visible. CI rejects a capability-map +change that is not reflected in that ledger. Node.js and Python gates build and +load their native modules before exercising the public wrappers; compile-only +Rust checks are not counted as SDK runtime evidence. Go runs through its +versioned bridge with the race detector. + +Performance checks distinguish work amplification from timing. Ordinary CI +gates deterministic ceilings such as provider requests, vector bytes, scratch +space, retries, and post-close retention. The dedicated release-profile +workflow uses warmups and repeated samples, emits machine-readable JSON, and +retains it as a CI artifact. Network-dependent DeepSeek and browser timings are +reported separately, because combining them with local execution would make a +regression indistinguishable from provider or network variance. + +The v8.5.1 live-model matrix uses every model declared by the qualification +ACL. It checks model-selected Tool calls and Hook rewrites, an evidence-gated +multi-file coding task, automatic and explicit SubAgents, Skill discovery and +execution, concurrent PTC reads, persisted A3S Flow replay, and public +steer/interrupt behavior. Deterministic fixtures independently cover the same +control paths, including stale and idempotent receipts, denial, budgets, +cancellation, and cleanup. + +### Latest controlled qualification + +The 2026-08-18 release-profile run on four logical x86-64 Linux CPUs passed all +six reports: + +| Profile | Fixed workload | Observed p95 | Objective | +| ------------------- | -------------------------------------------- | ------------------------------: | ---------------: | +| Agent convergence | Four completion, guard, and recovery cases | 4/4 cases | 4/4 cases | +| Workspace Retrieval | 25,000 × 384 exact / deterministic hybrid | 15.590 / 38.506 ms | ≤ 30 / 100 ms | +| Flow / State Graph | 1,000-step projection / 11,008-record replay | 130.067 / 125.526 ms | each ≤ 2,000 ms | +| Code Intelligence | 5,000 files; cold / warm workspace symbols | 754.397 ms cold / 0.519 ms warm | ≤ 5,000 / 250 ms | +| Context / memory | 25,000 context inputs / 2,500 memories | 136.740 / 0.123 ms | ≤ 500 / 250 ms | +| File persistence | 1,272,624-byte synchronized save / load | 338.887 / 1.028 ms | ≤ 1,000 / 500 ms | + +Resource gates also passed for request amplification, vector and rerank bytes, +RSS deltas, serialized graph bytes, process cleanup, overwrite without +accumulation, and delete cleanup. Companion hermetic CI completed a MinIO +roundtrip, the production Chrome/CDP and Google-parser path against controlled +HTTPS, and exact service/span receipt by a local OpenTelemetry Collector. + +See the +[performance qualification record](https://github.com/A3S-Lab/Code/blob/main/manual/PERFORMANCE_QUALIFICATION.md) +for p50/p95/max values, inclusion rules, machine metadata, resource counts, +workflow links, and Artifact SHA-256 digests. These are regression ceilings for +the locked profiles, not universal hardware or remote-service SLAs. + +See the +[capability verification and performance contract](https://github.com/A3S-Lab/Code/blob/main/manual/CAPABILITY_VERIFICATION.md) +for the current evidence ledger, gap closure, external boundaries, +qualification commands, and the completion rule. A repository-wide claim is +not complete while that ledger contains an unresolved Code-owned gap. + +## Why This Matters + +Without verification, an agent run ends on the model's word. With it, the run +ends on observable evidence: a build that compiled, a test suite that passed, a +linter that stayed quiet. The summary text gives you the audit trail; the +counts on the result let you fail closed in automation. + +## Related + +- [Telemetry](/guide/telemetry) — inspect trace events and verification reports as runtime evidence. +- [Limits](/guide/limits) — bound how much work a turn can do before verification runs. diff --git a/website/docs/v8.5.1/en/guide/workspace-backends.mdx b/website/docs/v8.5.1/en/guide/workspace-backends.mdx new file mode 100644 index 00000000..067412f4 --- /dev/null +++ b/website/docs/v8.5.1/en/guide/workspace-backends.mdx @@ -0,0 +1,541 @@ +--- +title: 'Workspace Backends' +description: 'Run A3S Code sessions against local files, S3-compatible object storage, and remote git services.' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Workspace Backends + +Workspace backends decide where built-in workspace tools read and write files. +The default is the local filesystem rooted at the session workspace. All four +SDKs expose configuration for local files, S3-compatible object storage, and an +optional HTTP/JSON remote git provider. Go uses value configurations where +Node.js and Python use backend objects. + +Use this surface when the host owns workspace placement: local development, +browser or container workspaces, object-storage workspaces, or managed sessions. + +## Capability Matrix + +| Backend | File tools | Search tools | Shell and local git | +| ---------------------------------- | -------------------------------------------------- | --------------------------------------------- | ----------------------------------- | +| Default local workspace | `read`, `write`, `edit`, `patch`, `download`, `ls` | `search` (`grep`, `glob`, `bm25` modes) | `bash`, `git` | +| `LocalWorkspaceBackend` | Same as default local workspace | Same as default local workspace | Same as default local workspace | +| `S3WorkspaceBackend` | `read`, `write`, `edit`, `patch`, `ls` | Optional `search` when `searchEnabled` is set | Not registered | +| `S3WorkspaceBackend` + `remoteGit` | S3 file tools | Optional degraded S3 search | `git` through remote git, no `bash` | + +Object storage cannot service local processes. Do not promise shell execution on +an S3 workspace unless the host provides a separate sandbox through MCP or +A3S Box. + +## Local Backend + +The explicit local backend is useful when host code wants one option surface for +both local and remote sessions. + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, WorkspaceServices}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let backend = WorkspaceServices::local("/repo"); + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_workspace_backend(backend)) + .build() + .await?; + + println!("{}", session.read_file("Cargo.toml").await?); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, LocalWorkspaceBackend } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + workspaceBackend: new LocalWorkspaceBackend('/repo'), +}); +``` + + + + +```python +from a3s_code import Agent, LocalWorkspaceBackend, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = LocalWorkspaceBackend("/repo") +session = agent.session("/repo", opts) +``` + + + + +The workspace path already selects the default local backend. Use +`WorkspaceBackendConfig` when one option shape must cover local and remote +sessions: + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "local", + Root: "/repo", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + content, err := session.ReadFile(ctx, "go.mod", nil) + if err != nil { + log.Fatal(err) + } + fmt.Println(content) +} +``` + + + + +### Local download sink + +A writable local backend also registers `download`. The tool accepts a public +HTTP(S) `url` and writes only to a workspace-relative `file_path`; omitting the +path produces a sanitized inferred filename. It is absent from S3 and other +non-local backends because atomic adjacent temporary files, cancellation +cleanup, symlink checks, and final promotion require local filesystem +semantics. + +`overwrite` defaults to `false`. Transfers default to a 512 MiB `max_bytes` +limit and a 300-second `timeout`, with hard limits of 8 GiB and 3600 seconds. +Range concurrency is adaptive or explicitly bounded to 1–4 connections, and +`expected_sha256` can require a 64-character hexadecimal digest before the +destination is promoted. See [Tools](/guide/tools#binary-safe-local-downloads) +for transport and metadata details. + +## S3 Backend + +`S3WorkspaceBackend` points the built-in file tools at any S3-compatible service, +including AWS S3, MinIO, RustFS, Cloudflare R2, and Backblaze B2. + + + + +The Rust crate must be built with its `s3` feature. + +```rust +use std::env; + +use a3s_code_core::{Agent, S3BackendConfig, SessionOptions, WorkspaceServices}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let bucket = env::var("WORKSPACE_S3_BUCKET") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let access_key = env::var("S3_ACCESS_KEY_ID") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let secret_key = env::var("S3_SECRET_ACCESS_KEY") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + + let mut config = + S3BackendConfig::new(bucket, "sessions/example", access_key, secret_key) + .region(env::var("WORKSPACE_S3_REGION").unwrap_or_else(|_| "us-east-1".into())) + .force_path_style(true) + .enable_search(false); + if let Ok(endpoint) = env::var("WORKSPACE_S3_ENDPOINT") { + config = config.endpoint(endpoint); + } + + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("s3://workspace-bucket/sessions/example") + .options( + SessionOptions::new().with_workspace_backend(WorkspaceServices::s3(config)), + ) + .build() + .await?; + + println!("{}", session.read_file("README.md").await?); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, S3WorkspaceBackend } from '@a3s-lab/code'; + +const backend = new S3WorkspaceBackend({ + endpoint: process.env.WORKSPACE_S3_ENDPOINT, + region: process.env.WORKSPACE_S3_REGION ?? 'us-east-1', + accessKeyId: process.env.S3_ACCESS_KEY_ID!, + secretAccessKey: process.env.S3_SECRET_ACCESS_KEY!, + bucket: process.env.WORKSPACE_S3_BUCKET!, + prefix: 'sessions/example', + forcePathStyle: true, + searchEnabled: false, +}); + +const agent = await Agent.create('agent.acl'); +const session = agent.session('s3://workspace-bucket/sessions/example', { + workspaceBackend: backend, +}); +``` + + + + +```python +import os + +from a3s_code import Agent, S3WorkspaceBackend, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = S3WorkspaceBackend( + bucket=os.environ["WORKSPACE_S3_BUCKET"], + prefix="sessions/example", + access_key_id=os.environ["S3_ACCESS_KEY_ID"], + secret_access_key=os.environ["S3_SECRET_ACCESS_KEY"], + endpoint=os.environ.get("WORKSPACE_S3_ENDPOINT"), + region=os.environ.get("WORKSPACE_S3_REGION", "us-east-1"), + force_path_style=True, + search_enabled=False, +) +session = agent.session("s3://workspace-bucket/sessions/example", opts) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "os" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + forcePathStyle := true + searchEnabled := false + session, err := agent.Session( + ctx, + "s3://workspace-bucket/sessions/example", + &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "s3", + S3: &code.S3BackendConfig{ + Endpoint: os.Getenv("WORKSPACE_S3_ENDPOINT"), + Region: os.Getenv("WORKSPACE_S3_REGION"), + AccessKeyID: os.Getenv("S3_ACCESS_KEY_ID"), + SecretAccessKey: os.Getenv("S3_SECRET_ACCESS_KEY"), + Bucket: os.Getenv("WORKSPACE_S3_BUCKET"), + Prefix: "sessions/example", + ForcePathStyle: &forcePathStyle, + SearchEnabled: &searchEnabled, + }, + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + content, err := session.ReadFile(ctx, "README.md", nil) + if err != nil { + log.Fatal(err) + } + fmt.Println(content) +} +``` + + + + +S3 search is intentionally opt-in. When enabled, `grep` / `glob` degrade to +object listing and bounded downloads. Configure `maxObjectsScanned`, +`maxGrepBytesPerObject`, and `searchConcurrency` when the bucket can be large or +the endpoint rate-limits. + +### S3 Options + +| Node.js option | Python option | Required | Purpose | +| ----------------------- | --------------------------- | -------- | ------------------------------------------------------------------------------- | +| `bucket` | `bucket` | yes | S3 bucket that stores the workspace objects. | +| `prefix` | `prefix` | yes | Logical workspace root inside the bucket; use `""` for the bucket root. | +| `accessKeyId` | `access_key_id` | yes | Access key id, normally read from the host environment. | +| `secretAccessKey` | `secret_access_key` | yes | Secret access key, normally read from the host environment or a secret manager. | +| `endpoint` | `endpoint` | no | Custom S3-compatible endpoint; omit for AWS S3 defaults. | +| `region` | `region` | no | Region, defaulting to `us-east-1` when omitted. | +| `sessionToken` | `session_token` | no | STS session token when temporary credentials are used. | +| `forcePathStyle` | `force_path_style` | no | Set `true` for MinIO, RustFS, and most non-AWS endpoints. | +| `maxReadBytes` | `max_read_bytes` | no | Per-read size ceiling; defaults to 10 MiB. | +| `searchEnabled` | `search_enabled` | no | Enables degraded S3 `grep` / `glob`; defaults to false. | +| `maxObjectsScanned` | `max_objects_scanned` | no | Per-search object scan cap; used only when search is enabled. | +| `maxGrepBytesPerObject` | `max_grep_bytes_per_object` | no | Per-object download cap for `grep`; used only when search is enabled. | +| `searchConcurrency` | `search_concurrency` | no | Concurrent object downloads during `grep`; used only when search is enabled. | + +Go exposes the same fields on `S3BackendConfig`: `Bucket`, `Prefix`, +`AccessKeyID`, `SecretAccessKey`, `Endpoint`, `Region`, `SessionToken`, +`ForcePathStyle`, `MaxReadBytes`, `SearchEnabled`, `MaxObjectsScanned`, +`MaxGrepBytesPerObject`, and `SearchConcurrency`. + +## Remote Git + +`remoteGit` attaches an HTTP/JSON git provider on top of `workspaceBackend`. It is +designed for non-local workspaces where the built-in `git` tool cannot use a +local `.git` directory. + +`remoteGit` requires `workspaceBackend`; passing it alone is rejected. + + + + +```rust +use std::env; + +use a3s_code_core::{ + Agent, RemoteGitBackendConfig, S3BackendConfig, SessionOptions, WorkspaceServices, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let bucket = env::var("WORKSPACE_S3_BUCKET") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let access_key = env::var("S3_ACCESS_KEY_ID") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let secret_key = env::var("S3_SECRET_ACCESS_KEY") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let base_url = env::var("REMOTE_GIT_BASE_URL") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let token = env::var("REMOTE_GIT_TOKEN") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + + let storage = WorkspaceServices::s3(S3BackendConfig::new( + bucket, + "sessions/example", + access_key, + secret_key, + )); + let backend = storage.with_remote_git( + RemoteGitBackendConfig::new(base_url, "sessions/example").bearer_token(token), + )?; + + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("s3://workspace-bucket/sessions/example") + .options(SessionOptions::new().with_workspace_backend(backend)) + .build() + .await?; + + let status = session + .tool("git", serde_json::json!({ "command": "status" })) + .await?; + println!("{}", status.output); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, S3WorkspaceBackend } from '@a3s-lab/code'; + +const backend = new S3WorkspaceBackend({ + endpoint: process.env.WORKSPACE_S3_ENDPOINT, + region: process.env.WORKSPACE_S3_REGION ?? 'us-east-1', + accessKeyId: process.env.S3_ACCESS_KEY_ID!, + secretAccessKey: process.env.S3_SECRET_ACCESS_KEY!, + bucket: process.env.WORKSPACE_S3_BUCKET!, + prefix: 'sessions/example', + forcePathStyle: true, +}); + +const agent = await Agent.create('agent.acl'); +const session = agent.session('s3://workspace-bucket/sessions/example', { + workspaceBackend: backend, + remoteGit: { + baseUrl: process.env.REMOTE_GIT_BASE_URL!, + repoId: 'sessions/example', + bearerToken: process.env.REMOTE_GIT_TOKEN, + }, +}); +``` + + + + +```python +import os + +from a3s_code import ( + Agent, + RemoteGitBackendConfig, + S3WorkspaceBackend, + SessionOptions, +) + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = S3WorkspaceBackend( + bucket=os.environ["WORKSPACE_S3_BUCKET"], + prefix="sessions/example", + access_key_id=os.environ["S3_ACCESS_KEY_ID"], + secret_access_key=os.environ["S3_SECRET_ACCESS_KEY"], + endpoint=os.environ.get("WORKSPACE_S3_ENDPOINT"), + region=os.environ.get("WORKSPACE_S3_REGION", "us-east-1"), + force_path_style=True, +) +opts.remote_git = RemoteGitBackendConfig( + base_url=os.environ["REMOTE_GIT_BASE_URL"], + repo_id="sessions/example", + bearer_token=os.environ["REMOTE_GIT_TOKEN"], +) +session = agent.session("s3://workspace-bucket/sessions/example", opts) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "os" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + forcePathStyle := true + session, err := agent.Session( + ctx, + "s3://workspace-bucket/sessions/example", + &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "s3", + S3: &code.S3BackendConfig{ + Endpoint: os.Getenv("WORKSPACE_S3_ENDPOINT"), + Region: os.Getenv("WORKSPACE_S3_REGION"), + AccessKeyID: os.Getenv("S3_ACCESS_KEY_ID"), + SecretAccessKey: os.Getenv("S3_SECRET_ACCESS_KEY"), + Bucket: os.Getenv("WORKSPACE_S3_BUCKET"), + Prefix: "sessions/example", + ForcePathStyle: &forcePathStyle, + }, + }, + RemoteGit: &code.RemoteGitBackendConfig{ + BaseURL: os.Getenv("REMOTE_GIT_BASE_URL"), + RepoID: "sessions/example", + BearerToken: os.Getenv("REMOTE_GIT_TOKEN"), + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + status, err := session.Git(ctx, code.GitOptions{Command: "status"}) + if err != nil { + log.Fatal(err) + } + fmt.Println(status.Output) +} +``` + + + + +Keep remote git credentials out of `agent.acl` and agent directories. Inject them +from the host environment or secret manager. + +### Remote Git Options + +| Node.js option | Python option | Required | Purpose | +| ------------------ | -------------------- | ---------- | -------------------------------------------------------------------------------------- | +| `baseUrl` | `base_url` | yes | Remote git service base URL, without a trailing slash. | +| `repoId` | `repo_id` | yes | Opaque repository id negotiated with the remote git service. | +| `bearerToken` | `bearer_token` | production | Bearer credential for the remote git service; omit only in trusted development setups. | +| `clientCertPem` | `client_cert_pem` | no | mTLS client certificate path; must be paired with the client key. | +| `clientKeyPem` | `client_key_pem` | no | mTLS client key path; must be paired with the certificate. | +| `requestTimeoutMs` | `request_timeout_ms` | no | Per-call HTTP timeout in milliseconds; defaults to 30000. | +| `maxDiffBytes` | `max_diff_bytes` | no | Client-side cap on `diff` response bytes; defaults to 1 MiB. | +| `maxLogEntries` | `max_log_entries` | no | Client-side cap on `log` entries; defaults to 200. | + +Go exposes the same fields on `RemoteGitBackendConfig`: `BaseURL`, `RepoID`, +`BearerToken`, `ClientCertPEM`, `ClientKeyPEM`, `RequestTimeoutMS`, +`MaxDiffBytes`, and `MaxLogEntries`. + +## Choosing A Backend + +- Use the default local workspace for normal developer machines and CI checkouts. +- Use `LocalWorkspaceBackend` when your host always passes a typed backend object. +- Use `S3WorkspaceBackend` when workspace state must live in object storage. +- Add `remoteGit` when a non-local workspace still needs the built-in `git` tool. +- Use MCP or A3S Box for shell-like execution that cannot run inside the + selected workspace backend. diff --git a/website/docs/v8.5.1/en/index.mdx b/website/docs/v8.5.1/en/index.mdx new file mode 100644 index 00000000..863ffb70 --- /dev/null +++ b/website/docs/v8.5.1/en/index.mdx @@ -0,0 +1,7 @@ +--- +pageType: home +title: A3S Code +description: A governed coding-agent runtime with asynchronous workspace retrieval, model-bound evidence, event streaming, and recovery. Available for Rust, Node.js, Python, and Go. +sidebar: false +outline: false +--- diff --git a/website/docs/v8.5.1/zh/_meta.json b/website/docs/v8.5.1/zh/_meta.json new file mode 100644 index 00000000..aa467abc --- /dev/null +++ b/website/docs/v8.5.1/zh/_meta.json @@ -0,0 +1,12 @@ +[ + { + "type": "dir", + "name": "guide", + "label": "文档" + }, + { + "type": "dir", + "name": "api", + "label": "API" + } +] diff --git a/website/docs/v8.5.1/zh/_nav.json b/website/docs/v8.5.1/zh/_nav.json new file mode 100644 index 00000000..9690bb3f --- /dev/null +++ b/website/docs/v8.5.1/zh/_nav.json @@ -0,0 +1,42 @@ +[ + { + "text": "文档", + "link": "/guide/", + "activeMatch": "^/guide/(?!examples/)" + }, + { + "text": "示例", + "link": "/guide/examples/", + "activeMatch": "^/guide/examples/" + }, + { + "text": "API", + "link": "/api/", + "activeMatch": "^/api/" + }, + { + "text": "资源", + "items": [ + { + "text": "GitHub", + "link": "https://github.com/A3S-Lab/Code" + }, + { + "text": "更新日志", + "link": "https://github.com/A3S-Lab/Code/blob/main/CHANGELOG.md" + }, + { + "text": "Rust API", + "link": "https://docs.rs/a3s-code-core" + }, + { + "text": "npm", + "link": "https://www.npmjs.com/package/@a3s-lab/code" + }, + { + "text": "PyPI", + "link": "https://pypi.org/project/a3s-code/" + } + ] + } +] diff --git a/website/docs/v8.5.1/zh/api/index.mdx b/website/docs/v8.5.1/zh/api/index.mdx new file mode 100644 index 00000000..dbb7bee5 --- /dev/null +++ b/website/docs/v8.5.1/zh/api/index.mdx @@ -0,0 +1,306 @@ +--- +title: SDK 与 API +description: 安装 A3S Code 的 Rust、Node.js、Python 和 Go SDK,并找到对应 API 文档。 +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# SDK 与 API + +A3S Code 提供 Rust、Node.js、Python 和 Go SDK。想直接使用终端应用,则安装 +`a3s` CLI。版本号与发布状态以对应 Registry 和 +[GitHub Releases](https://github.com/A3S-Lab/Code/releases) 为准。 + +| 入口 | 包或命令 | 文档 | 适用场景 | +| -------- | ----------------------------------- | -------------------------------------------------- | ------------------------------------- | +| Terminal | `a3s code` | [A3S CLI](https://github.com/A3S-Lab/a3s) | 直接在终端运行编码 Agent | +| Rust | `a3s-code-core` | [docs.rs](https://docs.rs/a3s-code-core) | 使用完整 Runtime API 或实现扩展 Trait | +| Node.js | `@a3s-lab/code` | [npm](https://www.npmjs.com/package/@a3s-lab/code) | 在 Node.js 应用中订阅异步事件流 | +| Python | `a3s-code` | [PyPI](https://pypi.org/project/a3s-code/) | 在 Python 中使用同步或异步 API | +| Go | `github.com/A3S-Lab/Code/sdk/go/v8` | [见下方安装说明](#go-module-与桥接程序) | 通过纯 Go API 使用原生 Runtime | + +## 安装 + +```bash +# Rust +cargo add a3s-code-core + +# Node.js +npm install @a3s-lab/code + +# Python +python -m pip install a3s-code + +# Go +go get github.com/A3S-Lab/Code/sdk/go/v8 +``` + +## Python Wheel 平台(v8.5.1) + +PyPI 上的 `a3s-code` 是一个纯 Python Bootstrap。首次导入时,它会从 v8.5.1 +GitHub Release 下载匹配的 Native Wheel,并根据 SHA-256 Manifest 校验。Native +Wheel 使用 CPython 3.10 Stable ABI(`cp310-abi3`),同一个 Asset 支持 CPython +3.10 至 3.14: + +| 主机 | Wheel Platform Tag | 基线 | 内置浏览器 | +| ------------------- | ------------------------ | ----------- | ---------- | +| Apple Silicon macOS | `macosx_11_0_arm64` | macOS 11+ | Moli arm64 | +| Intel macOS | `macosx_12_0_x86_64` | macOS 12+ | Moli x64 | +| Linux x86_64 | `manylinux_2_28_x86_64` | glibc 2.28+ | Moli x64 | +| Linux arm64 | `manylinux_2_39_aarch64` | glibc 2.39+ | Moli arm64 | +| Windows x86_64 | `win_amd64` | Windows 10+ | Moli x64 | +| Windows arm64 | `win_arm64` | Windows 10+ | Moli arm64 | + +每个 Wheel 都包含 `a3s_code/moli/` 和来源记录。Bootstrap 会把 +Native Extension 与 Sidecar 解压到按用户共享的缓存;进程锁与原子替换确保多个 +应用同时首次启动时只安装一次,后续进程复用同一个已校验的 Moli。Linux musl +没有列出,因为上游 Moli 没有 musl 资产;该平台请使用系统/显式 Moli,或选择 +Chrome/Lightpanda 后端。 + +Intel Mac 使用 macOS 12 或更高版本时,应使用实际运行应用的同一个解释器安装: + +```bash +python3.14 -m ensurepip --upgrade # 只有该解释器没有 pip 时才需要 +python3.14 -m pip install --upgrade pip +python3.14 -m pip install a3s-code +``` + +如果 `python3.14 -m pip` 报告 `No module named pip`,问题发生在 Python 环境中, +还没有进入 A3S Code。请先为该解释器初始化或重新安装 pip,再重试。Intel Wheel +使用 `x86_64` 并以 macOS 12 为最低版本;该构建不包含可选的本地 ONNX Embedding +适配器。在 Intel 平台请保持 Workspace Retrieval 的 Model-free 模式,或配置明确 +授权的远程 Embedding Provider。 + +## Go 模块与桥接程序 + +Go 1.23 及以上版本使用纯 Go API,不需要 CGO。一个长驻的 +`a3s-code-go-bridge` 进程持有原生 Runtime,通过带版本号的 JSONL 协议传输 +多路复用请求和 `EventEnvelopeV1`。 + +包含 Go SDK 的仓库 Release 会发布路径前缀 Tag `sdk/go/vX.Y.Z`,它与 +`vX.Y.Z` Release 对应,同时提供 `a3s-code-go-bridge-SHA256SUMS`、独立桥接 +程序和包含对应 Moli Sidecar 的 Bundle: + +| 系统 | 资产目标 | Bundle 浏览器 | +| ------- | ----------------------------- | ------------- | +| Linux | `x86_64-unknown-linux-gnu` | Moli x64 | +| Linux | `aarch64-unknown-linux-gnu` | Moli arm64 | +| macOS | `x86_64-apple-darwin` | Moli x64 | +| macOS | `aarch64-apple-darwin` | Moli arm64 | +| Windows | `x86_64-pc-windows-msvc.exe` | Moli x64 | +| Windows | `aarch64-pc-windows-msvc.exe` | Moli arm64 | + +从 [GitHub Releases](https://github.com/A3S-Lab/Code/releases) 下载桥接程序, +使用发布的 SHA-256 文件校验,并确保它与 Go module 版本相同。可以将程序加入 +`PATH`、设置 `A3S_CODE_GO_BRIDGE`,或传入 `code.WithBridgePath`: + +```bash +export A3S_CODE_GO_BRIDGE=/opt/a3s/bin/a3s-code-go-bridge +``` + +```powershell +$env:A3S_CODE_GO_BRIDGE = 'C:\a3s\a3s-code-go-bridge.exe' +``` + +没有对应架构的 Release 资产时,可以从源码构建: + +```bash +bash .github/setup-workspace.sh +cargo build --release --package a3s-code-go-bridge --bin a3s-code-go-bridge +``` + +`code.Create` 会对传输协议、事件协议和完整操作清单执行 fail-closed 握手。 +Go 错误使用稳定的 `*code.Error` 错误码,Context 取消和 Deadline 仍可通过 +`errors.Is` 判断。桥接程序覆盖完整的可序列化 Agent/Session 能力,以及由 Go +实现的 Hook、预算守卫、斜杠命令和流水线回调。任意 Rust trait 对象仍属于 +Rust 原生扩展机制;其他 SDK 通过等价的值配置、回调、直接工具或 MCP 边界接入。 + +## 四种 SDK 的共用能力 + +四种 SDK 使用相同的 Session 生命周期、事件格式和 Snapshot。界面可以订阅同一套 +`AgentEvent` / `EventEnvelopeV1`,保存后也可以按 Session ID 恢复。 + +### 优先级调度器接口 + +v6.9 为四种 SDK 增加相同的 Agent 级调度器控制。创建 Session 时选择 `urgent`、 +`interactive`、`foreground`、`background` 或 `maintenance`,再从 Agent 或任意 +同级 Session 读取共享占用快照: + +| SDK | Session 配置 | Agent / Session 快照 | +| ------- | -------------------------------------------------- | ------------------------------ | +| Rust | `SessionOptions::with_task_priority(TaskPriority)` | `task_scheduler_stats().await` | +| Node.js | `taskPriority` | `taskSchedulerStats()` | +| Python | `SessionOptions.task_priority` | `task_scheduler_stats()` | +| Go | `SessionOptions.TaskPriority` | `TaskSchedulerStats(ctx)` | + +快照包含全局容量、活动与等待总数、按优先级分组的计数和关闭状态。顺序、老化、取消、 +配置与完整示例见[任务优先级调度器](/guide/tasks#agent-wide-priority-scheduler)。 + +### 安全点运行控制接口 + +四种 SDK 都能在不发起第二个对话操作的情况下,调整或中断当前活动 Run: + +| SDK | 调整方向 | 中断 | 快照 | +| ------- | ---------------------------- | ----------------------------------- | -------------------------------- | +| Rust | `steer(SteerRequest).await` | `interrupt(InterruptRequest).await` | `run_control_snapshot().await` | +| Node.js | `steer(input, options)` | `interrupt(options)` | `runControlSnapshot()` | +| Python | `steer` / `steer_async` | `interrupt` / `interrupt_async` | 同步/异步 `run_control_snapshot` | +| Go | `Steer(ctx, input, options)` | `Interrupt(ctx, options)` | `RunControlSnapshot(ctx)` | + +请求使用不可变 Run ID、可选乐观回合守卫、截止时间与幂等键。回执会区分 `accepted`、 +`applied`、`settled` 与 `rejected`;共用 `run_control_applied` 事件记录安全点应用。 +完整示例与生命周期语义见[会话](/guide/sessions#安全点运行控制)。 + +### 工具结果投影接口 + +四种 SDK 都能把同一个带版本的确定性投影策略固定到 Session: + +| SDK | Session 配置或 Builder | +| ------- | ---------------------------------------------------------------- | +| Rust | `with_tool_result_transform_policy(ToolResultTransformPolicyV1)` | +| Node.js | `toolResultTransformPolicy` | +| Python | `SessionOptions.tool_result_transform_policy` | +| Go | `SessionOptions.ToolResultTransformPolicy` | + +Rust 与 Python 提供 `context_efficient()` 预设;Node.js 与 Go 接受相同的显式字段。策略 +会写入快照,每个工具结果都会携带 `a3s.code.tool-result-evidence.v1` 元数据。字段值、 +执行顺序、边界、损失模式与四种 SDK 示例见 +[工具](/guide/tools#确定性的工具结果投影)。 + +共用指南会在完整的共用 SDK 能力面中,将 Go 与 Node.js、Python 并列展示。 +可以从[快速开始](/guide/examples/quick-start)开始,再阅读 +[流式事件](/guide/examples/streaming)、[直接工具](/guide/examples/direct-tools)、 +[会话](/guide/sessions)、[验证](/guide/verification)、[MCP](/guide/mcp)和 +[持久化](/guide/persistence)。 + +四种 SDK 都能配置持久化、记忆、Local/S3 Workspace、Remote Git、权限与确认、 +Hook、MCP、队列、确定性重放和编排。Rust 还可以接收自定义 `LlmClient` 或 +`ContextProvider` 等任意进程内 trait 实现;其他语言通过回调、直接工具或 MCP +接入自定义服务。界面接入可从 [Session 与事件流](/guide/sessions) 开始。 + +## 产品能力发现 + +本版本只有一份产品级能力契约。每个官方 SDK 的 +`sdk_capabilities()`(或对应语言名称)都会返回相同顺序的 带 baseline/advanced 分级的能力清单。 +每条记录包含稳定 ID、分类、规范操作名、描述和 `host_owned` 标志。 +`host_owned` 只表示策略、凭据或外部生命周期由谁提供,不表示 SDK 不支持该操作。 + +能力清单覆盖 Agent/Runtime 生命周期、受治理工具、代码智能、Workspace 检索与工具、 +记忆与 Cognitive Package、A3S Use 任务、模型适配器、结构化输出、MCP 与 Skill、 +规划与优先级调度、可编程工作流、持久化、State Graph、发布/协议契约、网页搜索、 +Moli、S3、Agent Server、OpenTelemetry、对话、运行观测和治理。应用应通过清单 +协商可选能力,不要根据包文件或版本号猜测支持情况。 + + + + +```rust +use a3s_code_core::{sdk_capabilities, sdk_capabilities_schema}; + +let capabilities = sdk_capabilities(); +assert_eq!(sdk_capabilities_schema(), "a3s-code/sdk-capabilities/v2"); +assert!(capabilities.iter().any(|item| item.id == "web_search")); +``` + + + + +```ts +import { sdkCapabilities, sdkCapabilitiesSchema } from '@a3s-lab/code'; + +const capabilities = sdkCapabilities(); +console.log(sdkCapabilitiesSchema(), capabilities.length); +``` + + + + +```python +from a3s_code import sdk_capabilities, sdk_capabilities_schema + +capabilities = sdk_capabilities() +assert sdk_capabilities_schema() == "a3s-code/sdk-capabilities/v2" +``` + + + + +```go +capabilities, err := code.SDKCapabilities(ctx) +if err != nil { + return err +} +fmt.Println(code.SDKCapabilitiesSchema(), len(capabilities)) +``` + + + + +## Moli Runtime 与网页搜索 + +`web_search` 使用 `a3s-search` v3.1.0,并默认用 Moli 执行 JavaScript 渲染搜索。 +Runtime 的解析顺序固定为:显式 `browserPath`/`A3S_CODE_MOLI_EXECUTABLE`、打包 +Sidecar、已校验的共享缓存、可发现的系统 Moli,最后才通过 HTTPS 下载固定版本。 +缓存按用户、版本和目标平台隔离;独占安装锁与原子收据确保多个 +`a3s-code` 进程同时启动时不会重复安装。严格离线时可设置 +`autoDownloadMoli: false`(或对应语言字段)。 + +诊断调用只读;Ensure 调用只有在默认自动配置开启且 Release Manifest/SHA-256 校验 +通过后才可能下载。v8.5.1 的上游 Moli 没有 Linux musl 资产;该平台请提供系统/显式 +Moli,或选择 Chrome/Lightpanda 后端。 + + + + +```rust +use a3s_code_core::{ensure_moli, moli_runtime_info, HeadlessConfig}; +use std::time::Duration; + +let config = HeadlessConfig::default(); +let status = moli_runtime_info(Some(&config)); +let executable = ensure_moli(&config, Duration::from_secs(120)).await?; +println!("{} {:?} {}", status.version, status.executable, executable.display()); +``` + + + + +```ts +import { BrowserBackend, ensureMoli, moliRuntimeInfo } from '@a3s-lab/code'; + +const status = moliRuntimeInfo({ + backend: BrowserBackend.Moli, + autoDownloadMoli: true, +}); +const executable = await ensureMoli({ backend: BrowserBackend.Moli }); +console.log(status.version, executable); +``` + + + + +```python +from a3s_code import ensure_moli_async, moli_runtime_info + +status = moli_runtime_info() +executable = await ensure_moli_async() +print(status["version"], executable) +``` + + + + +```go +status, err := code.MoliRuntimeInfo(ctx, code.NewMoliHeadlessConfig()) +if err != nil { + return err +} +executable, err := code.EnsureMoli(ctx, code.NewMoliHeadlessConfig()) +if err != nil { + return err +} +fmt.Println(status.Version, executable) +``` + + + diff --git a/website/docs/v8.5.1/zh/guide/_meta.json b/website/docs/v8.5.1/zh/guide/_meta.json new file mode 100644 index 00000000..ebb47e89 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/_meta.json @@ -0,0 +1,72 @@ +[ + "index", + "tui", + { + "type": "dir", + "name": "examples", + "label": "示例", + "collapsible": true, + "collapsed": true + }, + { + "type": "section-header", + "label": "文件系统优先" + }, + "filesystem-first", + "convention-over-configuration", + "agents-md", + "filesystem-instructions", + "filesystem-config", + "agent-dir", + "filesystem-agents", + "filesystem-skills", + "filesystem-tools", + "filesystem-schedules", + { + "type": "section-header", + "label": "运行时核心" + }, + "api-contract", + "sessions", + "commands", + "tools", + "verification", + "tasks", + "teams", + "orchestration", + "skills", + { + "type": "section-header", + "label": "治理工程" + }, + "security", + "hooks", + "limits", + "isolation", + { + "type": "section-header", + "label": "基础设施" + }, + "architecture", + "lane-queue", + "workspace-backends", + "multi-machine", + "cluster-extension-points", + { + "type": "section-header", + "label": "扩展" + }, + "providers", + "mcp", + "context", + "memory", + "persistence", + "telemetry", + { + "type": "dir", + "name": "rfcs", + "label": "RFC", + "collapsible": true, + "collapsed": true + } +] diff --git a/website/docs/v8.5.1/zh/guide/agent-dir.mdx b/website/docs/v8.5.1/zh/guide/agent-dir.mdx new file mode 100644 index 00000000..c7162939 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/agent-dir.mdx @@ -0,0 +1,194 @@ +--- +title: '智能体目录' +description: '智能体目录结构、加载映射、serve 守护进程和持久化边界。' +--- + +# 智能体目录 + +import { Tab, Tabs } from '@rspress/core/theme'; + +AgentDir 是按约定定义长期 Agent 的单一目录。一个文件夹承载主 Agent 的角色、运行配置、私有 skills、目录级 tools 和周期性 schedules。`AgentDir::load` 读取该目录并合成已有的 A3S Code 配置对象;它不引入新的 runtime,也不引入新的 prompt 系统。 + +这是一个有意为之的设计选择:`instructions.md` 作为 prompt **slot** 注入,而不是作为 system-prompt 覆盖。harness 始终保持 `BOUNDARIES`、response-format 契约、tool 可见性、安全门以及 verification 的权威性。一次调度运行始终是一个完整的 harness turn(`AgentSession::send`),绝不是裸的模型调用。 + +核心实现由 Rust 的 `serve` Cargo feature 控制。Rust、Node.js、Python 和 Go +都通过各自的原生 SDK 暴露同一套守护进程生命周期。 + +## 目录结构 + +```text +my-agent/ +├── instructions.md (required) Role and guidelines. Injected as a prompt slot. +├── agent.acl (optional) Model, providers, queue, and CodeConfig. +├── skills/ (optional) Private *.md skills. +├── schedules/ (optional) Cron jobs: frontmatter + body prompt. +└── tools/ (optional) Tool specs: kind: mcp or kind: script. +``` + +只有 `instructions.md` 是必需的。其余一切都是可选的,缺失的子目录只是不贡献对应能力。 + +`AgentDir::load` 将该目录映射到已有对象上: + +| Path | Becomes | Notes | +| ----------------- | ------------------------ | -------------------------------------------------------------------------------------------------- | +| `instructions.md` | `SystemPromptSlots.role` | 主 Agent 的角色 slot,详见 [instructions.md](/guide/filesystem-instructions)。 | +| `agent.acl` | `CodeConfig` | 模型、provider、队列和目录发现配置,详见 [agent.acl](/guide/filesystem-config)。 | +| `skills/` | `skill_dirs` | AgentDir 私有技能,详见 [skills/ 技能目录](/guide/filesystem-skills)。 | +| `schedules/*.md` | `Vec` | 每个文件一条周期性 turn,详见 [schedules/ 调度目录](/guide/filesystem-schedules)。 | +| `tools/*.md` | `Vec` | 每个文件一个 `kind: mcp` 或 `kind: script` 工具,详见 [tools/ 工具目录](/guide/filesystem-tools)。 | + +## `serve` 守护进程 + +`serve_agent_dir` 将调度加载到各自独立的 session 中,并运行它们的 cron 循环,直到某个 cancellation token 触发。每次触发都会把该调度的 prompt 经由 `AgentSession::send` 路由。 + + + + +```rust +use a3s_code_core::config::AgentDir; +use a3s_code_core::serve::serve_agent_dir; +use a3s_code_core::{Agent, SessionOptions}; +use tokio_util::sync::CancellationToken; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let agent_dir = AgentDir::load("./my-agent")?; + let agent = Agent::from_config(agent_dir.config.clone()).await?; + let cancel = CancellationToken::new(); + let shutdown = cancel.clone(); + tokio::spawn(async move { + let _ = tokio::signal::ctrl_c().await; + shutdown.cancel(); + }); + + let options = SessionOptions::new().with_file_session_store("./sessions"); + serve_agent_dir( + &agent, + &agent_dir, + "./workspace", + Some(options), + cancel, + ) + .await?; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('./my-agent/agent.acl'); +const daemon = await agent.serveAgentDir('./my-agent', './workspace', { + sessionStore: new FileSessionStore('./sessions'), +}); + +await new Promise((resolve) => process.once('SIGINT', resolve)); +await daemon.stop(); +await agent.close(); +``` + + + + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +agent = Agent.create('./my-agent/agent.acl') +options = SessionOptions() +options.session_store = FileSessionStore('./sessions') +daemon = agent.serve_agent_dir('./my-agent', './workspace', options) + +input('按 Enter 停止 AgentDir 守护进程……') +daemon.stop() +agent.close() +``` + + + + +```go +package main + +import ( + "context" + "log" + "os" + "os/signal" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx, stopSignal := signal.NotifyContext(context.Background(), os.Interrupt) + defer stopSignal() + + agent, err := code.NewAgent(ctx, "./my-agent/agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + daemon, err := agent.ServeAgentDir(ctx, "./my-agent", "./workspace", &code.SessionOptions{ + FileSessionStoreDir: "./sessions", + }) + if err != nil { + log.Fatal(err) + } + + <-ctx.Done() + if err := daemon.Stop(context.Background()); err != nil { + log.Fatal(err) + } +} +``` + + + + +直接使用 Rust 的项目还需要在 `Cargo.toml` 中启用该 feature: + +```toml +[dependencies] +a3s-code-core = { version = "...", features = ["serve"] } +``` + +## 覆盖会话选项 + +Rust `serve_agent_dir` 的第四个参数,或 `serveAgentDir` / `serve_agent_dir` / +`ServeAgentDir` 的 `options` 参数,会合并进每一个调度 session,可用于固定 model、 +session store、权限或 prompt slots。守护进程始终为每个 schedule 分配稳定的 +`session_id`:`schedule:`;extra options 中的 `session_id` 会被有意忽略, +避免多个 schedule 写入同一个 store id。若未提供 prompt slots,则使用 AgentDir +的 `instructions.md` slot。 + +优雅取消会让进行中的 turn 完成其循环迭代后再停止;一旦每个 job 循环都已退出,守护进程返回 `Ok(())`。一个没有任何已启用调度的 AgentDir 会立即返回。 + +## 持久性 + +默认情况下,每次守护进程启动都会全新启动每个调度 session。在 +`SessionOptions` 中传入文件 session store——如上例中的 +`with_file_session_store`、`FileSessionStore` 或 `FileSessionStoreDir`——即可 +从已保存的对话历史恢复现有的 `schedule:` session。 + +恢复只还原历史。当前的 `instructions.md`、`skills/` 和 `tools/` 会在每次启动时重新应用,因此即便是被恢复的 session,编辑 AgentDir 也会在下一次重启时生效。 + +## 状态 + +| Area | State | +| ----------------------------------------- | -------------------------------------------------- | +| `instructions.md`、`agent.acl`、`skills/` | 已加载并使用。 | +| `schedules/` + serve 守护进程 | 已实现(`serve` feature)。 | +| 启动时重新水合 | 已实现,配合 `SessionStore` 可在重启后恢复上下文。 | +| `tools/`(`kind: mcp`) | 已实现,声明式 MCP server 注册进每个调度 session。 | +| `tools/`(`kind: script`) | 已实现,基于 `program` 路径的沙箱化 QuickJS tool。 | + +## 备注 + +- `instructions.md` 是一个角色 slot,因此 harness 的边界、响应契约以及 verification 保持权威。 +- AgentDir 是主 Agent 的目录;它不同于 `agent_dirs` / `registerAgentDir`,后者扫描 [agents/ 角色目录](/guide/filesystem-agents) 以获取 worker/subagent 定义。 +- secret 应放在环境变量或宿主密钥系统中,并从 `agent.acl` 引用,不要内联写进目录里。 +- `tools/` 由 serve 守护进程按调度 session 安装。普通交互式 session 应优先使用 direct tools、MCP 或 SDK 注册路径。 diff --git a/website/docs/v8.5.1/zh/guide/agents-md.mdx b/website/docs/v8.5.1/zh/guide/agents-md.mdx new file mode 100644 index 00000000..54f8abb1 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/agents-md.mdx @@ -0,0 +1,73 @@ +--- +title: 'AGENTS.md' +description: '作为一等上下文的项目指令' +--- + +# AGENTS.md + +`AGENTS.md` 是 workspace 级项目指令文件。它让项目规则随仓库一起版本化,避免每次 prompt 都重复说明构建命令、代码风格、安全边界和发布流程。 + +在文件系统优先架构里,`AGENTS.md` 负责“这个项目如何工作”;AgentDir 的 `instructions.md` 负责“这个长期 Agent 是谁”。两者都会进入上下文组合流程,但都不能覆盖 harness 的权限门、响应契约或验证要求。 + +```md +# 项目说明 + +- Core 变更运行 `cargo test -p a3s-code-core`。 +- 不要提交 `.a3s/config.acl` 里的真实密钥。 +- 搜索优先使用 `rg`。 +- 发布检查必须包含包元数据、CI 和 provider 验证。 +``` + +## 适合内容 + +- 构建、测试、lint、格式化和发布命令。 +- 目录职责、模块边界和代码风格。 +- 安全规则,例如 secrets、权限、外部副作用和数据处理要求。 +- 验证政策,例如哪些变更必须跑哪些检查。 +- 项目专用术语和常见工作流。 + +## 不适合内容 + +- 密钥、token、私有凭据或临时个人路径。 +- 某个 worker agent 的角色说明;放进 `.a3s/agents/`。 +- 某个长期 Agent 的身份和默认输出风格;放进 `instructions.md`。 +- 可复用 checklist;放进 `.a3s/skills/`。 + +## 嵌套规则 + +A3S Code 会在 session 启动时构建一条项目指令链:先找到最近的 Git 根目录,再从 +Git 根目录逐级走到当前 workspace,每层目录最多选择一份文件。每层的查找顺序是: + +1. `AGENTS.override.md` +2. `AGENTS.md` +3. `project_doc_fallback_filenames` 中按顺序配置的后备文件名 + +指令按“根目录到 workspace”顺序拼接,因此越靠近当前 workspace 的局部规则越晚出现, +优先级也越高。`AGENTS.override.md` 只替换同一目录中的普通 `AGENTS.md`,不会清除父目录 +规则。找不到 Git 根目录时,只检查当前 workspace。 + +空文件会跳过。默认总预算为 32 KiB,可通过 `project_doc_max_bytes` 调整;设为 0 可关闭 +项目指令加载。A3S Code 只接受项目根目录内的 UTF-8 普通文件,会忽略符号链接候选和不安全 +的后备文件名。经过边界约束的最终指令链属于 session 必读上下文,不会被通用检索预算静默丢弃。 + +只有当子目录规则确实不同,才添加嵌套 `AGENTS.md`。例如桌面端、API 端或某个 SDK +的构建工具链不同,可以在对应目录下放局部规则。不要为了重复根部说明而复制文件; +重复会让长期 Agent 难以判断哪份规则才是最新。 + +```acl +project_doc_max_bytes = 65536 +project_doc_fallback_filenames = ["TEAM_GUIDE.md", ".agents.md"] +``` + +## 与其他约定的关系 + +```text +repo/ +├── AGENTS.md # 项目级长期指令 +├── agent.acl # 运行配置 +└── .a3s/ + ├── agents/ # worker/subagent 定义 + └── skills/ # 可复用技能 +``` + +`AGENTS.md` 应该告诉 Agent 项目事实和工作边界;`agent.acl` 告诉运行时如何连接模型和目录;`agents/` 与 `skills/` 提供可被发现的角色和流程。 diff --git a/website/docs/v8.5.1/zh/guide/api-contract.mdx b/website/docs/v8.5.1/zh/guide/api-contract.mdx new file mode 100644 index 00000000..be7cbdef --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/api-contract.mdx @@ -0,0 +1,807 @@ +--- +title: 'API 契约' +description: '本地集成检查覆盖的 SDK 机制' +--- + +# API 契约 + +本页只记录 `scripts/docs_api_contract_smoke.mjs` 覆盖过的 A3S Code Node.js +SDK 行为。该脚本会启动一个临时的 OpenAI 兼容测试服务,创建真实 SDK 会话, +调用原生绑定并断言返回值,不需要先构建文档。 + +在仓库根目录运行: + +```bash +node scripts/docs_api_contract_smoke.mjs +``` + +## 智能体 + +已验证入口: + +```ts +const agent = await Agent.create(aclSource); +await agent.refreshMcpTools(); + +const session = agent.session(workspace, options); +const named = agent.sessionForAgent(workspace, 'explore', [], options); +``` + +`Agent.create()` 接受 ACL 源文本或 `.acl` 文件路径。JSON 配置不在 +已验证契约内。集成检查覆盖 `apiKey`/`baseUrl` 和 +`api_key`/`base_url` 两组服务提供商别名。`sessionForAgent()` 已用内置 +`explore` 智能体验证。 + +Node.js 工厂在 JavaScript 接口上保持同步命名,但原生实现会把资源解析委托给核心 +的异步会话构建路径。Rust 嵌入方应使用 `SessionBuilder::build().await` 或异步工厂; +同步 Rust 兼容工厂要求显式传入预先初始化的记忆存储。 + +集成检查也覆盖会创建文件型 session store 的 ACL 字段: + +```acl +storage_backend = "file" +sessions_dir = "/tmp/a3s-doc-stores/acl-storage" +``` + +本契约不覆盖 `storage_url`。它不是本地文件型 session persistence 路径; +ACL 中使用 `sessions_dir`,或在 SDK options 中传 `sessionStore`。 + +## 会话选项 + +已验证的 option 形状: + +```ts +const session = agent.session(workspace, { + model: 'openai/docs-alt', + builtinSkills: true, + planningMode: 'disabled', + memoryStore: new FileMemoryStore(memoryDir), + sessionStore: new FileSessionStore(sessionDir), + sessionId: 'docs-contract', + autoSave: true, + securityProvider: new DefaultSecurityProvider(), + skillDirs: [path.join(workspace, 'skills')], + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: 'Always separate blockers from nice-to-have improvements.', + }, + ], + maxToolRounds: 24, + maxParseRetries: 3, + toolTimeoutMs: 120000, + circuitBreakerThreshold: 4, + autoCompact: true, + autoCompactThreshold: 0.75, + continuationEnabled: true, + maxContinuationTurns: 3, + maxExecutionTimeMs: 300000, // 5 分钟超时 + confirmationPolicy: { + enabled: true, + defaultTimeoutMs: 60000, + timeoutAction: 'reject', + }, +}); +``` + +`model` 是 per-session override。检查已覆盖使用 +`model: 'openai/docs-alt'` 创建的 session 会把 `docs-alt` 发送给本地 +provider。 + +基础 session accessor 已验证: + +```ts +console.log(session.sessionId); +console.log(session.workspace); +console.log(session.initWarning); +console.log(session.history()); +console.log(session.cancel()); +``` + +`workspace` 返回 SDK 规范化后的工作区路径。 + +`planningMode` 使用显式三态:`'auto'`、`'enabled'`、`'disabled'`。 + +检查也覆盖 session 创建时接受这个 `permissionPolicy` 形状: + +```ts +agent.session(workspace, { + permissionPolicy: { + deny: ['write(**/.env*)', 'bash(rm -rf*)'], + ask: ['bash(git push*)', 'bash(npm publish*)'], + allow: ['read(*)', 'search(*)', 'bash(npm run build*)'], + defaultDecision: 'ask', + enabled: true, + }, +}); +``` + +提示词插槽选项都是字符串: + +```ts +agent.session(workspace, { + role: 'release-readiness reviewer', + guidelines: + 'Find blockers before improvements. Require command evidence for done claims.', + responseStyle: 'concise, findings first', + goalTracking: true, +}); +``` + +## 结果结构 + +`session.send()` 返回的 `AgentResult` 字段在 result 对象自身: + +```ts +const result = await session.send('Return a short answer'); + +console.log(result.text); +console.log(result.toolCallsCount); +console.log(result.promptTokens); +console.log(result.completionTokens); +console.log(result.totalTokens); +console.log(result.verificationStatus); +console.log(result.pendingVerificationCount); +console.log(result.failedVerificationCount); +console.log(result.verificationReportCount); +console.log(result.verificationSummaryJson); +console.log(result.verificationSummaryText); +``` + +trace events 和 verification reports 是 session API,不是 `AgentResult` 字段。 + +## 流式事件 + +`session.stream()` 返回 `EventStream`。已验证的读取方式是 `.next()`: + +```ts +const stream = await session.stream('Stream one sentence'); + +while (true) { + const { value: event, done } = await stream.next(); + if (done) break; + if (!event) continue; + if (event.text) process.stdout.write(event.text); +} +``` + +冒烟检查会在循环结束后立即发起下一次 `send()`。因此这里验证的不只是事件交付, +还包括流式调用的单任务准入租约已经释放;不需要人为增加重试延时。 + +每个 SDK 事件都是 envelope v1 投影:`version === 1`、开放的 `type` 字符串、 +完整的 `payload` 和可选 `metadata`。消费者必须为未来事件类型保留默认分支。 +Node 提供 `payloadJson` / `metadataJson` 字符串视图;Python 提供 +`payload_json` / `metadata_json`,并保留 `event_type` 作为 `type` 的别名。 + +不要依赖 `for await`,除非你安装的 SDK 版本已经单独验证支持异步 +iteration。 + +## 直接工具 + +> 完整指南:[工具](/guide/tools)。 + +已验证的宿主侧直接工具调用: + +```ts +await session.readFile('README.md'); +await session.glob('src/*.rs'); +await session.grep('PermissionPolicy'); +await session.bash('printf docs-bash'); +await session.tool('read', { file_path: 'README.md' }); +await session.git('status'); +await session.git('diff'); +await session.git( + 'log', + undefined, + undefined, + undefined, + undefined, + undefined, + undefined, + 5, +); +await session.tool('search_skills', { query: 'release blockers', limit: 5 }); + +session.toolNames(); +session.toolDefinitions(); +session.registerAgentDir(path.join(workspace, 'agents')); +``` + +已验证的本地工作区 `toolNames()` 集合包含 `read`、`write`、`edit`、`patch`、 +`download`、`search`、`ls`、`bash`、`task`、`search_skills`、`Skill`、 +`program`、`git`、`batch`、`web_fetch` 和 `web_search`。 + +直接工具调用是宿主侧特权能力。把它暴露给最终用户之前,应在宿主应用内做权限判断。 + +`download` 通过通用直接工具 API 调用: + +```ts +const result = await session.tool('download', { + url: 'https://example.com/archive.tar.zst', + file_path: 'artifacts/archive.tar.zst', + overwrite: false, + connections: 4, + max_bytes: 536870912, + timeout: 300, + expected_sha256: + '0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef', +}); +``` + +只有 `url` 必填。`connections` 限制为 1–4,`max_bytes` 默认 512 MiB、最大 +8 GiB,`timeout` 默认 300 秒、最大 3600 秒,`expected_sha256` 必须正好包含 +64 位十六进制字符。`file_path` 是工作区相对路径,也可以省略并安全推断文件名; +`overwrite` 默认 `false`。 + +该工具只存在于可写本地工作区。模型驱动调用仍是受权限与 HITL 治理的工作区修改。 +传输会为请求保留签名查询参数,但从结果元数据中移除它们;逐跳执行 SSRF 与直接 DNS +目标校验,严格验证 Range 响应,只做有限重试,必要时顺序回退,并且只有完整下载和 +可选摘要校验成功后才把相邻临时文件提升为目标。 + +## AGENTS.md + +脚本会在工作区写入一个 `AGENTS.md`,并断言其中的指令标记 +出现在本地服务提供商请求体中: + +```md +# 项目说明 + +Always mention docs-contract-agents-md-token when asked for project instructions. +``` + +项目指令应保持可操作,并且不要包含密钥。 + +## 程序化工具调用 + +`session.program()` 在内嵌 QuickJS runtime 中运行有边界的 JavaScript: + +```ts +const result = await session.program({ + source: ` + export default async function run(ctx, inputs) { + const text = await ctx.readFile('README.md'); + const hits = await ctx.grep(inputs.q, { glob: '*.md' }); + return { summary: 'ok', hasHits: text.includes(inputs.q) && hits.includes(inputs.q) }; + } + `, + inputs: { q: 'planningMode' }, + allowedTools: ['read', 'grep'], + limits: { timeoutMs: 30000, maxToolCalls: 12, maxOutputBytes: 65536 }, +}); + +const meta = JSON.parse(result.metadataJson); +console.log(meta.script_result); +console.log(meta.program.tool_calls); +``` + +已验证的 `ctx` helper:`readFile`、`read`、`grep`、`glob`、`ls`、`bash`、 +`git` 和通用 `tool(name, args)`。`allowedTools` 限制脚本能调用的已注册 +tool。`program` 不会出现在它自己的默认 tool set 里。 + +## 验证 + +> 完整指南:[验证](/guide/verification)。 + +验证信息是 session 级能力: + +```ts +const report = await session.verifyCommands('docs api check', [ + { + id: 'echo', + kind: 'command', + description: 'echo works', + command: 'printf verify', + required: true, + }, +]); + +console.log(report.subject); +console.log(session.verificationReports()); +console.log(session.verificationSummary()); +console.log(session.verificationSummaryText()); +console.log(session.verificationPresets()); +console.log(formatVerificationSummary(session.verificationSummary())); +``` + +## 记忆 + +> 完整指南:[记忆](/guide/memory)。 + +Node memory 已用 `FileMemoryStore` 验证: + +```ts +const session = agent.session(workspace, { + memoryStore: new FileMemoryStore(memoryDir), +}); + +console.log(session.hasMemory); +await session.rememberSuccess('docs memory success', ['grep'], 'remembered'); +await session.rememberFailure('docs memory failure', 'expected failure', [ + 'bash', +]); +await session.memoryRecent(10); +await session.recallSimilar('docs memory', 5); +await session.recallByTags(['grep'], 10); +``` + +当前 Node.js SDK 已验证的近期记忆方法是 `memoryRecent()`。`recallRecent()` +不存在于当前 Node.js SDK 接口。 + +## 技能 + +> 完整指南:[技能](/guide/skills)。 + +文件型和内联技能已通过 `search_skills` 验证: + +```ts +const session = agent.session(workspace, { + skillDirs: [path.join(workspace, 'skills')], + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: 'Always separate blockers from nice-to-have improvements.', + }, + ], +}); + +await session.tool('search_skills', { query: 'release blockers', limit: 5 }); +await session.tool('search_skills', { + query: 'strict release review', + limit: 5, +}); +``` + +skill-file 检查使用带 YAML frontmatter 的 Markdown,并覆盖了 +`allowed-tools` key。 + +## 临时提问 + +> 完整指南:[会话](/guide/sessions)。 + +当前 SDK surface 没有专用的临时提问 helper。需要不改变 session transcript +时,使用显式 history: + +```ts +const snapshot = session.history(); +const side = await session.send('What is this test?', snapshot); + +console.log(side.text); +console.log(session.history().length === snapshot.length); +``` + +## 运行与取消 + +> 完整指南:[会话](/guide/sessions)。 + +每次 `send()` 或 `stream()` 都会记录可回放的 run state: + +```ts +const runs = await session.runs(); +const latest = runs.at(-1); + +if (latest) { + console.log(await session.runSnapshot(latest.id)); + console.log(await session.runEvents(latest.id)); +} + +const current = await session.currentRun(); +if (current?.id && current.status === 'running') { + await session.cancelRun(current.id); +} + +console.log(session.traceEvents()); +``` + +Headless 宿主可以准入精确且不可变的运行标识,并立即取得权威快照,而无需等待运行结束: + +```ts +const admitted = await session.spawnRunWithId( + 'release-42/run-7', + 'Verify the release', +); +console.log(admitted.snapshot.id, admitted.replayed); + +const recovered = await session.spawnRecoveryWithRunId( + 'checkpoint-run-6', + 'release-42/recovery-7', +); +console.log(recovered.snapshot.status, recovered.replayed); +``` + +重复提交兼容的不可变输入会返回 `replayed: true`,不会启动重复工作。输入冲突会返回 +`RUN_IDENTITY_CONFLICT`;关闭会话会取消已分离的后台 worker。 + +`currentRun()` 用于读取当前操作。空闲时,它可能返回 `null`,也可能 +因前序控制流保留一个快照。已完成历史请使用 `runs()`。 + +同一个会话中影响对话记录的操作采用单任务准入。重叠的发送、事件流、附件调用、 +斜杠命令或 `resumeRun` 会立即返回 `SessionBusy`,不会排队。即使公开句柄被丢弃, +事件流也会保留准入状态,直到生产者停止。 + +## 持久化 + +> 完整指南:[持久化](/guide/persistence)与[会话](/guide/sessions)。 + +文件型会话持久化已验证稳定的 `sessionId`、`autoSave`、显式 +`save()` 和 `resumeSession()`: + +```ts +const session = agent.session(workspace, { + sessionStore: new FileSessionStore(sessionDir), + sessionId: 'docs-contract', + autoSave: true, +}); + +await session.save(); + +const resumed = agent.resumeSession('docs-contract', { + sessionStore: new FileSessionStore(sessionDir), +}); +console.log(resumed.history()); +``` + +核心持久化会把对话、制品、追踪、运行记录、验证报告和子智能体任务快照一起提交为 +带版本号的 `SessionSnapshotV1`。文件或内存存储会原子发布这个聚合快照。旧式分片 +记录仍可加载;自定义存储必须显式实现聚合保存。 + +Node 进程需要及时释放会话级后台资源时,调用 `session.close()`。`close()` 是完整的 +优雅停止入口:把 `session.isClosed` 设为 `true`(之后 `send` / `stream` 会以 +`Session closed` 错误立即返回),触发会话级 `CancellationToken`,让所有进行中的运行、 +委派子智能体任务和待人工确认项全部中止。重复调用 `close()` 不会重复操作。 + +控制面只持有会话 ID 时,可以从智能体侧触发同样的清理: + +```ts +await agent.listSessions(); // ['session-a', 'session-b'] +await agent.closeSession('session-a'); // 若原本处于打开状态,返回 true +await agent.close(); // 关闭所有活动会话并断开全局 MCP +``` + +`agent.close()` 之后,再调用 `agent.session(...)` / `agent.resumeSession(...)` 会立即 +抛出 `Session closed`。该操作幂等。建议在进程退出处理函数中调用,保证没有会话级 +工作进程比智能体存活更久。 + +## 委派 + +> 完整指南:[任务](/guide/tasks)与[编排](/guide/orchestration)。 + +已验证核心委派工具的直接辅助方法: + +```ts +await session.task({ + agent: 'general', + description: 'docs delegated check', + prompt: 'Return a short response.', + maxSteps: 1, +}); + +await session.tasks([ + { + agent: 'general', + description: 'one', + prompt: 'Return one response.', + maxSteps: 1, + }, + { + agent: 'general', + description: 'two', + prompt: 'Return another response.', + maxSteps: 1, + }, +]); +``` + +两个辅助方法都返回来自模型可见 `task` 工具的 `ToolResult`。 + +## 钩子 + +> 完整指南:[钩子](/guide/hooks)。 + +已验证的钩子管理入口: + +```ts +session.registerHook( + 'docs-observer', + 'pre_tool_use', + { tool: 'bash' }, + { priority: 1, timeoutMs: 1000 }, + () => ({ action: 'continue' }), +); + +console.log(session.hookCount()); +session.unregisterHook('docs-observer'); +``` + +把钩子行为作为生产关卡前,需要对你依赖的具体事件路径做集成测试。 + +## 斜杠命令 + +> 完整指南:[命令](/guide/commands)。 + +自定义斜杠命令通过 `session.send()` 触发: + +```ts +session.registerCommand( + 'docs_status', + 'Return docs command status', + (args, ctx) => { + return `status args=${args}; session=${ctx.sessionId}; workspace=${ctx.workspace}`; + }, +); + +console.log(session.listCommands()); +const result = await session.send('/docs_status check'); +console.log(result.text); +``` + +## 执行通道队列 + +> 完整指南:[执行通道队列](/guide/lane-queue)。 + +队列基础设施需要显式启用: + +```ts +const queued = agent.session(workspace, { + queueConfig: { enableDlq: true, enableMetrics: true }, +}); + +console.log(queued.hasQueue()); +await queued.setLaneHandler('execute', { mode: 'external', timeoutMs: 1000 }); +await queued.pendingExternalTasks(); +await queued.completeExternalTask('missing', { + success: true, + result: { ok: true }, +}); +await queued.queueStats(); +await queued.queueMetrics(); +await queued.deadLetters(); +``` + +没有传入 `queueConfig` 的普通会话不会启用队列。 + +## MCP + +> 完整指南:[MCP](/guide/mcp)。闲置断开见[集群扩展点](/guide/cluster-extension-points)。 + +集成检查覆盖一个真实的标准输入输出 MCP 服务器: + +```ts +const count = await session.addMcp({ + name: 'echo', + transport: { + type: 'stdio', + command: process.execPath, + args: ['tools/mcp_echo_server.mjs', 'example-value'], + }, + timeoutMs: 30000, +}); + +console.log(count); +console.log(await session.mcpStatus()); +console.log( + session.toolNames().filter((name) => name.startsWith('mcp__echo__')), +); + +await session.tool('mcp__echo__echo', { message: 'docs mcp ok' }); +await session.removeMcpServer('echo'); +``` + +服务器注册出的工具名称格式是 `mcp____`。 +`addMcpServer(...)` 和 `addMcpServerConfig(...)` 仍是兼容别名;新示例使用参数对象 +更紧凑的 `addMcp(...)` API。 + +运行时添加或移除只作用于当前会话的私有管理器。智能体全局和宿主提供的管理器是 +继承的只读能力来源,因此一个会话不能修改同级会话或全局 MCP 配置。 + +## 集群级扩展点 + +> 完整指南:[集群扩展点](/guide/cluster-extension-points)(身份标签、预算守卫、集群事件、确定性 ID/回放、循环检查点、保留上限)。 + +这些契约让集群控制面在**不派生框架分支**的前提下接入多租户、成本管控和容错运行。 +框架定义“决策点”和“结构化事件”,**策略实现由宿主提供**。 + +### 身份标签 + +`SessionOptions` 上四个可选字段会透传到钩子、追踪与 `SessionData`,框架本身不解释: + +```ts +const session = agent.session(workspace, { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', + sessionStore: new FileSessionStore('./sessions'), +}); +session.tenantId; // -> 'tenant-example' +session.correlationId; // -> 'trace-example' +``` + +恢复时,`apply_persisted_runtime_options` 会从持久化快照中还原标签;但**调用方在 +`resume_session` 时传入的选项优先**,可以借此重新设置标签。 + +### 预算 / 成本守卫 + +`BudgetGuard` 会在每个属于运行的服务提供商调用前检查,并在成功响应后记录用量; +每个受治理的工具调用(包括嵌套调用与可信宿主直接调用)也会在执行前检查。 +`Deny` 返回 `CodeError::BudgetExhausted { resource, reason }`;`SoftLimit` 发射 `AgentEvent::BudgetThresholdHit { kind: "soft", .. }` 后继续执行。 + +Rust 宿主直接注入 trait。Node.js、Python 与 Go 的回调桥接见本节后文: + +```rust +let guard: Arc = /* host-supplied impl */; +let opts = SessionOptions::new().with_budget_guard(guard); +``` + +### 集群事件词汇 + +`AgentEvent`(`#[non_exhaustive]`)新增三类平台级事件,host 通过 `HookExecutor` 注入: + +- `BudgetThresholdHit { resource, kind, consumed, limit, message? }` +- `PassivationRequested { reason, deadline_ms? }` +- `PeerInvocation { from_session_id, from_tenant_id?, correlation_id? }` + +session 内部 hook 可统一订阅,不必关心 host 用什么传输发过来。 + +### 确定性标识与时钟 + +`HostEnv { id_generator, clock }` 替换默认的 `uuid::Uuid::new_v4()` + 墙上时钟。Replay 工具传入 `SequentialIdGenerator` + `FixedClock` 即可在另一台机器上 bit-identical 重放一个 run。 + +### 循环检查点与运行恢复 + +配置了 `SessionStore` 后,agent loop **每次 tool round 结束**会持久化一个 `LoopCheckpoint`(按 `run_id` 索引)。任何拥有同一个 store 的节点都能从最近的边界 rehydrate: + +```ts +// Node.js:宿主探测到 A 节点失效后,在 B 节点上恢复: +const session = agentB.session(workspace, { + sessionStore: new FileSessionStore('./sessions'), + sessionId: 'session-from-node-a', +}); +const result = await session.resumeRun('run-id-from-node-a'); +``` + +```python +# Python 等价写法 +opts = SessionOptions() +opts.session_store = FileSessionStore('./sessions') +opts.session_id = 'session-from-node-a' +session = agent_b.session(workspace, opts) +result = session.resume_run('run-id-from-node-a') +``` + +```go +// Go 等价写法 +session, err := agent.Session(ctx, workspace, &code.SessionOptions{ + FileSessionStoreDir: "./.a3s/sessions", + SessionID: "session-from-node-a", +}) +if err != nil { + return err +} +result, err := session.ResumeRun(ctx, "run-id-from-node-a") +``` + +resume 出来的会**分配一个全新的 run id** — 框架不假装旧 run 还在继续,新旧 run 的关系是 host 的元数据。两个可区分的错误路径方便 host 端调度分支: + +- `"resume_run requires a session_store"` — host 应该回退到新建 session。 +- `"no loop checkpoint found for run 'X'"`:宿主可以稍后重试(可能正好遇到检查点写入竞态),也可以把该运行视为已丢失。 + +**边界策略**:checkpoint 只在 tool round **之间**取,不在工具执行中途取。进程在工具执行中途死掉时,这一轮的工作会丢失,LLM 从前一个边界重新思考。这是用"重试成本"换"正确性" — 把非幂等工具(write、bash)在边界两侧重跑比让 LLM 重想要糟得多。 + +### 长时间会话的保留上限 + +`SessionRetentionLimits` 让 host 给四种 in-memory 存储设上限:run 记录、每 run 的事件、 +trace 事件、**终态的** subagent 任务快照。每个字段都是可选的;省略时保留框架的有限 +默认值,只有明确设置 `unbounded: true` 才恢复无限保留。FIFO 严格按插入序丢; +**Running 状态的** subagent 任务永不被丢。 + +```rust +use a3s_code_core::retention::SessionRetentionLimits; + +let limits = SessionRetentionLimits::new() + .with_max_runs(100) + .with_max_events_per_run(5_000) + .with_max_trace_events(10_000) + .with_max_terminal_subagent_tasks(1_000); + +let opts = SessionOptions::new().with_retention_limits(limits); +``` + +上限建议跟 host 自己 Prometheus / 观测系统的内存预算保持一致。Node 暴露为 +`retentionLimits`;Python 暴露为 `opts.retention_limits`;Go 暴露为 +`SessionOptions.RetentionLimits`。 + +### MCP 闲置断开 + +`Agent::disconnect_idle_mcp(threshold_ms)` 扫描所有已连接的 MCP server,把"最后活跃时间"早于 `now - threshold_ms` 的全部断开。注册的配置**保留** — 后续 tool 调用会按需重连。返回被断开的 server 名称列表。 + +```ts +// Node.js:定期回收闲置的 MCP 子进程 +setInterval(async () => { + const dropped = await agent.disconnectIdleMcp(5 * 60 * 1000); // 5min + if (dropped.length) { + console.log('reaped idle MCP servers:', dropped); + } +}, 60_000); +``` + +```python +# Python 等价写法 +dropped = agent.disconnect_idle_mcp(5 * 60 * 1000) +``` + +```go +// Go 同样传入毫秒数 +dropped, err := agent.DisconnectIdleMCP(ctx, 5*60*1000) +``` + +每次 `connect` 和成功的 `call_tool` 都会刷新活跃时间。Host 走旁路通道路由 tool 时,可以手动 `McpManager.touch(name)` 把 server 保温。 + +### `BudgetGuard` 的 SDK 桥接 + +所有支持回调的 SDK 共用同一个决策返回形状: + +| 返回值 | 效果 | +| -------------------------------------------------------- | ------------------------------------------------------------------------------- | +| `None` / `null` / `{decision:'allow'}` | 静默放行 | +| `{decision:'soft', resource, consumed, limit, message?}` | 发射 `BudgetThresholdHit('soft')` 事件,继续执行 | +| `{decision:'deny', resource, reason}` | 中止调用,Python 抛 `RuntimeError("Budget exhausted...")`/Node reject 同样的错误 | + +guard 对象上缺失的方法会使用宽松默认值(Allow / no-op)。Python 回调异常会回退为 +Allow;Go 回调报错、超时或返回无法解析的决策时会 fail-closed 为 Deny。 + +```python +# Python:通过 SessionOptions 在创建会话前挂载 +class MyGuard: + def check_before_llm(self, session_id, estimated_tokens): + return {"decision": "deny", "resource": "llm_tokens", "reason": "cap"} + def record_after_llm(self, session_id, usage): + track(session_id, usage["total_tokens"]) + +opts = SessionOptions() +opts.budget_guard = MyGuard() +session = agent.session(workspace, opts) +``` + +```ts +// Node.js:创建会话后通过 setBudgetGuard 挂载。 +// JsFunction 不能放进值类型的 SessionOptions,因此预算守卫注册在 Session 上, +// 下一次 send/stream 生效。 +session.setBudgetGuard({ + checkBeforeLlm: (ctx) => { + if (overBudget(ctx.sessionId)) { + return { decision: 'deny', resource: 'llm_tokens', reason: 'cap' }; + } + return null; + }, + recordAfterLlm: (ctx) => { + track(ctx.sessionId, ctx.usage.totalTokens); + }, +}); +``` + +```go +err := session.SetBudgetGuard(ctx, &code.BudgetGuardHandlers{ + CheckBeforeLLM: func( + _ context.Context, + call code.BudgetLLMContext, + ) (*code.BudgetDecision, error) { + if overBudget(call.SessionID) { + return &code.BudgetDecision{ + Decision: "deny", + Resource: "llm_tokens", + Reason: "cap", + }, nil + } + return &code.BudgetDecision{Decision: "allow"}, nil + }, +}) +``` + +Node 回调接收单个 `ctx` 对象,且**绝不能 throw**;请用 `try/catch` 包裹并返回显式决策。 +卡住或无法解析的 `check*` 回调会 fail-closed 为 `deny`。Python 回调使用位置参数, +异常会被捕获并视为 Allow。Go handler 接收带类型的 context,报错或超时会 fail-closed。 + +Node 用 `setBudgetGuard(null)` 清除;Python 把 `opts.budget_guard` 设回 `None` 后重建 +session;Go 调用 `session.SetBudgetGuard(ctx, nil)`。 diff --git a/website/docs/v8.5.1/zh/guide/architecture.mdx b/website/docs/v8.5.1/zh/guide/architecture.mdx new file mode 100644 index 00000000..76db476c --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/architecture.mdx @@ -0,0 +1,182 @@ +--- +title: '架构' +description: '会话构建、运行作用域、受控调用、事件协议与持久化' +--- + +# 架构 + +A3S Code 把配置解析、逐次运行、稳定的线协议契约与持久化分成清晰边界。 +TUI 和 SDK 使用同一个运行时内核,不存在另一套执行路径。 + +```text +CodeConfig + SessionOptions + -> 校验并异步解析资源 + -> ResolvedSessionConfig + -> AgentSession + -> single-flight run admission + -> safe-point run-control inbox + -> InvocationContext + ├─ LLM invoker -> provider calls + ├─ tool invoker -> model、nested、delegated 与 host-direct calls + └─ events -> EventEnvelopeV1 -> Rust / Node / Python / Go consumers + -> SessionSnapshotV1 -> atomic store generation +``` + +## 异步优先的会话构建 + +`SessionOptions` 是公开的配置补丁,不是半初始化的运行时状态。异步构建路径会把 +它与 `CodeConfig` 合并,校验冲突选项,初始化异步资源,再生成唯一的内部 +`ResolvedSessionConfig`。后续会话组装只消费这个已解析值,不会让不同 +层各自重复决定同一个配置。 + +Rust 宿主应优先使用: + +```rust +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +`Agent::session_async`、`resume_session_async`、`session_for_agent_async` 与 +`session_for_worker_async` 使用同一个构建内核。文件型记忆和会话存储、队列、 +轨迹记录器与会话 MCP 发现都在异步阶段初始化;失败时返回带具体资源信息的 +`SessionConfiguration` 或 `SessionInitialization` 类型错误。 + +同步 `Agent::session` 只为已经显式传入预初始化记忆存储、并初始化好其他全部资源的 +宿主保留兼容。它不会启动或阻塞 Tokio 运行时。任何需要异步工作的配置都会返回 +`CodeError::AsyncSessionBuildRequired`;运行时不会悄悄换成更容易构建的后端。 +通过 `SessionOptions::with_mcp` 传入的管理器始终需要异步发现能力;同步路径只能 +继承智能体启动时已经缓存的全局 MCP 工具。 + +## 对话状态的单任务约束 + +对话历史通过准入机制串行化,而不是依赖乐观锁。同一个会话同时只能有一个会影响 +对话记录的操作,包括 `send`、`stream`、两种附件变体、斜杠命令与 `resume_run`。 +重叠调用会在读取历史或派发命令前立即返回 +`CodeError::SessionBusy`。 + +流式调用会一直持有准入租约,直到流式运行时真正结束。丢弃或中止公开句柄不会在原 +生产者仍写入事件和历史时短暂放行第二个操作。直接宿主工具调用属于控制面操作, +不占用对话租约。 + +## 调用上下文 + +每个已准入运行都会创建一个不可变的 `InvocationContext`,其中包含: + +- 运行标识与会话标识 +- 运行取消令牌 +- 事件发送器 +- 治理快照,包括当前预算守卫 + +它是服务提供商与工具工作的单一事实来源,也会把同一个取消令牌和会话身份安装进 +`ToolContext`。因此取消可以传递到排队工具、嵌套的 `batch`/`program`、委派任务、 +规划、结构化输出修复、压缩以及其他属于该运行的辅助调用。 + +## 分层提示词与运行连续性 + +默认系统提示词由少量互不重叠的契约组成。共用运行时契约只定义权威边界、自主性、 +证据要求、连续性与停止语义;执行风格提示词只补充该模式需要的行为;响应格式片段 +只描述目标输出。文件系统指令、Skills 与宿主上下文仍是独立输入,因此优化默认提示词 +不会遮蔽运行时策略,也不会移除工具能力。 + +每个活动 Run 都拥有一个有界控制收件箱。`steer` 在下一个服务提供商或工具安全点把 +新用户指令追加到同一份对话记录;`interrupt` 发出协作取消请求,同时让正常清理、 +检查点、钩子和事件持久化完成。幂等键、不可变 Run ID、可选回合修订号与截止时间会 +阻止重试或过期界面状态影响较新的回合。上下文压缩与恢复会保留继续任务所需的目标、 +约束、验收标准、已作决定与验证证据,不会静默改变范围。 + +## 作用域能力组合 + +一个 Session 会将 Tool、Skill、Agent、Command、Hook、MCP、Context、Flow、 +Knowledge 与 UI 贡献发布为单个不可变能力目录代次。准备过程遵循有界依赖图,校验 +覆盖完整投影,最后通过一次比较并交换让新代次可见。准备失败或取消时,已完成的副 +作用会回滚,不会暴露部分目录。 + +每个已准入 Run 都会固定一个投影;当投影来自 A3S Use 时,还会取得与之匹配且不可 +克隆的 Use 快照租约。Turn 与 Subtask 是有类型的子作用域:它们共享 Run 的结构代次, +只能收窄权限,不能重新发现 Session 的更新目录。Run 关闭时会先收敛受监督工作与可 +逆副作用,再释放精确的代次租约。 + +`RunCapabilityBindingV1` 会把 Code 代次、目录摘要、完整权限上限摘要以及可选的 Use +游标写入 Run 与逻辑检查点。因此恢复会在目标 Run 准入前拒绝 N/N+1 漂移,包括准备 +期间发生的切换。尚未发布能力的全新 Session 可以且只能引导一次完整历史批次;已经 +发布能力的 Session 不能回退到旧代次。 + +## 大语言模型调用边界 + +属于运行的服务提供商工作统一经过有作用域的大语言模型调用器。它在每次服务提供商 +调用前检查预算与取消,在成功响应后记录用量;流式路径会代理最终用量,并合并调用方 +取消与运行取消。普通回合、规划、结构化输出及修复、压缩、记忆和辅助路径都使用这个 +边界,不再各自维护预算逻辑。 + +硬预算拒绝会作为错误返回,绝不会转换成不受治理的回退。软上限会发出 +`budget_threshold_hit` 事件,然后继续当前调用。 + +## 工具调用边界 + +工具调用器是模型选择、嵌套、程序化、委派与宿主直接调用的统一治理内核。对模型发起 +的工作,它会执行当前技能限制、权限策略、前后置钩子、预算检查、人工确认、队列与 +超时、取消、递归调用保护及输出净化。`batch` 和 `program` 接收的是有作用域的调用器, +而不是原始注册表,因此内部调用不能绕过这些检查。 + +直接 SDK 辅助方法使用显式的 `HostDirectPolicy::TrustedControlPlane` 来源。宿主已经 +是选择该操作的权威,因此跳过面向模型的权限与人工确认决策;但前置钩子仍可阻止调用, +预算、队列与超时、取消、递归保护、后置钩子和输出净化仍然生效。应用在把这条特权 +路径暴露给终端用户前,必须自行完成授权。 + +## 稳定事件协议 + +`AgentEvent` 是 Rust 内部运行时枚举。跨语言契约是无损信封: + +```json +{ + "version": 1, + "type": "tool_end", + "payload": {}, + "metadata": {} +} +``` + +v1 事件目录与 Rust 穷尽映射共享一个事实源,因此新增运行时变体却没有规范线协议 +名称会直接导致编译失败。Node.js 与 Python 使用统一投影生成 `text`、`toolName`、 +`tool_name` 等便捷字段。`type` 保持开放字符串:未来未知类型仍会完整保留载荷与 +元数据,不会被压成 `unknown` 哨兵值。 + +## 原子会话持久化 + +`SessionSnapshotV1` 表示一个带版本号的完整持久化代次,包含对话、制品、追踪事件、 +运行记录、验证报告与委派任务快照。`session.save()` 会物化这个聚合快照,并且只调用一次 +`SessionStore::save_snapshot`。 + +文件存储先写入并同步临时文件,再通过原子替换发布一个完整 JSON 信封;内存存储在 +同一把锁下替换整个聚合条目。两者都会声明原子快照能力。历史裸 `SessionData` 与 +分片目录仍可读取以便迁移,但新保存不会发布碎片化代次。自定义存储必须显式实现 +聚合保存;默认方法返回错误,不会把部分写入或空操作写入当成成功。 + +## MCP 所有权与隔离 + +MCP 管理器的所有权是显式的: + +1. 智能体全局管理器持有从全局配置加载的服务器。 +2. 会话选项中由宿主传入的管理器是继承的只读能力来源。 +3. 每个会话都新建一个私有实时管理器。 + +能力按以上顺序组装,因此会话本地工具只能在当前会话内遮蔽继承工具。实时 +`add_mcp_server` / `remove_mcp_server` 只改变私有管理器;移除本地遮蔽后会重新显露 +继承能力。同级会话不能互相修改;委派的子智能体会继承调用同一批工具所需的有序 +管理器来源,但不会取得它们的所有权。 + +## 程序化工具调用 + +`program` 工具在内嵌 QuickJS 虚拟机中运行 JavaScript,只暴露受控 `ctx` 对象。虚拟机 +没有直接文件系统、网络、子进程或环境变量访问权;有用能力都通过 scoped tool +调用器回到 A3S Code。下一步需要判断时使用普通模型工具调用;动作序列已经确定、 +只需让模型理解结果时,使用有边界的程序。 + +## 扩展点 + +通过带类型的会话选项、技能、智能体定义、钩子、MCP 服务器、记忆和会话存储、 +安全提供程序、队列配置与工作区服务扩展运行时。优先使用显式策略和可回放证据, +不要引入平行执行路径。 diff --git a/website/docs/v8.5.1/zh/guide/cluster-extension-points.mdx b/website/docs/v8.5.1/zh/guide/cluster-extension-points.mdx new file mode 100644 index 00000000..d765ff98 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/cluster-extension-points.mdx @@ -0,0 +1,598 @@ +--- +title: '集群扩展点' +description: '集群宿主用来在多节点上运行长时会话、且无需分叉框架的接缝。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 集群扩展点 + +集群宿主平台在多个节点上运行长时运行的智能体会话。框架本身并不附带调度器或放置引擎,而是暴露一小组接缝:它定义决策点、发出结构化事件,并把策略交给宿主来提供。下文的所有内容都是你从框架外部接入的——你永远不需要分叉框架。 + +本页给出 Rust、Node.js、Python 和 Go 中等价的原生接缝。配置形式遵循各语言习惯, +底层能力和生命周期契约保持一致。 + +## 身份标签 + +每个会话都可以携带四个不透明的身份标签。框架从不解释它们——它会把它们传播到钩子、追踪和 `SessionData`,并在恢复时还原它们。宿主正是借此把一个会话归属到租户、主体、智能体模板以及更广的关联链。 + +请将身份标签与 `sessionStore` / `session_store` 搭配使用,使标签在进程重启后依然保留。恢复时,**由调用方提供的选项优先生效**,因此你可以在节点之间迁移会话时为其重新打标签。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let options = SessionOptions::new() + .with_tenant_id("tenant-example") + .with_principal("principal-example") + .with_agent_template_id("agent-template-example") + .with_correlation_id("trace-example") + .with_file_session_store("./.a3s/sessions"); + let session = agent + .session_builder("/path/to/project") + .options(options) + .build() + .await?; + + println!("tenant: {:?}", session.tenant_id()); + println!("principal: {:?}", session.principal()); + println!("template: {:?}", session.agent_template_id()); + println!("correlation: {:?}", session.correlation_id()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +const session = agent.session('/path/to/project', { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', +}); + +// 读取器返回 string | null +console.log(session.tenantId); // 'tenant-example' +console.log(session.principal); // 'principal-example' +console.log(session.agentTemplateId); // 'agent-template-example' +console.log(session.correlationId); // 'trace-example' +``` + + + + +```python +opts = SessionOptions() +opts.tenant_id = 'tenant-example' +opts.principal = 'principal-example' +opts.agent_template_id = 'agent-template-example' +opts.correlation_id = 'trace-example' +session = agent.session('/path/to/project', opts) + +# 读取器是属性,返回 str | None +print(session.tenant_id) # 'tenant-example' +print(session.principal) # 'principal-example' +print(session.agent_template_id) # 'agent-template-example' +print(session.correlation_id) # 'trace-example' +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func value(value *string) string { + if value == nil { + return "" + } + return *value +} + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, "/path/to/project", &code.SessionOptions{ + TenantID: "tenant-example", + Principal: "principal-example", + AgentTemplateID: "agent-template-example", + CorrelationID: "trace-example", + FileSessionStoreDir: "./.a3s/sessions", + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + info, err := session.Info(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("tenant:", value(info.TenantID)) + fmt.Println("principal:", value(info.Principal)) + fmt.Println("template:", value(info.AgentTemplateID)) + fmt.Println("correlation:", value(info.CorrelationID)) +} +``` + + + + +## 预算 / 成本守卫 + +预算守卫让宿主针对成本或令牌预算对每一次 LLM 调用进行把关。框架会在每次 LLM 请求*之前*调用你的守卫,并在请求返回*之后*再次调用。守卫是你自己拥有的策略;框架只负责执行你返回的决策。 + + + + +```rust +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; + +use a3s_code_core::{ + budget::{BudgetDecision, BudgetGuard}, + llm::TokenUsage, + Agent, SessionOptions, +}; + +#[derive(Debug)] +struct TokenGuard { + limit: usize, + used: AtomicUsize, +} + +#[async_trait::async_trait] +impl BudgetGuard for TokenGuard { + async fn check_before_llm( + &self, + _session_id: &str, + estimated_prompt_tokens: usize, + ) -> BudgetDecision { + let used = self.used.load(Ordering::Relaxed); + if used.saturating_add(estimated_prompt_tokens) > self.limit { + BudgetDecision::Deny { + resource: "tokens".into(), + reason: "monthly cap reached".into(), + } + } else { + BudgetDecision::Allow + } + } + + async fn record_after_llm(&self, _session_id: &str, usage: &TokenUsage) { + self.used.fetch_add(usage.total_tokens, Ordering::Relaxed); + } +} + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let guard = Arc::new(TokenGuard { + limit: 50_000, + used: AtomicUsize::new(0), + }); + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_budget_guard(guard)) + .build() + .await?; + + let result = session.send("总结这个仓库。", None).await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +session.setBudgetGuard({ + checkBeforeLlm(ctx) { + if (overLimit(ctx.sessionId, ctx.estimatedTokens)) { + return { + decision: 'deny', + resource: 'tokens', + reason: 'monthly cap reached', + }; + } + return { decision: 'allow' }; + }, + recordAfterLlm(ctx) { + meter(ctx.sessionId, ctx.usage); + }, +}); + +// 清除预算守卫 +session.setBudgetGuard(null); +``` + +Node 回调接收单个 `ctx` 对象,且**绝不能抛出异常**。请用 `try/catch` 包裹逻辑并返回显式决策。卡住或无法解析的 `check*` 回调会 fail-closed 为 `deny`。 + + + + +```python +class MyGuard: + def check_before_llm(self, session_id, estimated_tokens): + if over_limit(session_id, estimated_tokens): + return {'decision': 'deny', 'resource': 'tokens', 'reason': 'monthly cap reached'} + return {'decision': 'allow'} + + def record_after_llm(self, session_id, usage): + meter(session_id, usage) + +opts = SessionOptions() +opts.budget_guard = MyGuard() +session = agent.session('/path/to/project', opts) + +# 如需清除:把 opts.budget_guard 设为 None,再重新创建会话。 +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "sync/atomic" + "time" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + const limit uint64 = 50_000 + var used atomic.Uint64 + err = session.SetBudgetGuard(ctx, &code.BudgetGuardHandlers{ + CheckBeforeLLM: func( + _ context.Context, + call code.BudgetLLMContext, + ) (*code.BudgetDecision, error) { + if used.Load()+uint64(call.EstimatedTokens) > limit { + return &code.BudgetDecision{ + Decision: "deny", + Resource: "tokens", + Reason: "monthly cap reached", + }, nil + } + return &code.BudgetDecision{Decision: "allow"}, nil + }, + RecordAfterLLM: func( + _ context.Context, + call code.BudgetUsageContext, + ) error { + used.Add(uint64(call.Usage.TotalTokens)) + return nil + }, + Timeout: 2 * time.Second, + }) + if err != nil { + log.Fatal(err) + } + defer session.SetBudgetGuard(context.Background(), nil) + + result, err := session.Run(ctx, "总结这个仓库。") + if err != nil { + log.Fatal(err) + } + fmt.Printf("%s\nspent %d / %d tokens\n", result.Text, used.Load(), limit) +} +``` + +Go 回调遵循相同的 fail-closed 契约:回调报错、超时或返回无法解析的决策时, +受保护的 LLM 或工具调用会被拒绝。 + + + + +四种 SDK 的原生守卫具有等价的决策结构: + +| 返回值 | 效果 | +| ----------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | +| `None` / `null` / `{ decision: 'allow' }` | 继续执行 LLM 调用。 | +| `{ decision: 'soft', resource, consumed, limit, message? }` | 发出 `BudgetThresholdHit`(kind 为 `soft`)并继续执行。 | +| `{ decision: 'deny', resource, reason }` | 中止 LLM 调用。Python 抛出 `RuntimeError("Budget exhausted...")`;Node 以 `"Budget exhausted..."` 拒绝(reject)。 | + +这种健壮性是刻意为之,但各 SDK 的失败语义略有不同:**缺失的守卫方法**会被当作宽松默认值处理。Python 回调出错会回退为 Allow。Node 回调不能 throw。Go 回调报错、超时或返回无法解析的 `check*` 决策时,会 fail-closed 为 `deny`。 + +## 集群事件词汇 + +宿主通过其钩子执行器,将集群级别的决策作为结构化的 `AgentEvent` 变体发出。会话内的钩子以统一方式订阅它们——与它们观察其他任何事件的方式相同——因此在宿主处编写的策略会原样呈现给智能体自身的钩子,无需特殊处理。 + +集群词汇如下: + +- **`BudgetThresholdHit { resource, kind, consumed, limit, message? }`** —— 预算守卫返回了 `soft` 决策(或宿主越过了它自己跟踪的某个阈值)。`kind` 用于区分软性警告与更硬性的限制。 +- **`PassivationRequested { reason, deadline_ms? }`** —— 宿主请求会话进入一个安全、可持久化的状态,以便将其从当前节点驱逐。`deadline_ms` 若存在,则表示强制驱逐前的宽限窗口。 +- **`PeerInvocation { from_session_id, from_tenant_id?, correlation_id? }`** —— 另一个会话调用了本会话。这些标签让接收方能够把调用归属回其源租户和关联链。 + +这些事件通过你的会话内钩子已经在使用的、经过验证的同一套钩子 API 来观察——Node 中为 `session.registerHook`,Python 中为 `session.register_hook`(参见[钩子](/guide/hooks))。请将上述三个变体视为已记录在案的契约;宿主负责通过其钩子执行器发出它们。 + +## 确定性 ID 与时间(重放) + +希望在另一节点上对某次运行进行**逐位一致重放**的集群,必须消除常规运行中两处不确定性的来源:随机 ID 和挂钟时间。Rust 核心将二者建模在一个 `HostEnv { id_generator, clock }` 之后。默认实现把 UUID 生成器与系统时钟配对;重放工具会换入 `SequentialIdGenerator` 和 `FixedClock`,使得对相同输入的重新执行在任意节点上都产生相同的 ID 和时间戳,从而产生相同的输出。 + +四种 SDK 都提供同一确定性配置。请同时设置 ID 前缀和固定时间戳,并在每次重放开始时 +用相同的值重新创建配置。 + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{ + host_env::{FixedClock, HostEnv, SequentialIdGenerator}, + Agent, SessionOptions, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let host_env = Arc::new(HostEnv::new( + Arc::new(SequentialIdGenerator::new("replay")), + Arc::new(FixedClock::new(1_700_000_000_000)), + )); + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_host_env(host_env)) + .build() + .await?; + + println!("{}", session.session_id()); + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = await agent.sessionAsync('.', { + hostEnv: { + sequentialIdPrefix: 'replay', + fixedTimeMs: 1_700_000_000_000, + }, +}); + +console.log(session.sessionId); +await session.closeAsync(); +await agent.close(); +``` + + + + +```python +from a3s_code import Agent, HostEnvConfig, SessionOptions + +agent = Agent.create('agent.acl') +options = SessionOptions() +options.host_env = HostEnvConfig( + sequential_id_prefix='replay', + fixed_time_ms=1_700_000_000_000, +) +session = agent.session('.', options) + +print(session.session_id) +session.close() +agent.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.NewAgent(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + prefix := "replay" + fixedTime := uint64(1_700_000_000_000) + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + HostEnv: &code.HostEnvConfig{ + SequentialIDPrefix: &prefix, + FixedTimeMS: &fixedTime, + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + fmt.Println(session.SessionID()) +} +``` + + + + +## 循环检查点与运行恢复 + +配置了 `sessionStore` / `session_store` 后,智能体循环会在**每一轮工具调用完成之后**持久化一个检查点,以运行 id 作为键。任何共享同一存储的节点都可以重新水合该运行并继续它。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_file_session_store("./.a3s/sessions") + .with_session_id("session-from-node-a"), + ) + .build() + .await?; + + let result = session.resume_run("run-id-from-node-a").await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session(workspace, { + sessionStore: new FileSessionStore('./.a3s/sessions'), + sessionId: 'session-from-node-a', +}); + +const result = await session.resumeRun('run-id-from-node-a'); +``` + + + + +```python +from a3s_code import FileSessionStore + +opts = SessionOptions() +opts.session_store = FileSessionStore('./.a3s/sessions') +opts.session_id = 'session-from-node-a' +session = agent.session(workspace, opts) + +result = session.resume_run('run-id-from-node-a') +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + options := &code.SessionOptions{ + SessionID: "session-from-node-a", + FileSessionStoreDir: "./.a3s/sessions", + } + session, err := agent.Session(ctx, ".", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + result, err := session.ResumeRun(ctx, "run-id-from-node-a") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +系统会为恢复的工作分配一个**新的运行 id**——存储中的原始运行保持不变。有两条错误路径值得处理: + +- **`resume_run requires a session_store`** —— 未配置存储;回退到一个全新会话。 +- **`no loop checkpoint found for run 'X'`** —— 该运行从未到达其第一个检查点,或已被清理;稍后重试,或将该运行视为丢失。 + +由于检查点只在**工具轮次之间、绝不在工具执行中途**生成,恢复的运行永远不会重放一个执行到一半的工具。存储细节参见[持久化](/guide/persistence)。 + +## 长时运行会话的保留上限 + +运行数小时或数天的会话会在四个内存存储中累积状态:运行记录、每次运行的事件缓冲区、追踪事件,以及终态子智能体任务快照。若不加限制,它们会随会话寿命增长——对短寿命会话无妨,对长寿命会话则是真实的泄漏。 + +`SessionRetentionLimits` 为这些存储设置上限。省略字段时保留框架的有限默认值; +只有明确需要无限保留时才设置 `unbounded: true`。驱逐采用严格的 **FIFO**,并且 +**正在运行的子智能体任务永不被丢弃**——只有终态快照会被驱逐。 + +Node 使用 `retentionLimits`,Python 使用 `opts.retention_limits`,Go 使用 +`SessionOptions.RetentionLimits`。Rust 宿主使用 +`SessionOptions::with_retention_limits(...)`。字段名和示例见[限制](/guide/limits)。 + +--- + +**另见:** [多机部署](/guide/multi-machine) · [持久化](/guide/persistence) · [限制](/guide/limits) · [钩子](/guide/hooks) diff --git a/website/docs/v8.5.1/zh/guide/commands.mdx b/website/docs/v8.5.1/zh/guide/commands.mdx new file mode 100644 index 00000000..9fee7988 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/commands.mdx @@ -0,0 +1,59 @@ +--- +title: '命令' +description: '命令表面与推荐控制流' +--- + +# 命令 + +A3S Code 主要是 SDK 驱动。CLI 和 UI 通常把命令映射到 session API,而不是依赖 +一个庞大的公开命令协议。 + +本页说明 SDK 命令注册表。`a3s code` 终端应用中的内置命令见 +[A3S Code TUI](/guide/tui)。 + +这两个 surface 刻意不同: + +| Surface | 所有者 | 用途 | +| ------------------------- | -------- | -------------------------------------------------------------------------------------------------- | +| `a3s code` slash commands | CLI/TUI | `/model`、`/flow`、`/memory`、`/kb`、`/update`、`/exit` 等终端控制。 | +| SDK 命令注册表 | 宿主应用 | 用 `session.registerCommand(...)` 注册、再通过 `session.send("/name args")` 触发的产品自定义命令。 | + +SDK command 会在 LLM 看到输入之前执行。handler 接收原始参数字符串和 session +元数据,并返回展示文本。命令适合薄薄的控制面动作;workflow、工具、验证和持久化 +应直接使用普通 SDK 方法。 + +| 用户动作 | Session API | +| ---------------------- | ----------------------------------------------------------------- | +| 发送 prompt | `session.send(prompt)` | +| 流式 prompt | `session.stream(prompt)` | +| 临时问题 | 创建隔离 session,或用显式隔离的 history 调用 `send` / `stream`。 | +| 直接工具 | `session.tool(name, args)` | +| 历史 | `session.history()` | +| 保存 | `session.save()` | +| 恢复 | `agent.resumeSession(id, options)` | +| 工具列表 | `session.toolNames()` / `session.toolDefinitions()` | +| 验证 | `session.verifyCommands(subject, commands)` | +| 回放 run 状态 | `session.runs()` / `session.runEvents(runId)` | +| 检查或取消 current run | `session.currentRun()` / `session.cancelRun(runId)` | + +如果产品提供 slash command,请让它们薄薄地映射到这些 API。注册 handler、列出 +可用命令,并通过 `session.send()` 触发: + +```ts +session.registerCommand( + 'docs_status', + 'Return docs command status', + (args, ctx) => { + return `status args=${args}; session=${ctx.sessionId}; workspace=${ctx.workspace}`; + }, +); + +console.log(session.listCommands()); + +const result = await session.send('/docs_status check'); +console.log(result.text); +``` + +command handler 不应隐藏长时间运行的工作。如果一个命令需要运行工具、验证输出、 +fan out 到 subagent 或调度周期性自动化,让 handler 返回简短确认,再由明确的宿主 +代码执行这些工作。这样取消、审计和权限边界仍然可见。 diff --git a/website/docs/v8.5.1/zh/guide/context.mdx b/website/docs/v8.5.1/zh/guide/context.mdx new file mode 100644 index 00000000..f169c0fc --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/context.mdx @@ -0,0 +1,221 @@ +--- +title: '上下文' +description: '可靠编程智能体的预算化上下文组装' +--- + +# 上下文 + +A3S Code 把上下文当作有预算的资源。 + +上下文来源包括用户 prompt、历史、AGENTS.md、skills、memory、文件搜索、直接工具结果、MCP、委派子运行摘要和 trace event。AGENTS.md 注入、skills discovery、memory API、直接工具结果、委派 helper 和 `traceEvents()` 属于 Node SDK 文档表面;MCP context 行为应针对你的 live integration 单独验证。 + +```text +sources -> ContextItem -> rank -> dedupe -> budget -> render +``` + +长 grep 输出、日志和委派子运行 transcript 不应直接塞进 prompt;应保存在 prompt 之外,再总结成适合 prompt 的证据。紧凑运行证据可通过 `session.traceEvents()` 获取。 + +## 精确认知包(Rust 宿主) + +嵌入式 Rust 宿主可以把会话绑定到某个精确的 A3S Use cognitive-package 代。A3S Code +不会安装包、解析 Registry 条目或选择 `latest`;宿主负责注入不可变的 +`CognitivePackageBindingV1`,以及持有对应 Knowledge lease 的 Provider。 + +```rust +use a3s_code_core::{CognitiveContextSession, SessionOptions}; + +let cognitive_context = CognitiveContextSession::new(binding, provider)?; +let options = SessionOptions::new().with_cognitive_context(cognitive_context); +``` + +持久化的 `a3s.code.cognitive-package-session-binding.v1` 身份包含包 ID/版本、生命周期代、 +代摘要、能力快照摘要、精确 Knowledge surface 和 prompt 注入上限。每个带类型的请求与 +带引用的 Markdown 响应都会重复该绑定,并在内容进入模型上下文前完成校验。 + +硬上限是 4 个文档、每个文档 6 KiB、总计 6 KiB;宿主可在绑定中选择更小的限制。 +Provider 失败、引用非法、请求不匹配或代身份漂移都会关闭失败,不会回退到无关检索。 + +必须使用 `with_cognitive_context`。通过通用 context-provider 列表加入认知 Provider 会 +被拒绝,因为这种方式无法正确持久化绑定。精确认知包不能与通用 RAG 或图 Provider +共存,并且会抑制个人 memory 召回;Code 自己管理的工作区指令与 skills 仍然可用。 + +绑定会写入会话快照,并发出 `cognitive_context_bound` 事件。恢复时,宿主必须重新注入 +具有相同绑定的 Provider;缺失绑定或使用不同的代都会被拒绝。这个带类型的边界目前是 +Rust 宿主集成面,不是 Node.js、Python 或 Go 的会话选项。 + + + +## Session 绑定的工作区检索 + +A3S Code 8 可以为一个工作区的一个 Session 构建有界检索索引。构建过程异步运行, +结果会与当前源文件再次校验,Session 关闭时所有向量都会释放。Runtime 不会安装或 +要求使用向量数据库。 + +Workspace Retrieval 是宿主能力,不是模型可以自行打开的开关。省略 Typed Option +即可保持关闭。关闭状态不会构建额外目录、不会调用 Embedding Provider,也不会向 +模型暴露 Semantic 或 Hybrid Search Mode。 + +启用后不会另外注册一个 `vector_db` 工具。模型侧唯一的 `search` Schema 会增加 +`mode: "semantic"` 与 `mode: "hybrid"`。这两个模式和 Grep、Glob、BM25 一样经过 +统一的受治理工具路径,并查询 Session 独占的 Projection;关闭状态会直接从 Schema +移除这两个模式,而不是暴露无法执行的调用。 + +### 选择能完成任务的最小检索面 + +| 需求 | 使用方式 | 是否需要 Embedding 模型 | +| ------------------------ | ------------------- | ------------------------------------ | +| 已知标识符或精确文本 | Exact、Glob 或 Grep | 不需要 | +| 用自然语言查询项目文本 | 增量 BM25 | 不需要 | +| 定义、引用或诊断 | Code Intelligence | 不需要 | +| 词汇不一致或跨语言语义 | Semantic Retrieval | 需要宿主提供回调 | +| 同时包含标识符和自然语言 | Hybrid RRF | 可选;没有向量时仍保留词法与符号通道 | +| 减少近重复证据 | 确定性 Reranker | 不需要;本地 CPU 算法 | + +Dense Semantic Search 必然需要 Text-to-vector 函数,但该函数可以是进程内 CPU +回调。A3S Code 不要求远程 API、GPU、内置模型或运行时下载。模型 Revision、License、 +Artifact 校验、Cache 与凭据仍由宿主负责。 + +### 异步向量投影生命周期(非持久化向量库) + +这里使用的是 Session 独占的精确内存向量索引,不是持久化或跨 Session 共享的 +Vector Database。Session 构建会在 Corpus Embedding 完成前返回;后台 Indexer 读取 +已准入文本,以 File 为单位原子发布不可变 Partition,并在源文件 Revision 变化后异步 +Reconcile,未变化的文件不会重复 Embedding。重新创建 Session 会生成新的 Projection; +关闭 Session 会取消未完成的 Provider Work,并要求 Vector Record 与已计量 Byte 归零。 + +```text +构建 Session -> 立即返回 + `-> 后台构建文本目录与向量 Partition + +查询 -> 融合独立 Rank -> 校验当前源 -> 输出证据 + +关闭 Session -> 取消 Provider -> Join Indexer -> 释放全部向量 +``` + +状态依次为 `building`、`ready`、`degraded` 和 `closed`。构建期间,查询可以使用已经 +发布的 Coverage。需要更强首查边界的宿主可等待 Ready,最长 30 秒;超时后保留 +Partial Fallback,取消查询或关闭 Session 会中断等待。 + +| 状态 | 向量投影 | 查询行为 | +| ---------- | ------------------------------------- | ----------------------------------------------------- | +| `building` | 完成的 File Partition 会逐个原子发布 | Exact/BM25 始终可用,Semantic Coverage 可能不完整 | +| `ready` | 当前源 Generation 已达到完整 Coverage | Semantic 与 Hybrid Query 使用完整的已发布 Generation | +| `degraded` | 保留有效 Partition,并报告有界失败 | 词法路径继续服务,Semantic Result 显示 Partial Status | +| `closed` | 取消 Indexing 并释放全部向量 | 再次使用 Semantic Retrieval 前必须重建 Session | + +### 排序 + +同一个不可变 Chunk Catalog 支持增量 BM25、可选的精确内存向量、稳定源 Anchor, +以及 Exact-literal 或 Code Intelligence Candidate。Hybrid Mode 使用 +Reciprocal-rank Fusion(`k = 60`)合并各通道从 1 开始的独立 Rank,不混合不可比较的 +Raw Score。RRF-only 是默认值;可选确定性 Reranker 是有界、无模型的 CPU 代码, +用于减少重复证据并保护精确 Identifier。 + +### 文本准入与切块 + +只有 Manifest 准入的 UTF-8 文本和源文件会进入目录。Generated、Oversized、 +Credential、Key Material、`.a3s` 控制路径和非文本 Asset 会在切块与 Embedding 之前 +排除。PDF、Office、图片、音频、OCR 及其他知识编译属于独立 Knowledge Compiler; +Workspace Retrieval 不会猜测这些格式的解析方式。 + +内置 Typed Strategy 包括 Line/byte、Fixed UTF-8 Window 与 Recursive Separator。 +可信 Rust 宿主可以提供 Custom Splitter,但 Range 必须保持 UTF-8 Boundary、覆盖准入 +Byte 并始终向前推进。Node.js、Python 与 Go 接受带类型的内置 Strategy Object; +Primitive Strategy Name 会被拒绝。 + +### SDK 控制面 + +| 宿主 | 开启 | 保持关闭 | 查询与状态 | +| ------- | --------------------------------------------------- | ------------------------------- | ------------------------------------------------------------------ | +| Rust | `SessionOptions::with_workspace_retrieval(...)` | `without_workspace_retrieval()` | `workspace_retrieval_status`、`semantic_search`、`hybrid_search` | +| Node.js | 在 Session Option 中设置 Typed `workspaceRetrieval` | 省略 | `workspaceRetrievalStatus()`、`semanticSearch()`、`hybridSearch()` | +| Python | 设置 `SessionOptions.workspace_retrieval` | 赋值 `None` | `workspace_retrieval_status()`、异步 Semantic 与 Hybrid Search | +| Go | 设置 `SessionOptions.WorkspaceRetrieval` | 使用 `nil` | `WorkspaceRetrievalStatus`、`SemanticSearch`、`HybridSearch` | + +### CLI 启用方式 + +`a3s` CLI 默认关闭语义检索。只有受信任的用户 ACL,或通过 `--config` 显式选择的文件, +才能启用它。自动发现的工作区 `.a3s/config.acl` 只能关闭继承的检索路由,不能授权源代码 +出站,也不能选择 Embedding Backend。 + +远程 Embedding 需要独立 Provider 路由,并显式授权源代码出站: + +```acl +workspace_retrieval { + enabled = true + allow_source_egress = true + model = "openai/text-embedding-3-small" + dimension = 1536 + normalization = "none" +} +``` + +本地 CPU Embedding 与远程字段互斥,不需要出站授权: + +```acl +workspace_retrieval { + enabled = true + semantic_readiness_timeout_ms = 30000 + + local_cpu { + artifact_manifest = "models/multilingual-mini/model.acl" + intra_threads = 2 + } +} +``` + +本地 Artifact Manifest 会锁定 Revision 与 SHA-256,Runtime 不会下载模型文件。创建 +Session 前,应运行 `a3s config validate`,并检查 `a3s config show` 中已脱敏的 +`workspaceRetrieval` 段。Embedding 路由与 `default_model` 相互独立,因此选择 DeepSeek +执行 Chat 和 Tool Call,不会把该 Chat Endpoint 自动当作 Embedding 服务。 + +CLI 的 `local_cpu` 适配器支持 Linux x64/ARM64、Windows x64 和 Apple Silicon。Intel +macOS 12(`x86_64`)构建会刻意省略可选的 ONNX 适配器。在 Intel 平台请保持 +Model-free Retrieval,或使用单独明确授权的远程 Embedding 路由。 + +Provider Descriptor 会锁定 Identity、Model、Dimension 与 Normalization。Runtime 会 +拒绝 Partial、Duplicate、Unknown、Dimension-mismatched、Non-finite、 +Non-normalized 或 Descriptor-drifted Response。Diagnostic 不复制输入文本、向量、 +远程响应体、凭据或 Endpoint Value。 + +### 质量与安全证据 + +Status Snapshot 会报告 Coverage、Queue Depth、Failure、Vector Memory、Batching、 +Request Amplification、Non-text Provider Input 与关闭后的资源释放。发布评估使用 +Recall@5、MRR、Latency、Memory、Non-text Egress 和 Cleanup。锁定的 Cross-SDK +DeepSeek Fixture 完成 3/3 精确任务,Recall@5 为 1.0、MRR 为 0.5、Document Request +Amplification 为 1.0x、Non-text Input 为零,并完整释放向量。它是 Portability Gate, +不代表某个模型或 Reranker 对所有仓库都最优。 + +Code `5aa9642` 在 2026-08-17 完成了 `v7.0.1` 发布后复验: + +| SDK | 精确任务 / 单次 Search 协议 | DeepSeek Turn p50 / p95 | Tokens | +| ------- | --------------------------: | ----------------------: | -----: | +| Node.js | 3 / 3 | 16,033 / 16,538 ms | 14,540 | +| Python | 3 / 3 | 15,552 / 23,751 ms | 14,784 | +| Go | 3 / 3 | 16,636 / 19,009 ms | 14,171 | + +三个 SDK 均保持 Recall@5 1.0、MRR 0.5、1.0x Document Request Amplification、 +零 Non-text Input 与完整的关闭后释放。这些远程耗时只用于诊断,不是本地检索延迟目标。 + +渲染结果前,A3S Code 会重新读取权威文件,并校验 Full-file Digest 与精确 Chunk Byte +Range。Deleted、Stale、Unreadable 或 Superseded Candidate 不会暴露。生产阈值、回滚 +规则和可复现评估见 +[运维手册](https://github.com/A3S-Lab/Code/blob/main/manual/WORKSPACE_RETRIEVAL_OPERATIONS.md) +与[资格记录](https://github.com/A3S-Lab/Code/blob/main/manual/WORKSPACE_RETRIEVAL_QA.md)。 + +## 上下文压缩 + +长会话可启用自动压缩: + +```ts +const session = agent.session('/repo', { + autoCompact: true, + autoCompactThreshold: 0.75, + maxContextTokens: 128_000, +}); +``` + +未设置 `maxContextTokens` 时,Core 会优先使用当前模型声明的上下文窗口。每次模型请求前,Core 会统计 system prompt、会话消息、工具调用与结果以及暴露给模型的工具 schema;达到阈值后,它会限制过大的工具输出、总结较早且边界安全的消息、保留近期消息,并继续当前任务。压缩摘要还会参与后续压缩,因此长会话可以持续滚动压缩;这不会扩大模型单次请求的物理上下文窗口。压缩成功时,`context_compacted` 事件会携带累计摘要,使用外部历史的宿主可以把同一压缩代持久化到后续轮次。 + +Python SDK 中对应的字段是 `max_context_tokens`。 diff --git a/website/docs/v8.5.1/zh/guide/convention-over-configuration.mdx b/website/docs/v8.5.1/zh/guide/convention-over-configuration.mdx new file mode 100644 index 00000000..c12610a3 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/convention-over-configuration.mdx @@ -0,0 +1,230 @@ +--- +title: '约定大于配置' +description: '用文件系统优先的智能体、工具、连接、子智能体、调度、持久化和人工确认构建长期工作单元。' +--- + +# 约定大于配置 + +A3S Code 的约定大于配置能力不是一个单独的运行器,而是一套长期智能体组合: +[文件系统优先](/guide/filesystem-first)约定、会话运行时、工具权限、 +多智能体委派、可恢复编排、人工确认、运行回放和调度守护进程。它让一个智能体 +从“能聊天的 SDK 会话”变成“有角色、有工具、有团队、有状态、有接管点的工作单元”。 + +如果你在评估目录优先的智能体框架,可以把 A3S Code 看成更底层、更可嵌入的运行时。 +目录约定、工具、连接、子智能体、调度、持久化和观测都存在,但不会强制绑定到某个 +前端渠道或部署平台;宿主可以把它接入托管会话、开放平台 API、MCP、A3S Box 或 +自己的任务系统。 + +## 能力映射 + +| 智能体框架概念 | A3S Code 对应能力 | 说明 | +| ---------------- | ---------------------------------------------------------- | ---------------------------------------------------------------------------------------- | +| 智能体是一个目录 | [智能体目录](/guide/agent-dir) | `instructions.md`、`agent.acl`、`skills/`、`tools/`、`schedules/` 合成一个可运行智能体。 | +| Markdown 指令 | `instructions.md`、`AGENTS.md`、提示词插槽 | 指令作为插槽注入,不能覆盖驾驭层的边界、响应契约和安全门。 | +| Markdown 技能 | [技能](/guide/skills) | 文件型技能、内联技能和显式注册表使用同一套发现语义。A3S Code 不再内置默认技能。 | +| TypeScript 工具 | `program`、直接工具、MCP 工具、智能体目录脚本工具 | A3S Code 更关注工具注册、权限门、带类型错误和验证证据;脚本工具运行在 QuickJS 受限环境。 | +| 沙箱 | `program` 沙箱、权限策略、人工确认、A3S Box | A3S Code 控制工具权限和脚本沙箱;需要进程、文件系统、网络级隔离时接 A3S Box。 | +| 渠道 | 托管会话、开放平台 WebSocket/SSE、宿主自有界面 | A3S Code 不把渠道写死,宿主通过流式事件和运行回放接到任意前端。 | +| 连接 | MCP、服务提供商配置、宿主注入凭据 | 连接由宿主配置和授权,不建议把令牌写进智能体目录。 | +| 子智能体 | [任务](/guide/tasks)、[团队](/guide/teams)、`workerAgents` | 支持聚焦或多项 `task` 调用、自动委派和动态工作智能体。 | +| 调度 | 智能体目录 `schedules/` + `serve_agent_dir` | 每个调度有独立会话标识,可配合会话存储在重启后恢复上下文。 | +| 持久执行 | `SessionStore`、运行回放、`parallelResumable` | 会话历史、运行事件和可恢复工作流能落盘;失败步骤可在恢复时重试。 | +| 人工确认 | [安全](/guide/security)、确认继承、工具确认 | 高风险工具可请求确认;子运行可配置自动批准、遇到询问时失败或继承父级策略。 | +| 评测 | [验证](/guide/verification)、报告、回归证据 | A3S Code 把评测产品化为验证命令、证据摘要、运行回放与发布门禁,而不是单独的评测套件。 | + +## 最小工作形态 + +交互式智能体团队可以从一个普通仓库开始: + +```text +repo/ +├── agent.acl +├── AGENTS.md +├── .a3s/ +│ ├── agents/ +│ │ ├── release-reviewer.md +│ │ ├── security-reviewer.md +│ │ └── verification-runner.md +│ └── skills/ +│ └── release-readiness.md +└── src/ +``` + +`agent.acl` 负责模型、服务提供商、并行度和自动委派: + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +agent_dirs = ["./.a3s/agents"] + +auto_delegation { + enabled = true + min_confidence = 0.72 + max_tasks = 4 + auto_parallel = false +} +``` + +宿主启动一个会话,并把技能、智能体目录、持久化和委派策略接进去: + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + skillDirs: ['./.a3s/skills'], + agentDirs: ['./.a3s/agents'], + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, + autoSave: true, +}); + +const result = await session.send(` +准备一次发布就绪检查: +1. 让探索角色找出高风险变更。 +2. 让安全角色检查权限、secret、外部副作用。 +3. 让验证角色给出必须跑的回归命令。 +4. 合并成一个按阻塞程度排序的报告。 +`); + +console.log(result.text); +console.log(result.verificationSummaryText); +``` + +## 智能体目录形态 + +需要长期运行、调度或目录级工具时,使用文件系统优先的智能体: + +```text +release-agent/ +├── instructions.md +├── agent.acl +├── skills/ +│ └── release-readiness.md +├── schedules/ +│ └── daily.md +└── tools/ + ├── github.md + └── search-auth.md +``` + +`instructions.md` 是角色插槽: + +```md +You are a release-readiness agent for this repository. +Always separate blockers from follow-up work. +Never invent versions or CI status. Read evidence from the workspace. +``` + +`schedules/daily.md` 是周期性回合: + +```md +--- +cron: '0 9 * * *' +name: daily-release-check +enabled: true +--- + +Summarize merged changes since the last run, inspect release risks, +and report only blockers plus required verification. +``` + +`serve_agent_dir` 会为每个调度创建独立会话。传入 `SessionStore` 后,重启只重新加载 +当前目录配置和工具,历史上下文从存储中恢复。 + +## 工具与连接 + +A3S Code 的工具面分三层: + +| 层 | 用法 | 适合场景 | +| ------------------ | --------------------------------------------------------- | --------------------------------------------------------- | +| 内置与直接工具 | `session.tool(...)`、文件、命令行、Git、`generate_object` | 宿主明确知道要运行什么,或需要确定性调用。 | +| MCP | MCP 服务器 | 连接 GitHub、Linear、内部系统、远程宿主工具或跨进程能力。 | +| 智能体目录脚本工具 | `tools/*.md` + QuickJS `program` | 把一段受限脚本包装成模型可见工具。 | + +工具不是“文件存在就无限可用”。可见性、权限门、人工确认、允许列表和沙箱都由驾驭层 +或智能体目录加载器统一控制。高权限工具应只给可信目录,密钥应通过环境变量和宿主连接注入。 + +## 子智能体与团队 + +子智能体在 A3S Code 中拆成三种入口: + +| 入口 | 谁决定分工 | 适合场景 | +| ------------------------------------------------------------------------------------ | ---------- | -------------------------------------------- | +| `task` | 父智能体 | 模型决定委派单项,或并发扇出多个独立任务。 | +| `session.task(...)` / `session.tasks(...)` | 宿主代码 | 宿主已经知道执行通道,但仍想复用智能体能力。 | +| `session.parallel(...)` / `session.pipeline(...)` / `session.parallelResumable(...)` | 宿主代码 | 需要可复现、可测试、可恢复的固定工作流。 | + +自动委派依赖智能体描述和置信度评分。它适合“用户只描述目标,由运行时挑选专用智能体” +的场景;固定发布流程、批量审查、迁移任务更适合用编排显式表达。 + +## 可观测与接管 + +长期智能体必须能被观察和中止。A3S Code 的核心观察面包括: + +- `stream()` 输出增量事件。 +- `runs()`、`runSnapshot()`、`runEvents()` 查看当前和历史运行。 +- `toolNames()` / `toolDefinitions()` 查看可见工具表面。 +- `activeTools()` 查看当前正在运行的工具调用快照。 +- `cancelRun(runId)` 中止正在执行的回合。 +- `traceEvents()` 读取压缩、委派、工具和验证证据。 +- `verificationSummaryText` 给发布或审核流程使用。 + +宿主平台可以把这些事件转成 WebSocket/SSE、审计记录、调试面板和工作流节点状态。 + +## 与 A3S Box 的关系 + +A3S Code 负责智能体循环、工具、委派、状态和验证。A3S Box 负责更强的运行隔离:MicroVM、OCI workload、网络和 TEE。需要“模型能跑 shell,但进程必须隔离”的场景,应把 A3S Code 的工具执行放进 A3S Box 或由宿主通过 MCP 暴露隔离后的能力。 + +典型组合: + +```text +A3S Code session + -> permission policy and HITL + -> MCP tool adapter + -> A3S Box isolated workload + -> typed result and verification evidence +``` + +## 什么时候用 + +适合: + +- 发布巡检、依赖升级、代码库维护等长期工程 Agent。 +- 能拆成 explore / review / verify / implement 多角色协作的任务。 +- 需要自动委派,但仍要保留权限门、审计和验证证据的工作。 +- 需要 cron 调度和可恢复上下文的周期性报告。 +- 需要接入托管工作流、宿主 session 或外部协作渠道的 Agent。 + +不适合: + +- 只需要一次确定性 API 调用的工具型资产,直接用 tool contract 更简单。 +- 需要完整 GUI 自动化但没有结构化工具接口的任务,应优先提供 MCP 工具或浏览器工具,再交给 A3S Code 调度。 +- 需要强 OS 级隔离却只启用本地 shell 的任务,应接入 A3S Box。 + +## 阅读顺序 + +1. [文件系统优先](/guide/filesystem-first) +2. [Agent 目录](/guide/agent-dir) +3. [agents/ 角色目录](/guide/filesystem-agents) +4. [tools/ 工具目录](/guide/filesystem-tools) +5. [schedules/ 调度目录](/guide/filesystem-schedules) +6. [任务](/guide/tasks) +7. [团队](/guide/teams) diff --git a/website/docs/v8.5.1/zh/guide/examples/_meta.json b/website/docs/v8.5.1/zh/guide/examples/_meta.json new file mode 100644 index 00000000..6814f12c --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/_meta.json @@ -0,0 +1,22 @@ +[ + "index", + "quick-start", + "streaming", + "direct-tools", + "structured-output", + "batch", + "planning", + "orchestration", + "external-tasks", + "lane-queue", + "memory", + "auto-compact", + "prompt-slots", + "ripgrep-context", + "model-switching", + "skills", + "skill-tool", + "hooks", + "security", + "git-worktree" +] diff --git a/website/docs/v8.5.1/zh/guide/examples/auto-compact.mdx b/website/docs/v8.5.1/zh/guide/examples/auto-compact.mdx new file mode 100644 index 00000000..1047e4cf --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/auto-compact.mdx @@ -0,0 +1,168 @@ +--- +title: '自动压缩' +description: '让运行时自动将长会话保持在上下文预算内' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 自动压缩 + +A3S Code 可以替你将长对话保持在模型的上下文预算内。启用 `autoCompact` 后, +runtime 会持续监测上下文用量;一旦超过 `autoCompactThreshold`,较早的轮次就会被 +压缩成一份持续更新的摘要,让 agent 在大量步骤中始终保持连贯,而无需你手动管理 +token。续写则处理另一个方向:当单条回复因长度被截断时,runtime 会自动继续生成, +拼出完整回复。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let options = SessionOptions::new() + .with_auto_compact(true) + .with_auto_compact_threshold(0.75) + .with_continuation(true) + .with_max_continuation_turns(3); + let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; + + for step in 0..50 { + session + .send(&format!("第 {step} 步:继续重构解析器"), None) + .await?; + } + + println!("历史消息数:{}", session.history().len()); + if let Some(memory) = session.memory() { + println!("最近记忆:{:#?}", memory.get_recent(5).await?); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + // 上下文用量超过阈值后压缩较早的回合。 + autoCompact: true, + autoCompactThreshold: 0.75, + // 模型因长度截断单次响应时自动续写。 + continuationEnabled: true, + maxContinuationTurns: 3, +}); + +// 运行较长的多步骤任务。运行时会按需压缩较早的回合, +// 应用不需要自行计算令牌。 +for (let i = 0; i < 50; i++) { + await session.send(`Step ${i}: continue refactoring the parser`); +} + +// 检查会话当前携带的内容。 +console.log('history turns:', session.history().length); +console.log('recent memory:', await session.memoryRecent(5)); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +opts = SessionOptions() +# 上下文用量超过阈值后压缩较早的回合。 +opts.auto_compact = True +opts.auto_compact_threshold = 0.75 +# 模型因长度截断单次响应时自动续写。 +opts.continuation_enabled = True +opts.max_continuation_turns = 3 +session = agent.session("/repo", opts) + +# 运行较长的多步骤任务。运行时会按需压缩较早的回合, +# 应用不需要自行计算令牌。 +for i in range(50): + session.send(f"Step {i}: continue refactoring the parser") + +# 检查会话当前携带的内容。 +print("history turns:", len(session.history())) +print("recent memory:", session.memory_recent(5)) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + options := &code.SessionOptions{ + AutoCompact: code.Ptr(true), + AutoCompactThreshold: code.Ptr(float32(0.75)), + ContinuationEnabled: code.Ptr(true), + MaxContinuationTurns: code.Ptr(uint32(3)), + } + session, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + for step := 0; step < 50; step++ { + prompt := fmt.Sprintf("第 %d 步:继续重构解析器", step) + if _, err := session.Run(ctx, prompt); err != nil { + log.Fatal(err) + } + } + + history, err := session.History(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("历史消息数:", len(history)) +} +``` + + + + +长 session 若希望由 runtime 替你管理 context pressure,就使用自动压缩。 +`autoCompactThreshold` / `auto_compact_threshold` 是触发压缩的上下文窗口占比 +(0.0–1.0,默认 0.8);调低它可以更早触发压缩。四种 SDK 都能通过各自的 +history API 检查当前会话。四种 SDK 也都提供直接读取最近记忆的接口: +`memoryRecent` / `memory_recent` / `MemoryRecent`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/batch.mdx b/website/docs/v8.5.1/zh/guide/examples/batch.mdx new file mode 100644 index 00000000..16d43b5f --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/batch.mdx @@ -0,0 +1,189 @@ +--- +title: '批处理' +description: '通过组合 SDK 的确定性辅助方法来批量执行确定性操作' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 批处理 + +SDK 中并没有 `batch()` 方法。当宿主工作流或某个 agent 回合已经有一组清晰、相互独立、 +确定性的步骤时,你可以直接用会话提供的确定性辅助方法(`readFile`、`grep`、`glob`、`ls`、 +`git`)把它们组合起来,并自行汇总结果。这样"批处理"就完全在你的掌控之中:无需任何模型调用, +执行顺序明确,且除非你主动调用,否则不会执行任何破坏性操作。 + +下面的示例读取包元数据、changelog 和 release script,然后在不编辑任何文件的前提下报告版本不一致之处。 + + + + +```rust +use a3s_code_core::Agent; +use serde_json::Value; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/path/to/project") + .build() + .await?; + + let (package, changelog, release_script) = tokio::try_join!( + session.read_file("package.json"), + session.read_file("CHANGELOG.md"), + session.read_file("scripts/release.sh"), + )?; + + let package: Value = serde_json::from_str(&package)?; + let version = package["version"].as_str().unwrap_or_default(); + let mut mismatches = Vec::new(); + if !changelog.contains(version) { + mismatches.push(format!("CHANGELOG.md 缺少版本 {version}")); + } + if !release_script.contains(version) { + mismatches.push(format!("release.sh 缺少版本 {version}")); + } + + if mismatches.is_empty() { + println!("所有文件中的版本一致。"); + } else { + println!("{}", mismatches.join("\n")); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/path/to/project'); + +// Node.js 辅助方法是异步的,可用 Promise.all 组合互不依赖的读取。 +const [pkg, changelog, releaseScript] = await Promise.all([ + session.readFile('package.json'), + session.readFile('CHANGELOG.md'), + session.readFile('scripts/release.sh'), +]); + +const pkgVersion = JSON.parse(pkg).version; +const mismatches = []; +if (!changelog.includes(pkgVersion)) + mismatches.push(`CHANGELOG.md is missing ${pkgVersion}`); +if (!releaseScript.includes(pkgVersion)) + mismatches.push(`release.sh is missing ${pkgVersion}`); + +console.log( + mismatches.length ? mismatches.join('\n') : 'All files agree on the version.', +); + +session.close(); +``` + + + + +```python +import json +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session("/path/to/project", SessionOptions()) + +# Python 辅助方法是同步的,按顺序调用即可,无需 await。 +pkg = session.read_file("package.json") +changelog = session.read_file("CHANGELOG.md") +release_script = session.read_file("scripts/release.sh") + +pkg_version = json.loads(pkg)["version"] +mismatches = [] +if pkg_version not in changelog: + mismatches.append(f"CHANGELOG.md is missing {pkg_version}") +if pkg_version not in release_script: + mismatches.append(f"release.sh is missing {pkg_version}") + +print("\n".join(mismatches) if mismatches else "All files agree on the version.") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/project", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + pkg, err := session.ReadFile(ctx, "package.json", nil) + if err != nil { + log.Fatal(err) + } + changelog, err := session.ReadFile(ctx, "CHANGELOG.md", nil) + if err != nil { + log.Fatal(err) + } + releaseScript, err := session.ReadFile(ctx, "scripts/release.sh", nil) + if err != nil { + log.Fatal(err) + } + + var metadata struct { + Version string `json:"version"` + } + if err := json.Unmarshal([]byte(pkg), &metadata); err != nil { + log.Fatal(err) + } + var mismatches []string + if !strings.Contains(changelog, metadata.Version) { + mismatches = append(mismatches, "CHANGELOG.md 缺少版本 "+metadata.Version) + } + if !strings.Contains(releaseScript, metadata.Version) { + mismatches = append(mismatches, "release.sh 缺少版本 "+metadata.Version) + } + if len(mismatches) == 0 { + fmt.Println("所有文件中的版本一致。") + } else { + fmt.Println(strings.Join(mismatches, "\n")) + } +} +``` + + + + +不要把破坏性操作混入这些分组读取之中。任何破坏性宿主工作流(写入、`git` 提交、`bash`) +都应放在显式的应用确认或你的自动化关卡之后,并通过 `permissionPolicy` / +`permission_policy` 进行约束,避免出现意料之外的步骤被静默执行。 + +如果这些步骤并非相互独立——每一步都依赖上一步的结果,并且你希望由 agent 来驱动它们—— +请改用 [`session.pipeline(...)`](/guide/examples/orchestration),它会分阶段运行任务, +每个阶段都能接收到上一阶段的输出。 diff --git a/website/docs/v8.5.1/zh/guide/examples/direct-tools.mdx b/website/docs/v8.5.1/zh/guide/examples/direct-tools.mdx new file mode 100644 index 00000000..2d0a628a --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/direct-tools.mdx @@ -0,0 +1,296 @@ +--- +title: '直接工具' +description: '不消耗大语言模型回合即可运行确定性宿主工具' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 直接工具 + +`session.tool(name, args)`(以及 `glob`、`grep`、`readFile` 等类型化辅助方法)会直接运行宿主工具,循环中不发起任何模型调用。它们适用于测试、迁移以及宿主驱动的工作流——你需要确定性结果,而不是一次智能体回合。 + +直接调用是宿主控制面调用。调用前请先执行你的产品授权逻辑;`permissionPolicy` +管控的是 agent turn 内由模型选择的工具调用,不是你的应用代码是否可以调用 SDK helper。 + + + + +```rust +use a3s_code_core::Agent; +use serde_json::{json, Value}; +use std::collections::HashSet; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let allowed = HashSet::from(["search", "read", "generate_object"]); + let check = |name: &str| -> a3s_code_core::Result<()> { + allowed + .contains(name) + .then_some(()) + .ok_or_else(|| { + a3s_code_core::CodeError::Security(format!( + "此处不允许直接调用工具:{name}" + )) + }) + }; + + check("search")?; + let files = session.glob("**/*.rs").await?; + println!("glob 找到 {} 个 Rust 文件", files.len()); + + let matches = session.grep("Agent::new").await?; + println!("grep 找到 {} 行匹配", matches.lines().count()); + + check("read")?; + let readme = session.read_file("README.md").await?; + println!("README 大小为 {} 字节", readme.len()); + + let raw = session + .tool("read", json!({ "file_path": "Cargo.toml" })) + .await?; + println!("通过 tool() 读取了 {} 字节", raw.output.len()); + println!("会话暴露了 {} 个工具", session.tool_definitions().len()); + + check("generate_object")?; + let structured = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["count", "language"], + "properties": { + "count": { "type": "integer" }, + "language": { "type": "string" } + } + }, + "prompt": "这个项目中有多少个 Rust 文件?", + "schema_name": "file_stats" + }), + ) + .await?; + if structured.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "generate_object".into(), + message: structured.output, + }); + } + let generated: Value = serde_json::from_str(&structured.output)?; + println!("结构化输出:{}", generated["object"]); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const allowedDirectTools = new Set(['glob', 'grep', 'read', 'generate_object']); +function assertDirectToolAllowed(name: string) { + if (!allowedDirectTools.has(name)) { + throw new Error(`direct tool not allowed here: ${name}`); + } +} + +// Glob:按模式列出文件 +assertDirectToolAllowed('glob'); +const files = await session.glob('**/*.ts'); +console.log(`glob found ${files.length} TypeScript files`); + +// Grep:搜索文件内容 +assertDirectToolAllowed('grep'); +const matches = await session.grep('Agent.create'); +const matchCount = matches.split('\n').filter(Boolean).length; +console.log(`grep found ${matchCount} matching lines`); + +// 读取文件 +assertDirectToolAllowed('read'); +const readme = await session.readFile('README.md'); +console.log(`README is ${readme.length} bytes`); + +// 按名称直接调用工具 +assertDirectToolAllowed('read'); +const raw = await session.tool('read', { file_path: 'package.json' }); +console.log(`package.json via tool(): ${raw.output.length} bytes`); + +// 查看可用的工具模式 +const schemas = session.toolDefinitions(); +console.log(`session exposes ${schemas.length} tools`); + +// 结构化输出:生成通过模式校验的 JSON 对象 +assertDirectToolAllowed('generate_object'); +const structured = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['count', 'language'], + properties: { + count: { type: 'integer' }, + language: { type: 'string' }, + }, + }, + prompt: 'How many TypeScript files are in this project?', + schema_name: 'file_stats', +}); +if (structured.exitCode !== 0) { + throw new Error(structured.output); +} +console.log('structured output:', JSON.parse(structured.output).object); + +session.close(); +``` + + + + +```python +import json + +from a3s_code import Agent + +agent = Agent.create("agent.acl") +session = agent.session('.') + +ALLOWED_DIRECT_TOOLS = {'glob', 'grep', 'read', 'generate_object'} + + +def assert_direct_tool_allowed(name: str) -> None: + if name not in ALLOWED_DIRECT_TOOLS: + raise RuntimeError(f'direct tool not allowed here: {name}') + +# Glob:按模式列出文件 +assert_direct_tool_allowed('glob') +files = session.glob('**/*.py') +print(f'glob found {len(files)} Python files') + +# Grep:搜索文件内容 +assert_direct_tool_allowed('grep') +matches = session.grep('Agent.create') +match_count = len([line for line in matches.splitlines() if line]) +print(f'grep found {match_count} matching lines') + +# 读取文件 +assert_direct_tool_allowed('read') +readme = session.read_file('README.md') +print(f'README is {len(readme)} bytes') + +# 按名称直接调用工具 +assert_direct_tool_allowed('read') +raw = session.tool('read', {'file_path': 'pyproject.toml'}) +print(f'pyproject.toml via tool(): {len(raw.output)} bytes') + +# 查看可用的工具模式 +schemas = session.tool_definitions() +print(f'session exposes {len(schemas)} tools') + +# 结构化输出:生成通过模式校验的 JSON 对象 +assert_direct_tool_allowed('generate_object') +structured = session.tool('generate_object', { + 'schema': { + 'type': 'object', + 'required': ['count', 'language'], + 'properties': { + 'count': {'type': 'integer'}, + 'language': {'type': 'string'}, + }, + }, + 'prompt': 'How many Python files are in this project?', + 'schema_name': 'file_stats', +}) +if structured.exit_code != 0: + raise RuntimeError(structured.output) +print('structured output:', json.loads(structured.output)['object']) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func must[T any](value T, err error) T { + if err != nil { + log.Fatal(err) + } + return value +} + +func main() { + ctx := context.Background() + agent := must(code.Create(ctx, "agent.acl")) + defer agent.Close(context.Background()) + session := must(agent.Session(ctx, ".", nil)) + defer session.Close(context.Background()) + + files := must(session.Glob(ctx, "**/*.go")) + fmt.Printf("glob 找到 %d 个 Go 文件\n", len(files)) + + matches := must(session.Grep(ctx, "code.Create")) + fmt.Printf("grep 找到 %d 行结果\n", len(strings.Split(matches, "\n"))) + + readme := must(session.ReadFile(ctx, "README.md", nil)) + fmt.Printf("README 大小为 %d 字节\n", len(readme)) + + raw := must(session.Tool(ctx, "read", map[string]any{ + "file_path": "go.mod", + })) + fmt.Printf("通过 Tool() 读取 go.mod:%d 字节\n", len(raw.Output)) + + schemas := must(session.ToolDefinitions(ctx)) + fmt.Printf("Session 提供 %d 个工具\n", len(schemas)) + + structured := must(session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"count", "language"}, + "properties": map[string]any{ + "count": map[string]any{"type": "integer"}, + "language": map[string]any{"type": "string"}, + }, + }, + "prompt": "这个项目中有多少个 Go 文件?", + "schema_name": "file_stats", + })) + if structured.ExitCode != 0 { + log.Fatal(structured.Output) + } + var generated struct { + Object map[string]any `json:"object"` + } + if err := json.Unmarshal([]byte(structured.Output), &generated); err != nil { + log.Fatal(err) + } + fmt.Println("结构化输出:", generated.Object) +} +``` + + + + +直接工具在 session 工作区下执行,应视为宿主侧的特权操作。大多数调用(`read`、`glob`、`grep`)是纯确定性的;`generate_object` 是个例外——它仍会调用模型来填充经过 schema 校验的 JSON 对象,但由你显式驱动,而非通过自由形式的智能体回合。 + +可运行版本位于 `sdk/node/examples/basic/test_generate_object.ts` +(Python:`sdk/python/examples/test_generate_object.py`)。Go 的直接 +工具契约由 `sdk/go/session_test.go` 覆盖。 diff --git a/website/docs/v8.5.1/zh/guide/examples/external-tasks.mdx b/website/docs/v8.5.1/zh/guide/examples/external-tasks.mdx new file mode 100644 index 00000000..2c0ba829 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/external-tasks.mdx @@ -0,0 +1,247 @@ +--- +title: '外部任务' +description: '在智能体进程之外完成排队工作,并把结构化证据报告回去' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 外部任务 + +有些工作无法在 agent 进程内运行:它属于独立的 worker、CI runner,或位于另一个系统中的 +人工处理者。当某个 lane 被路由到外部处理器时,该 lane 上的工具会被**排队**而不是被执行—— +它们会作为外部任务等待。你的宿主代码会取出待处理队列、以任意方式完成工作,并通过 +`completeExternalTask` 把结果报告回去。仅当外部 worker 确实是你架构的一部分时,才使用此模式。 + +外部任务由 [lane 队列](/guide/examples/lane-queue) 产生:你必须先注册至少一个 +`external`(或 `hybrid`)lane 处理器,否则每个任务都会在进程内运行,也就没有什么可取出的。 + + + + +```rust +use a3s_code_core::{ + queue::{ + ExternalTaskResult, LaneHandlerConfig, SessionLane, SessionQueueConfig, + TaskHandlerMode, + }, + Agent, SessionOptions, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_queue_config(SessionQueueConfig::default()), + ) + .build() + .await?; + session + .set_lane_handler( + SessionLane::Execute, + LaneHandlerConfig { + mode: TaskHandlerMode::External, + timeout_ms: 300_000, + }, + ) + .await?; + + for task in session.pending_external_tasks().await { + println!( + "待处理:{},lane={:?},类型={}", + task.task_id, task.lane, task.command_type + ); + let completed = session + .complete_external_task( + &task.task_id, + ExternalTaskResult { + success: true, + result: json!({ + "summary": "worker 已完成测试", + "command": "npm run build", + "exit_code": 0 + }), + error: None, + }, + ) + .await; + println!("已完成:{completed}"); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { queueConfig: {} }); + +// 把执行通道路由到外部工作进程,使工具进入队列而不在本地执行。 +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); + +// 排空等待宿主处理的任务。 +const pending = await session.pendingExternalTasks(); + +for (const task of pending) { + console.log( + `pending: ${task.task_id} on ${task.lane} (${task.command_type})`, + ); + + try { + // ...the host does the real work here (run CI, call a service, ask a human)... + const ok = await session.completeExternalTask(task.task_id, { + success: true, + result: { + summary: 'worker completed the test run', + command: 'npm run build', + exitCode: 0, + }, + }); + console.log('completed:', ok); + } catch (err) { + await session.completeExternalTask(task.task_id, { + success: false, + error: String(err), + }); + } +} + +session.close(); +``` + + + + +```python +import os +from a3s_code import Agent, SessionOptions, SessionQueueConfig + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.queue_config = SessionQueueConfig() +session = agent.session(os.getcwd(), opts) + +# 把执行通道路由到外部工作进程,使工具进入队列而不在本地执行。 +session.set_lane_handler("execute", "external", 300000) + +# 排空等待宿主处理的任务。 +pending = session.pending_external_tasks() + +for task in pending: + print(f"pending: {task['task_id']} on {task['lane']} ({task['command_type']})") + + try: + # ...the host does the real work here (run CI, call a service, ask a human)... + ok = session.complete_external_task( + task["task_id"], + success=True, + result={ + "summary": "worker completed the test run", + "command": "npm run build", + "exit_code": 0, + }, + ) + print("completed:", ok) + except Exception as err: + session.complete_external_task( + task["task_id"], + success=False, + error=str(err), + ) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + QueueConfig: &code.SessionQueueConfig{}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + err = session.SetLaneHandler(ctx, code.LaneExecute, code.LaneHandlerConfig{ + Mode: "external", + TimeoutMS: 300_000, + }) + if err != nil { + log.Fatal(err) + } + + pending, err := session.PendingExternalTasks(ctx) + if err != nil { + log.Fatal(err) + } + for _, task := range pending { + fmt.Println("pending:", task.TaskID, task.Lane, task.CommandType) + completed, err := session.CompleteExternalTask( + ctx, + task.TaskID, + code.ExternalTaskResult{ + Success: true, + Result: json.RawMessage( + `{"summary":"worker completed the test run","exit_code":0}`, + ), + }, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println("completed:", completed) + } +} +``` + + + + +说明: + +- 每个待处理任务都携带 `task_id`、`session_id`、`lane`、`command_type`、`payload` 和 + `timeout_ms`。完成时请把 `task_id` 传回 `completeExternalTask` / `complete_external_task`, + 以便将本次完成匹配到正确的任务。 +- 结果的结构为 `{ success, result?, error? }`。其中 `result` 可以是任意可 JSON 序列化的负载; + `error` 是失败时的可选消息。 +- 成功时,返回紧凑的结构化证据(一段摘要加上关键事实),而不是只有原始日志——agent 会基于 + 该结果进行推理,所以请保持其精简且机器可读。 +- 当任务被找到并完成时,`completeExternalTask` / `complete_external_task` 返回 `true`,否则 + 返回 `false`。在 Python 中这些队列方法是同步的;在 Node 中 `pendingExternalTasks` 和 + `completeExternalTask` 返回 promise。 +- Go 使用 `PendingExternalTasks` 与 `CompleteExternalTask`,两者都接收调用方的 + `context.Context`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/git-worktree.mdx b/website/docs/v8.5.1/zh/guide/examples/git-worktree.mdx new file mode 100644 index 00000000..886f58c3 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/git-worktree.mdx @@ -0,0 +1,231 @@ +--- +title: 'Git 工作树' +description: '通过会话的 Git 工具操作仓库与工作树' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Git 工作树 + +session 的 `git` 工具以宿主侧特权操作的方式运行 git。它接收一个结构化的命令对象 +(`command`,worktree 操作还可带 `subcommand`/`name`/`path`),并返回包含 +`output` 与 `exitCode` 的工具结果。本示例先检查仓库,然后直接通过工具表面创建、 +列出并移除一个 worktree。 + + + + +```rust +use a3s_code_core::Agent; +use serde_json::json; +use std::path::Path; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder("/path/to/repo").build().await?; + + let status = session.tool("git", json!({ "command": "status" })).await?; + println!("{}", status.output); + + let worktree = Path::new("/path/to/repo").join("wt-feature-auth"); + let created = session + .tool( + "git", + json!({ + "command": "worktree", + "subcommand": "create", + "name": "feature-auth", + "path": worktree + }), + ) + .await?; + if created.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "git".into(), + message: format!("创建失败:{}", created.output), + }); + } + + let list = session + .tool( + "git", + json!({ "command": "worktree", "subcommand": "list" }), + ) + .await?; + println!("{}", list.output); + + let removed = session + .tool( + "git", + json!({ + "command": "worktree", + "subcommand": "remove", + "path": worktree + }), + ) + .await?; + if removed.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "git".into(), + message: format!("删除失败:{}", removed.output), + }); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; +import * as path from 'path'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/path/to/repo'); + +// 检查仓库 +const status = await session.git({ command: 'status' }); +console.log(status.output); + +// 在新分支上创建工作树 +const wtPath = path.join('/path/to/repo', 'wt-feature-auth'); +const created = await session.git({ + command: 'worktree', + subcommand: 'create', + name: 'feature-auth', + path: wtPath, +}); +if (created.exitCode !== 0) throw new Error(`create failed: ${created.output}`); + +// 列出工作树 +const list = await session.git({ command: 'worktree', subcommand: 'list' }); +console.log(list.output); + +// 完成后删除工作树 +const removed = await session.git({ + command: 'worktree', + subcommand: 'remove', + path: wtPath, +}); +if (removed.exitCode !== 0) throw new Error(`remove failed: ${removed.output}`); + +session.close(); +``` + + + + +```python +import os +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session("/path/to/repo", SessionOptions()) + +# 检查仓库 +status = session.git({"command": "status"}) +print(status.output) + +# 在新分支上创建工作树 +wt_path = os.path.join("/path/to/repo", "wt-feature-auth") +created = session.git({ + "command": "worktree", + "subcommand": "create", + "name": "feature-auth", + "path": wt_path, +}) +if created.exit_code != 0: + raise RuntimeError(f"create failed: {created.output}") + +# 列出工作树 +listing = session.git({"command": "worktree", "subcommand": "list"}) +print(listing.output) + +# 完成后删除工作树 +removed = session.git({ + "command": "worktree", + "subcommand": "remove", + "path": wt_path, +}) +if removed.exit_code != 0: + raise RuntimeError(f"remove failed: {removed.output}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "path/filepath" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/repo", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + status, err := session.Git(ctx, code.GitOptions{Command: "status"}) + if err != nil { + log.Fatal(err) + } + fmt.Println(status.Output) + + worktree := filepath.Join("/path/to/repo", "wt-feature-auth") + created, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "create", + Name: "feature-auth", Path: worktree, + }) + if err != nil || created.ExitCode != 0 { + log.Fatalf("创建失败:%v %s", err, created.Output) + } + list, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "list", + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(list.Output) + + removed, err := session.Git(ctx, code.GitOptions{ + Command: "worktree", Subcommand: "remove", Path: worktree, + }) + if err != nil || removed.ExitCode != 0 { + log.Fatalf("删除失败:%v %s", err, removed.Output) + } +} +``` + + + + +请传入命令对象,而不是位置参数:`{ command: 'status' }`、`{ command: 'diff' }` +或 `{ command: 'worktree', subcommand: 'list' }`。每次调用都会返回一个工具结果, +因此在使用 output 之前,应先检查 `exit_code`(Rust/Python)、`exitCode` +(Node.js)或 `ExitCode`(Go)。 + +直接 git 调用是宿主侧特权操作。push、publish 和 release workflow 应放在应用级确认 +或自动化关卡后面。 + +可运行版本位于 `sdk/node/examples/git/test_worktree_git.ts`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/hooks.mdx b/website/docs/v8.5.1/zh/guide/examples/hooks.mdx new file mode 100644 index 00000000..d4ba81da --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/hooks.mdx @@ -0,0 +1,199 @@ +--- +title: '生命周期钩子' +description: '注册、统计并注销用于观测和把控 Agent 活动的生命周期事件回调。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 生命周期钩子 + +钩子让你在 Agent 活动发生时进行观测和把控。你针对某个生命周期事件注册一个具名回调, +运行时会在该节点调用它,回调返回一个决策,例如 `{ action: "continue" }`。钩子可用于 +审计、脱敏、日志记录,或在不修改 Agent 提示词的前提下实施策略。 + +其生命周期是对称的:`registerHook` 按名称添加回调,`hookCount` 告诉你当前有多少个钩子 +处于活动状态,`unregisterHook` 则按名称移除某个钩子。 + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{ + hooks::{ + Hook, HookConfig, HookEvent, HookEventType, HookHandler, HookMatcher, HookResponse, + }, + Agent, +}; + +struct ContinueHandler; + +impl HookHandler for ContinueHandler { + fn handle(&self, event: &HookEvent) -> HookResponse { + println!("observed {}", event.event_type()); + HookResponse::continue_() + } +} + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let hook = Hook::new("observe-env-read", HookEventType::PreToolUse) + .with_matcher(HookMatcher::path("**/.env*")) + .with_config(HookConfig { + priority: 100, + ..HookConfig::default() + }); + session.register_hook(hook)?; + session.register_hook_handler("observe-env-read", Arc::new(ContinueHandler))?; + + println!("active hooks: {}", session.hook_count()); + session.send("读取项目 README 并总结。", None).await?; + + session.unregister_hook_handler("observe-env-read")?; + session.unregister_hook("observe-env-read")?; + println!("active hooks: {}", session.hook_count()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd()); + +// 为生命周期事件注册命名钩子。回调不能抛出异常, +// 必须返回形如 { action: 'continue' } 的决策。 +session.registerHook( + 'observe-env-read', + 'pre_tool_use', + { pathPattern: '**/.env*' }, + { priority: 100 }, + () => ({ action: 'continue' }), +); + +console.log('active hooks:', session.hookCount()); // 1 + +await session.run('Read the project README and summarize it.'); + +// 不再需要时按名称移除钩子。 +session.unregisterHook('observe-env-read'); +console.log('active hooks:', session.hookCount()); // 0 + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session('.', SessionOptions()) + +# 为生命周期事件注册命名钩子,回调会返回决策。 +session.register_hook( + 'observe-env-read', + 'pre_tool_use', + {'pathPattern': '**/.env*'}, + {'priority': 100}, + lambda: {'action': 'continue'}, +) + +print("active hooks:", session.hook_count()) # 1 + +session.run("Read the project README and summarize it.") + +# 不再需要时按名称移除钩子。 +session.unregister_hook('observe-env-read') +print("active hooks:", session.hook_count()) # 0 + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + priority := 100 + err = session.RegisterHookWithHandler(ctx, code.Hook{ + ID: "observe-env-read", + EventType: "pre_tool_use", + Matcher: &code.HookMatcher{PathPattern: "**/.env*"}, + Config: &code.HookConfig{Priority: &priority}, + }, func(_ context.Context, event json.RawMessage) (*code.HookResponse, error) { + fmt.Println("hook event:", string(event)) + return &code.HookResponse{Action: "continue"}, nil + }) + if err != nil { + log.Fatal(err) + } + count, err := session.HookCount(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("active hooks:", count) + + if _, err = session.Run(ctx, "读取项目 README 并总结。"); err != nil { + log.Fatal(err) + } + + if _, err = session.UnregisterHook(ctx, "observe-env-read"); err != nil { + log.Fatal(err) + } +} +``` + + + + +注意事项: + +- 钩子回调返回一个决策。返回 `{ action: "continue" }`(Node)/ + `{"action": "continue"}`(Python)/ + `&code.HookResponse{Action: "continue"}`(Go)即可让 Agent 继续执行。 +- 匹配器(`{ pathPattern: '**/.env*' }`)将钩子限定到路径匹配该模式的事件,而 + `{ priority: 100 }` 用于对同一事件上的多个钩子排序(数值越小越先执行)。 +- Node 钩子回调**不得**抛出异常——未捕获的抛出可能会终止进程。请让处理逻辑保持完备, + 并始终返回一个决策。 +- `hookCount` / `hook_count` / `HookCount` 反映当前已注册钩子的数量,在测试中可方便地断言注册与 + 清理是否生效。 +- `unregisterHook` / `unregister_hook` / `UnregisterHook` 接收你注册时使用的名称。请始终拆除不再需要的 + 钩子,以免它们在多次运行之间泄漏。 +- 把钩子当作生产关卡前,请先验证你所依赖的具体 event path。 diff --git a/website/docs/v8.5.1/zh/guide/examples/index.mdx b/website/docs/v8.5.1/zh/guide/examples/index.mdx new file mode 100644 index 00000000..0ff3c24e --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/index.mdx @@ -0,0 +1,22 @@ +--- +title: '示例' +description: '当前 A3S Code 运行时和 SDK 接口示例' +--- + +# 示例 + +这些示例使用当前 A3S Code runtime 的概念:ACL 配置、环境变量注入、session API、 +流式输出、结构化输出、基于 task 的委派与自动 subagent 委派、可编程编排 +(`parallel` / `pipeline` / `parallelResumable`)、`.a3s/agents`、技能、记忆、 +直接工具、workspace backend、验证、Git 工作流以及可选的 MCP/队列基础设施。 + +建议从[快速开始](/guide/examples/quick-start)开始,然后按主题阅读: + +- **Session 与运行时** — [快速开始](/guide/examples/quick-start)、[流式输出](/guide/examples/streaming)、[模型切换](/guide/examples/model-switching)、[自动压缩](/guide/examples/auto-compact) +- **结构化与可编程** — [结构化输出](/guide/examples/structured-output)、[编排](/guide/examples/orchestration)、[规划](/guide/examples/planning)、[批处理](/guide/examples/batch) +- **工具与上下文** — [直接工具](/guide/examples/direct-tools)、[ripgrep 上下文](/guide/examples/ripgrep-context)、[Prompt 插槽](/guide/examples/prompt-slots)、[Git worktree](/guide/examples/git-worktree) +- **技能与记忆** — [技能](/guide/examples/skills)、[技能工具](/guide/examples/skill-tool)、[记忆](/guide/examples/memory)、[Hooks](/guide/examples/hooks) +- **安全与验证** — [安全](/guide/examples/security) +- **MCP 与队列** — [Lane 队列](/guide/examples/lane-queue)、[外部任务](/guide/examples/external-tasks) + +参见[编排](/guide/examples/orchestration)示例,了解扇出(`parallel`)、分阶段(`pipeline`)和可恢复(`parallelResumable`)的多 agent 工作流。 diff --git a/website/docs/v8.5.1/zh/guide/examples/lane-queue.mdx b/website/docs/v8.5.1/zh/guide/examples/lane-queue.mdx new file mode 100644 index 00000000..81eb2ad2 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/lane-queue.mdx @@ -0,0 +1,215 @@ +--- +title: '执行通道队列' +description: '将某条执行通道路由到外部工作进程,并显式排空其待处理任务。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 执行通道队列 + +默认情况下,A3S Code 会在进程内运行每一个任务,没有任何队列。lane 队列属于**可选基础设施**:为某条 lane 注册一个外部 handler 后,路由到该 lane 的工具不会由 agent 执行,而是排入队列、等待外部 worker 取走。随后你自己排空这些待处理任务,按需运行它们,再把结果回报回来。只有当外部 worker 确实是你架构的一部分时,才需要用到它。 + +共有四条 lane:`control`、`query`、`execute` 和 `generate`。每个 handler 的 `mode` 可以是 `internal`(默认)、`external` 或 `hybrid`。 + + + + +```rust +use a3s_code_core::{ + queue::{ + ExternalTaskResult, LaneHandlerConfig, SessionLane, SessionQueueConfig, + TaskHandlerMode, + }, + Agent, SessionOptions, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_queue_config(SessionQueueConfig::default()), + ) + .build() + .await?; + + session + .set_lane_handler( + SessionLane::Execute, + LaneHandlerConfig { + mode: TaskHandlerMode::External, + timeout_ms: 300_000, + }, + ) + .await?; + println!("队列已启用:{}", session.has_queue()); + + for task in session.pending_external_tasks().await { + println!( + "待处理:{} {:?} {}", + task.task_id, task.lane, task.command_type + ); + session + .complete_external_task( + &task.task_id, + ExternalTaskResult { + success: true, + result: json!({ "note": "由外部 worker 完成" }), + error: None, + }, + ) + .await; + } + + println!("lane 队列已清空"); + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('./agent.acl'); +const session = agent.session(process.cwd(), { queueConfig: {} }); + +// 把 "execute" 执行通道路由到外部工作进程。 +// 智能体不会运行这条通道上的工具,而是把它们放入队列, +// 等待外部工作进程领取并完成。 +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); + +// 会话配置了 queueConfig,因此 hasQueue() 返回 true。 +console.log('queue active:', session.hasQueue()); + +// 排空等待外部工作进程处理的任务。 +const pending = await session.pendingExternalTasks(); +for (const task of pending) { + console.log('pending:', task.task_id, task.lane, task.command_type); + + // ... hand off to your worker, run it, then report the outcome back: + await session.completeExternalTask(task.task_id, { + success: true, + result: { note: 'done by external worker' }, + }); +} + +console.log('lane queue drained'); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, SessionQueueConfig + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.queue_config = SessionQueueConfig() +session = agent.session(".", opts) + +# 把 "execute" 执行通道路由到外部工作进程。 +# 智能体不会运行这条通道上的工具,而是把它们放入队列, +# 等待外部工作进程领取并完成。 +session.set_lane_handler("execute", "external", 300000) + +# 会话配置了 queue_config,因此 has_queue() 返回 True。 +print("queue active:", session.has_queue()) + +# 排空等待外部工作进程处理的任务。 +pending = session.pending_external_tasks() +for task in pending: + print("pending:", task["task_id"], task["lane"], task["command_type"]) + + # ... hand off to your worker, run it, then report the outcome back: + session.complete_external_task( + task["task_id"], + success=True, + result={"note": "done by external worker"}, + ) + +print("lane queue drained") +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + QueueConfig: &code.SessionQueueConfig{}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + err = session.SetLaneHandler(ctx, code.LaneExecute, code.LaneHandlerConfig{ + Mode: "external", + TimeoutMS: 300_000, + }) + if err != nil { + log.Fatal(err) + } + active, err := session.HasQueue(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("queue active:", active) + + pending, err := session.PendingExternalTasks(ctx) + if err != nil { + log.Fatal(err) + } + for _, task := range pending { + fmt.Println("pending:", task.TaskID, task.Lane, task.CommandType) + _, err = session.CompleteExternalTask(ctx, task.TaskID, code.ExternalTaskResult{ + Success: true, + Result: json.RawMessage(`{"note":"done by external worker"}`), + }) + if err != nil { + log.Fatal(err) + } + } +} +``` + + + + +说明: + +- **默认路径不含队列。** 仅当会话创建时传入 `queueConfig` / `queue_config` / + `QueueConfig`,`hasQueue()` / `has_queue()` / `HasQueue()` 才返回 `true`。未配置 + queue 时,lane handler 没有可修改的队列。 +- 每个待处理任务都带有 `task_id`、`session_id`、`lane`、`command_type`、`payload` 和 `timeout_ms`。工作完成后,把 `task_id` 传回给 `completeExternalTask` / `complete_external_task` / `CompleteExternalTask`。 +- 结果结构为 `{ success, result?, error? }`——`result` 可承载任意可 JSON 序列化的载荷,`error` 是失败时的可选消息。`completeExternalTask` / `complete_external_task` 在找到并完成任务时返回 `true`,否则返回 `false`。 +- 在 Python 中这些队列方法是同步的;在 Node 中 `setLaneHandler`、`pendingExternalTasks` 和 `completeExternalTask` 返回 promise,而 `hasQueue` 是同步的。 +- Go 队列调用接收 `context.Context`;Node 返回 promise;Python 使用同步方法。 diff --git a/website/docs/v8.5.1/zh/guide/examples/memory.mdx b/website/docs/v8.5.1/zh/guide/examples/memory.mdx new file mode 100644 index 00000000..fe371d99 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/memory.mdx @@ -0,0 +1,189 @@ +--- +title: '记忆' +description: '记录任务结果,之后按相似度、标签或时间召回。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 记忆 + +持久化记忆让会话能够记录哪些做法成功(以及哪些失败),并在之后取回这些事实。 +SDK session 默认在 `/.a3s/memory` 使用文件存储;TUI 默认使用 +`~/.a3s/memory`,让 `/memory` 面板和实时 session 浏览同一份长期记忆。只有当你想 +覆盖路径或后端时,才需要传入 `memoryStore`。你仍然可以用 `rememberSuccess` / +`rememberFailure` 显式写入结果,再用 `recallSimilar`、`recallByTags` 或 +`memoryRecent` 取回它们。 + +LLM 抽取也默认开启。它会在重要 turn 完成后整理可复用的偏好、workflow、decision +和失败教训,普通寒暄或短 turn 会被跳过。 +默认 store 会把完全重复和保守近重复 memory 合并到 canonical item;带冲突意味的 +memory 会保留为独立项,方便后续 recall。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_file_memory("./.a3s/memory")) + .build() + .await?; + let memory = session.memory().expect("已配置记忆"); + + memory + .remember_success( + "重构认证模块", + &["read".into(), "edit".into(), "bash".into()], + "提取 AuthService 后所有测试均通过", + ) + .await?; + memory + .remember_failure( + "迁移尝试", + "5432 端口上的 psql 连接被拒绝", + &["bash".into()], + ) + .await?; + + let recent = memory.get_recent(10).await?; + let by_tags = memory + .recall_by_tags(&["read".into(), "edit".into()], 5) + .await?; + let similar = memory.recall_similar("认证重构", 5).await?; + println!("{} {} {}", recent.len(), by_tags.len(), similar.len()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileMemoryStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), +}); + +// 随着智能体执行,记录工作结果。 +await session.rememberSuccess( + 'refactored auth module', + ['read', 'edit', 'bash'], + 'all tests passed after extracting AuthService', +); +await session.rememberFailure( + 'migration attempt', + ['bash'], + 'psql connection refused on port 5432', +); + +// 稍后按时间、工具标签或语义相似度召回。 +const recent = await session.memoryRecent(10); +const byTags = await session.recallByTags(['read', 'edit'], 5); +const similar = await session.recallSimilar('auth refactor', 5); + +console.log(recent.length, byTags.length, similar.length); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, FileMemoryStore + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.memory_store = FileMemoryStore("./.a3s/memory") +session = agent.session("/repo", opts) + +# 随着智能体执行记录工作结果。Python 辅助方法为同步接口,无需 await。 +session.remember_success( + "refactored auth module", + ["read", "edit", "bash"], + "all tests passed after extracting AuthService", +) +session.remember_failure( + "migration attempt", + ["bash"], + "psql connection refused on port 5432", +) + +# 稍后按时间、工具标签或语义相似度召回。 +recent = session.memory_recent(10) +by_tags = session.recall_by_tags(["read", "edit"], 5) +similar = session.recall_similar("auth refactor", 5) + +print(len(recent), len(by_tags), len(similar)) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + options := &code.SessionOptions{FileMemoryDir: "./.a3s/memory"} + + first, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + if err := first.RememberSuccess( + ctx, + "认证验证规则", + []string{"cargo", "test"}, + "认证改动必须通过 cargo test。", + ); err != nil { + log.Fatal(err) + } + if err := first.Close(ctx); err != nil { + log.Fatal(err) + } + + second, err := agent.Session(ctx, "/repo", options) + if err != nil { + log.Fatal(err) + } + defer second.Close(context.Background()) + items, err := second.RecallSimilar(ctx, "认证改动需要什么验证?", 5) + if err != nil { + log.Fatal(err) + } + for _, item := range items { + fmt.Println(item.Content) + } +} +``` + + + + +四种 SDK 都暴露显式的记忆写入与查询辅助方法。Go 对应 +`RememberSuccess`、`RememberFailure`、`RecallSimilar`、`RecallByTags`、 +`MemoryRecent` 以及 working/short-term memory 方法。remember 方法接收简短 +任务描述、涉及的工具列表(同时作为可搜索标签)以及结果文本。若默认文件 store +无法创建,会话会回退到进程内 memory,并暴露初始化告警。 diff --git a/website/docs/v8.5.1/zh/guide/examples/model-switching.mdx b/website/docs/v8.5.1/zh/guide/examples/model-switching.mdx new file mode 100644 index 00000000..4afc1cf4 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/model-switching.mdx @@ -0,0 +1,362 @@ +--- +title: '模型切换' +description: '为每个会话选择模型,并可针对每个 worker 智能体单独覆盖,以平衡成本与能力。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 模型切换 + +会话会使用你在 `model` 选项中传入的模型运行。先声明一次智能体可以访问的所有模型, +然后为每个会话选择其一——用快速模型处理高频、低风险的工作,用能力更强的模型进行 +评审。当你希望在不改动任何提示词的前提下,于成本与能力之间取得平衡时,可以使用这 +种方式。 + +## 声明模型 + +模型在智能体文件中配置。每个 provider 列出它对外暴露的模型,当会话未设置 `model` +时则使用 `default_model`。 + +```acl +default_model = "provider/fast-model" + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "fast-model" { tool_call = true } + models "review-model" { tool_call = true } +} +``` + +## 按会话设置模型 + +`model` 选项在打开会话时设置。该会话运行的一切——`send`、`run`、`task`、 +`parallel`、`pipeline`——都会使用该模型。同一个智能体配置可以为不同会话驱动不同的 +模型选择。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let fast = agent + .session_builder("/repo") + .options(SessionOptions::new().with_model("provider/fast-model")) + .build() + .await?; + let draft = fast + .send("为这个项目起草一段简短的 README 介绍。", None) + .await? + .text; + println!("草稿:{draft}"); + fast.close().await; + + let review = agent + .session_builder("/repo") + .options(SessionOptions::new().with_model("provider/review-model")) + .build() + .await?; + let critique = review + .send(&format!("审查这段 README 介绍:\n{draft}"), None) + .await?; + println!("审查意见:{}", critique.text); + + review.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +// 快速模型用于高频、低风险工作。 +const fast = agent.session('/repo', { model: 'provider/fast-model' }); +const draft = await fast.run('Draft a short README intro for this project.'); +console.log('draft:', draft); +await fast.close(); + +// 更强的模型用于审查或风险更高的推理。 +const review = agent.session('/repo', { model: 'provider/review-model' }); +const critique = await review.run(`Critique this README intro:\n${draft}`); +console.log('critique:', critique); +await review.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +# 快速模型用于高频、低风险工作。 +fast_opts = SessionOptions() +fast_opts.model = 'provider/fast-model' +fast = agent.session('/repo', fast_opts) +draft = fast.run('Draft a short README intro for this project.') +print('draft:', draft) +fast.close() + +# 更强的模型用于审查或风险更高的推理。 +review_opts = SessionOptions() +review_opts.model = 'provider/review-model' +review = agent.session('/repo', review_opts) +critique = review.run(f'Critique this README intro:\n{draft}') +print('critique:', critique) +review.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + fast, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/fast-model", + }) + if err != nil { + log.Fatal(err) + } + draft, err := fast.Run(ctx, "为这个项目起草一段简短的 README 介绍。") + if err != nil { + log.Fatal(err) + } + fmt.Println("草稿:", draft.Text) + fast.Close(ctx) + + review, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/review-model", + }) + if err != nil { + log.Fatal(err) + } + defer review.Close(context.Background()) + critique, err := review.Run(ctx, "审查这段 README 介绍:\n"+draft.Text) + if err != nil { + log.Fatal(err) + } + fmt.Println("审查意见:", critique.Text) +} +``` + + + + +## 按工作智能体覆盖模型 + +worker 智能体通过各自的规格进行注册。为某个 worker 指定自己的 `model`,即可让它运行 +在与委派它的会话不同(通常更小、更廉价)的模型上。编排会话保留自己的 `model`,只有被 +委派出去的工作才会运行在该 worker 的模型上。 + + + + +```rust +use a3s_code_core::{ + subagent::ModelConfig as WorkerModelConfig, Agent, SessionOptions, WorkerAgentSpec, +}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let mut scout = WorkerAgentSpec::read_only("scout", "读取文件并报告发现。"); + scout.model = Some(WorkerModelConfig::from_model_ref( + "provider/fast-model", + )); + let session = agent + .session_builder("/repo") + .options( + SessionOptions::new() + .with_model("provider/review-model") + .with_worker_agent(scout), + ) + .build() + .await?; + + let findings = session + .tool( + "task", + json!({ + "agent": "scout", + "description": "列出公共 API", + "prompt": "列出 src/ 中的所有公共 API。" + }), + ) + .await?; + let plan = session + .send( + &format!("根据这些发现提出重构方案:\n{}", findings.output), + None, + ) + .await?; + println!("{}", plan.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session('/repo', { + // 编排器继续使用更强的模型。 + model: 'provider/review-model', + // 高频探索使用成本更低的模型。 + workerAgents: [ + { + name: 'scout', + description: 'Reads files and reports findings.', + model: 'provider/fast-model', + }, + ], +}); + +// 把探索委派给低成本工作智能体,再由强模型完成推理。 +const findings = await session.task({ + agent: 'scout', + description: 'List public APIs', + prompt: 'List every public API in src/.', +}); +const plan = await session.run( + `Given these findings, propose a refactor:\n${findings.output}`, +); +console.log(plan); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, WorkerAgentSpec + +agent = Agent.create("agent.acl") + +opts = SessionOptions() +# 编排器继续使用更强的模型。 +opts.model = 'provider/review-model' +# 高频探索使用成本更低的模型。 +scout = WorkerAgentSpec( + name='scout', + description='Reads files and reports findings.', +) +scout.model = 'provider/fast-model' +opts.worker_agents = [scout] +session = agent.session('/repo', opts) + +# 把探索委派给低成本工作智能体,再由强模型完成推理。 +findings = session.task({ + "agent": "scout", + "description": "List public APIs", + "prompt": "List every public API in src/.", +}) +plan = session.run(f'Given these findings, propose a refactor:\n{findings.output}') +print(plan) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + Model: "provider/review-model", + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + findings, err := session.Task(ctx, code.DelegateTaskOptions{ + Agent: "explore", + Description: "列出公共 API", + Prompt: "列出 src/ 中的所有公共 API。", + Model: "provider/fast-model", + }) + if err != nil { + log.Fatal(err) + } + plan, err := session.Run( + ctx, + "根据这些发现提出重构方案:\n"+findings.Output, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(plan.Text) +} +``` + + + + +说明: + +- `model` 的取值是一个标识符字符串,由你的运行时解析为智能体文件中声明的某个模型—— + SDK 中没有任何硬编码的模型名称。 +- worker 智能体的 `model` 仅作用于该智能体被委派的工作。会话自身的 + `send`/`run`/`task` 调用仍使用会话的 `model`。 +- 未设置 `model` 的 worker 会继承会话的 `model`,因此你只需为那些换用不同模型确有收益 + 的智能体进行覆盖即可。 + +展示会话 `model` 选项的可运行版本位于 +`sdk/node/examples/basic/test_api_alignment.ts`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/orchestration.mdx b/website/docs/v8.5.1/zh/guide/examples/orchestration.mdx new file mode 100644 index 00000000..71c4ed3e --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/orchestration.mdx @@ -0,0 +1,785 @@ +--- +title: '编排' +description: '用 session.parallel 扇出独立任务,用 session.pipeline 构建按条目执行的多阶段链,用 session.parallelResumable 恢复带日志的运行。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 编排 + +本页展示 A3S Code 中的可编程编排原语:用于扇出的 `session.parallel`、用于按条目执行多阶段链的 `session.pipeline`,以及用于在崩溃后仍可恢复的带日志运行的 `session.parallelResumable`。`parallel` 还接受一个可选的 token 预算,让整个扇出共享同一个账本。当你有多个相互独立的子代理任务时使用 parallel;当你需要让每个输入流经一组有序阶段时使用 pipeline。 + +关于这些原语背后的概念模型,请参阅[编排](/guide/orchestration)。 + +## 用 `session.parallel` 扇出 + +`parallel` 接收一个 `AgentStepSpec` 数组并发执行它们,**按输入顺序**(而非完成顺序)为每个 spec 返回一个 `StepOutcome`。每个 spec 路由到一个具名子代理(`explore`、`plan`、`review`、`verification`、`general` 等)。在 spec 上设置 `outputSchema` / `output_schema` 即可拿到经过 schema 校验的 `structured` 结果。 + + + + +```rust +use a3s_code_core::{Agent, AgentStepSpec}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow(); + + let outcomes = workflow + .parallel(vec![ + AgentStepSpec::new( + "langs", + "general", + "列出语言", + "列出三种系统编程语言。", + ) + .with_max_steps(2), + AgentStepSpec::new( + "verdict", + "general", + "分类", + "Rust 是否无需 GC 就能保证内存安全?只回答是或否。", + ) + .with_max_steps(2), + ]) + .await; + + for outcome in outcomes { + println!( + "[parallel] {}: success={}", + outcome.task_id, outcome.success + ); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +// 各步骤互不依赖;结果按输入顺序返回,而不是按完成顺序返回。 +const outcomes = await session.parallel([ + { + taskId: 'langs', + agent: 'general', + description: 'list', + prompt: 'Name three systems languages.', + maxSteps: 2, + }, + { + taskId: 'safe', + agent: 'general', + description: 'classify', + prompt: 'Is Rust memory-safe without a GC? yes/no.', + maxSteps: 2, + }, +]); + +for (const o of outcomes) { + console.log(`[parallel] ${o.taskId}: success=${o.success}`); +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +# 各步骤互不依赖;结果按输入顺序返回,而不是按完成顺序返回。 +outcomes = session.parallel([ + { + "task_id": "langs", + "agent": "general", + "description": "list languages", + "prompt": "Name three systems programming languages, comma-separated.", + "max_steps": 2, + }, + { + "task_id": "verdict", + "agent": "general", + "description": "classify", + "prompt": "Is Rust memory-safe without a GC? Answer yes or no.", + "max_steps": 2, + # 这个步骤会返回经过模式校验的结构化输出。 + "output_schema": { + "type": "object", + "properties": {"memory_safe": {"type": "boolean"}}, + "required": ["memory_safe"], + }, + }, +]) + +for o in outcomes: + print(f"[parallel] {o['task_id']}: success={o['success']} structured={o.get('structured')}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + result, err := session.Parallel(ctx, []code.AgentStepSpec{ + { + TaskID: "langs", + Description: "列出语言", + Agent: "general", + Prompt: "列出三种系统编程语言,用逗号分隔。", + MaxSteps: &maxSteps, + }, + { + TaskID: "verdict", + Description: "分类", + Agent: "general", + Prompt: "Rust 是否无需 GC 就能保证内存安全?只回答是或否。", + MaxSteps: &maxSteps, + OutputSchema: json.RawMessage(`{ + "type":"object", + "properties":{"memory_safe":{"type":"boolean"}}, + "required":["memory_safe"] + }`), + }, + }, nil) + if err != nil { + log.Fatal(err) + } + for _, outcome := range result.Outcomes { + fmt.Printf( + "[parallel] %s: success=%t structured=%s\n", + outcome.TaskID, + outcome.Success, + outcome.Structured, + ) + } +} +``` + + + + +结果在 Python 中是字典,在 Node 中是对象,在 Go 中是 `StepOutcome` 值。会话选项 +`maxParallelTasks` / `max_parallel_tasks` / `MaxParallelTasks` 限制并发量;多出的 +spec 会排队,而返回的结果数组仍然完整且保持顺序。 + +## 用 `session.pipeline` 构建按条目执行的链 + +`pipeline` 接收一个输入 `items` 列表和一个有序的 `stages` 列表。每个条目独立地流经各个阶段——阶段之间**没有屏障**,因此一个较快的条目可以在一个较慢的条目仍处于阶段 1 时就到达阶段 2。阶段回调接收一个 `ctx`:第一个阶段看到 `ctx.item`,后续阶段看到 `ctx.previous`(上一个 `StepOutcome`,你可以基于其 `.output` 继续构建)。返回下一个 spec 以继续,或返回 `null` / `None` 以提前停止该条目的链。 + + + + +```rust +use std::sync::Arc; + +use a3s_code_core::{Agent, AgentStepSpec, PipelineStage}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow(); + + let stages: Vec> = vec![ + Arc::new(|_, item| { + Some( + AgentStepSpec::new( + "summarize", + "general", + "总结", + format!("用一句话说明什么是{item}。"), + ) + .with_max_steps(2), + ) + }), + Arc::new(|previous, _| { + previous.map(|outcome| { + AgentStepSpec::new( + "classify", + "general", + "分类", + format!( + "只回答是或否:下面描述的是编程语言吗?\n\n{}", + outcome.output + ), + ) + .with_max_steps(2) + }) + }), + ]; + + let results = workflow + .pipeline(vec!["Rust 编程语言".to_string()], stages) + .await; + for result in results.into_iter().flatten() { + println!("[pipeline] {}", result.output); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +// 第二阶段使用第一阶段的输出。阶段回调不能抛出异常; +// 返回 null 可停止当前条目的处理链。 +const results = await session.pipeline( + ['the Rust programming language'], + [ + (ctx) => ({ + taskId: 'sum', + agent: 'general', + description: 'summarize', + prompt: `In one sentence, what is ${ctx.item}?`, + maxSteps: 2, + }), + (ctx) => ({ + taskId: 'cls', + agent: 'general', + description: 'classify', + prompt: `Reply YES or NO: does this describe a programming language?\n\n${ctx.previous.output}`, + maxSteps: 2, + }), + ], +); + +for (const r of results) { + console.log( + `[pipeline] final=${r === null ? null : JSON.stringify(r.output.slice(0, 60))}`, + ); +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +# 每个条目依次经过各阶段,第二阶段使用第一阶段的输出。 +# 阶段返回 None(抛出的异常也会被视为 None)可停止当前条目的处理链。 +results = session.pipeline( + ["the Rust programming language"], + [ + lambda ctx: { + "task_id": "summarize", + "agent": "general", + "description": "summarize", + "prompt": f"In one sentence, what is {ctx['item']}?", + "max_steps": 2, + }, + lambda ctx: { + "task_id": "classify", + "agent": "general", + "description": "classify", + "prompt": "Reply with one word YES or NO: does this describe a " + f"programming language?\n\n{ctx['previous']['output']}", + "max_steps": 2, + }, + ], +) + +for r in results: + print(f"[pipeline] final={None if r is None else r['output'][:60]!r}") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + results, err := session.Pipeline( + ctx, + []any{"Rust 编程语言"}, + []code.PipelineStage{ + func( + _ context.Context, + stage code.PipelineContext, + ) (*code.AgentStepSpec, error) { + return &code.AgentStepSpec{ + TaskID: "summarize", + Agent: "general", + Description: "总结", + Prompt: fmt.Sprintf("用一句话说明什么是 %v。", stage.Item), + MaxSteps: &maxSteps, + }, nil + }, + func( + _ context.Context, + stage code.PipelineContext, + ) (*code.AgentStepSpec, error) { + return &code.AgentStepSpec{ + TaskID: "classify", + Agent: "general", + Description: "分类", + Prompt: "只回答是或否:下面描述的是编程语言吗?\n\n" + stage.Previous.Output, + MaxSteps: &maxSteps, + }, nil + }, + }, + ) + if err != nil { + log.Fatal(err) + } + for _, outcome := range results { + if outcome != nil { + fmt.Println(outcome.Output) + } + } +} +``` + + + + +与 `parallel` 的关键区别:阶段是有序且相互依赖的,但各条目在阶段之间**不会** +彼此等待。Node 的阶段回调绝不能抛出异常——出错时返回 `null`;Python 的阶段可以 +抛出并被视为 `None`;Go 阶段返回 `(*AgentStepSpec, error)`。 + +## 用 `session.parallelResumable` 恢复运行 + +`parallelResumable` 就是带日志的 `parallel`。它的第一个参数是 `specs`,第二个参数是 +稳定的 `workflowId`;每个步骤的结果都会被记录到会话的 store,因此如果进程在 +运行中途崩溃,你可以用同一个 `workflowId` 再次调用它,已完成步骤会从日志中重放。 +它**需要一个会话 store**——打开会话时传入 `sessionStore`、`session_store` 或 +Go 的 `FileSessionStoreDir`。 + + + + +Rust 通过具名 `workflow.phase` 暴露相同的检查点行为。配置会话存储后,每个 phase +都是可恢复的屏障。 + +```rust +use a3s_code_core::{Agent, AgentStepSpec, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_file_session_store("./.a3s/sessions") + .with_session_id("nightly-audit-session"), + ) + .build() + .await?; + let workflow = session.workflow(); + + let outcomes = workflow + .phase( + "nightly-audit", + vec![ + AgentStepSpec::new( + "deps", + "general", + "审计依赖", + "检查清单中的过时依赖。", + ) + .with_max_steps(2), + AgentStepSpec::new( + "tests", + "verification", + "运行测试", + "运行测试套件并总结失败。", + ) + .with_max_steps(2), + ], + ) + .await; + + for outcome in outcomes { + println!("{}:{}", outcome.task_id, outcome.success); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +// parallelResumable 会把日志写入会话存储;没有存储时会抛出异常。 +const session = agent.session('.', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +// 签名是 (specs, workflowId):先传规格,再传稳定的工作流标识。 +const outcomes = await session.parallelResumable( + [ + { + taskId: 'deps', + agent: 'general', + description: 'audit deps', + prompt: 'Check manifests for outdated dependencies.', + maxSteps: 2, + }, + { + taskId: 'tests', + agent: 'verification', + description: 'run tests', + prompt: 'Run the test suite and summarize failures.', + maxSteps: 2, + }, + ], + 'nightly-audit', +); + +// 中断后使用相同 workflowId 重启。已完成步骤从日志加载; +// 完全成功的运行会删除检查点。 +console.log(outcomes.map((o) => `${o.taskId}:${o.success}`).join(' ')); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, FileSessionStore + +agent = Agent.create("agent.acl") +# parallel_resumable 会把日志写入会话存储;没有存储时会抛出异常。 +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session(".", opts) + +# 签名是 (specs, workflow_id):先传规格,再传稳定的工作流标识。 +outcomes = session.parallel_resumable( + [ + {"task_id": "deps", "agent": "general", "description": "audit deps", "prompt": "Check manifests for outdated dependencies.", "max_steps": 2}, + {"task_id": "tests", "agent": "verification", "description": "run tests", "prompt": "Run the test suite and summarize failures.", "max_steps": 2}, + ], + "nightly-audit", +) + +# 中断后使用相同 workflow_id 重启。已完成步骤从日志加载; +# 完全成功的运行会删除检查点。 +print(" ".join(f"{o['task_id']}:{o['success']}" for o in outcomes)) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + options := &code.SessionOptions{ + SessionID: "nightly-audit-session", + FileSessionStoreDir: "./.a3s/sessions", + } + session, err := agent.Session(ctx, ".", options) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + maxSteps := uint(2) + outcomes, err := session.ParallelResumable(ctx, []code.AgentStepSpec{ + { + TaskID: "deps", + Agent: "general", + Description: "审计依赖", + Prompt: "检查 manifest 中是否存在过期依赖。", + MaxSteps: &maxSteps, + }, + { + TaskID: "tests", + Agent: "verification", + Description: "运行测试", + Prompt: "运行测试套件并总结失败。", + MaxSteps: &maxSteps, + }, + }, "nightly-audit") + if err != nil { + log.Fatal(err) + } + for _, outcome := range outcomes { + fmt.Printf("%s:%t ", outcome.TaskID, outcome.Success) + } +} +``` + + + + +## 用 `session.parallel` 做预算受限的扇出 + +传入 token 预算后,所有子代理就会汇入**同一个账本**。传入预算时, +`parallel` 解析为 `{ outcomes, budget }`(账本快照)而非原来的结果数组;一旦达到 +上限,之后启动的 step 会被拒绝(`success: false`)。它是软上限——宽扇出可能冲过 +上限几个在飞回合;在飞工作绝不会被强杀。 + + + + +```rust +use a3s_code_core::{Agent, AgentStepSpec}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + let workflow = session.workflow_with_token_budget(Some(50_000)); + + let outcomes = workflow + .parallel(vec![ + AgentStepSpec::new("a", "general", "问题一", "回答:准备。") + .with_max_steps(2), + AgentStepSpec::new("b", "general", "问题二", "回答:开始。") + .with_max_steps(2), + ]) + .await; + for outcome in outcomes { + println!("{}: success={}", outcome.task_id, outcome.success); + } + + if let Some(budget) = workflow.budget_snapshot() { + println!( + "spent {} / {} tokens", + budget.consumed_tokens, + budget.limit_tokens.unwrap_or_default() + ); + } + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.', {}); + +const specs = [ + { + taskId: 'a', + agent: 'general', + description: 'q1', + prompt: 'Reply with one word: ready.', + maxSteps: 2, + }, + { + taskId: 'b', + agent: 'general', + description: 'q2', + prompt: 'Reply with one word: go.', + maxSteps: 2, + }, +]; + +// 传入预算时,parallel() 解析为 { outcomes, budget } —— 所有子代理共享一个账本。 +// (不传时,parallel(specs) 返回原来的数组。) +const { outcomes, budget } = await session.parallel(specs, 50_000); +for (const o of outcomes) + console.log(`[budget] ${o.taskId}: success=${o.success}`); +console.log(`spent ${budget.consumedTokens} / ${budget.limitTokens} tokens`); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +session = agent.session(".", SessionOptions()) + +specs = [ + {"task_id": "a", "agent": "general", "description": "q1", "prompt": "Reply with one word: ready.", "max_steps": 2}, + {"task_id": "b", "agent": "general", "description": "q2", "prompt": "Reply with one word: go.", "max_steps": 2}, +] + +# 传入预算时,parallel() 返回 {"outcomes", "budget"}(共享账本)。 +res = session.parallel(specs, budget_tokens=50_000) +for o in res["outcomes"]: + print(f"[budget] {o['task_id']}: success={o['success']}") +print(f"spent {res['budget']['consumed_tokens']} / {res['budget']['limit_tokens']} tokens") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + maxSteps := uint(2) + budgetTokens := uint64(50_000) + result, err := session.Parallel(ctx, []code.AgentStepSpec{ + {TaskID: "a", Agent: "general", Description: "问题 1", Prompt: "只回答一个词:ready。", MaxSteps: &maxSteps}, + {TaskID: "b", Agent: "general", Description: "问题 2", Prompt: "只回答一个词:go。", MaxSteps: &maxSteps}, + }, &budgetTokens) + if err != nil { + log.Fatal(err) + } + for _, outcome := range result.Outcomes { + fmt.Printf("[budget] %s: success=%t\n", outcome.TaskID, outcome.Success) + } + if result.Budget != nil && result.Budget.LimitTokens != nil { + fmt.Printf( + "spent %d / %d tokens\n", + result.Budget.ConsumedTokens, + *result.Budget.LimitTokens, + ) + } +} +``` + + + + +说明: + +- 三个原语都返回按输入顺序对齐的结果。Go 使用 `StepOutcome`;Node 使用对象; + Python 使用字典。 +- 在 spec 上设置 `outputSchema` / `output_schema` / `OutputSchema`,可在 + `structured` / `Structured` 中拿到解析后的结果。 +- `maxSteps` / `max_steps` / `MaxSteps` 限制每个子代理的步数;会话选项 + `maxParallelTasks` / `max_parallel_tasks` / `MaxParallelTasks` 限制扇出并发量。 +- 给 `parallel` 传入 token 预算即可让整个扇出对同一个账本计数。Go 把 `*uint64` + 作为 `Parallel` 的第三个参数,并从 `ParallelResult.Budget` 读取账本。它是软上限。 +- Node 的 pipeline 阶段回调绝不能抛出异常——出错时返回 `null`。Python 的阶段可以 + 抛出并被视为 `None`;Go 阶段返回 `error`。 + +可运行的 Node.js 和 Python 版本见 +`sdk/node/examples/orchestration/parallel-pipeline.mjs` 和 +`sdk/python/examples/orchestration_workflow.py`;上面的 Rust 与 Go 页签均为自包含示例。 diff --git a/website/docs/v8.5.1/zh/guide/examples/planning.mdx b/website/docs/v8.5.1/zh/guide/examples/planning.mdx new file mode 100644 index 00000000..48c361c7 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/planning.mdx @@ -0,0 +1,134 @@ +--- +title: '规划模式' +description: '使用 planningMode 让智能体先规划再行动' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 规划模式 + +Planning 模式让 session 在开始调用工具之前先生成一份结构化计划。它适合多步骤工作 +(重构、发布评审、审计),你希望 agent 先把目标拆解清楚,而不是直接动手修改。 + +通过会话的规划选项设置:Rust 和 Go 使用 `PlanningMode`,Node.js 使用 +`planningMode`,Python 使用 `planning_mode`。可接受的值为: + +| 值 | 行为 | +| ------------ | ----------------------------------------- | +| `"auto"` | 由 runtime 根据消息内容判断何时值得规划。 | +| `"enabled"` | 对每个请求都强制规划,即使是简单请求。 | +| `"disabled"` | 完全跳过规划,走最低延迟路径。 | + +`"auto"` 是默认的结构化预分析路径。 + + + + +```rust +use a3s_code_core::{Agent, PlanningMode, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_planning_mode(PlanningMode::Auto)) + .build() + .await?; + + let result = session.send("规划并完成发布就绪审查。", None).await?; + println!("{}", result.text); + println!("执行了 {} 次工具调用", result.tool_calls_count); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + planningMode: 'auto', +}); + +const result = await session.send( + 'Plan and complete the release-readiness review.', +); +console.log(result.text); +console.log(`${result.toolCallsCount} tool calls executed`); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.planning_mode = "auto" +session = agent.session("/repo", opts) + +result = session.send("Plan and complete the release-readiness review.") +print(result.text) +print(f"{result.tool_calls_count} tool calls executed") + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PlanningMode: code.PlanningAuto, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "规划并完成发布就绪审查。") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) + fmt.Printf("执行了 %d 次工具调用\n", result.ToolCallsCount) +} +``` + + + + +Planning 状态会挂到 run 上,因此宿主 UI 可以从 run events 渲染任务清单,并在 agent +工作时更新完成情况。Planning 负责组织工作;完成证据仍然来自验证命令。 + +可运行的 session 示例位于 +`sdk/node/examples/orchestration/parallel-pipeline.mjs` 和 +`sdk/python/examples/orchestration_workflow.py`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/prompt-slots.mdx b/website/docs/v8.5.1/zh/guide/examples/prompt-slots.mdx new file mode 100644 index 00000000..b71675f3 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/prompt-slots.mdx @@ -0,0 +1,358 @@ +--- +title: '提示词插槽' +description: '在不覆盖核心行为的前提下,定制智能体的角色设定、准则与回复风格。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 提示词插槽 + +Prompt slots 是以声明式方式塑造智能体系统提示词的会话选项。请把它们用于宿主级行为—— +角色设定、编码规范、输出风格——这些内容不应写在每次的用户 prompt 里。这些槽位叠加在 +智能体内置指令之上,因此核心工具行为(读取、写入、运行命令)会被保留。 + +共有四个槽位: + +| 槽位 | 用途 | +| ---------------------------------- | ---------------------------- | +| `role` / `role` | 智能体采用的角色设定。 | +| `guidelines` / `guidelines` | 智能体必须遵循的规范与规则。 | +| `responseStyle` / `response_style` | 智能体回复的格式方式。 | +| `extra` / `extra` | 逐字追加的自由格式指令。 | + +## 基本用法 + +在打开会话时设置任意子集的槽位。它们会应用于该会话的每一轮对话。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, SystemPromptSlots}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let slots = SystemPromptSlots::default() + .with_role("发布就绪审查员") + .with_guidelines("先找阻塞项,再提改进;完成结论必须有命令证据。") + .with_response_style("简洁,发现优先"); + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots(slots)) + .build() + .await?; + + let result = session.send("这个仓库可以发布了吗?", None).await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +async function main() { + // create() 接受 .acl 文件路径或内联 ACL 字符串。 + const agent = await Agent.create('agent.acl'); + + const session = agent.session('/repo', { + role: 'release-readiness reviewer', + guidelines: + 'Find blockers before improvements. Require command evidence for done claims.', + responseStyle: 'concise, findings first', + }); + + const result = await session.send('Is this repo ready to ship?'); + console.log(result.text); + + session.close(); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +def main(): + # create() 接受 .acl 文件路径或内联 ACL 字符串。 + agent = Agent.create('agent.acl') + + opts = SessionOptions() + opts.role = 'release-readiness reviewer' + opts.guidelines = 'Find blockers before improvements. Require command evidence for done claims.' + opts.response_style = 'concise, findings first' + session = agent.session('/repo', opts) + + result = session.send('Is this repo ready to ship?') + print(result.text) + + session.close() + +main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PromptSlots: &code.PromptSlots{ + Role: "发布就绪审查员", + Guidelines: "先找阻塞项,再提改进;完成结论必须有命令证据。", + ResponseStyle: "简洁,发现优先", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "这个仓库可以发布了吗?") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +## 逐个槽位说明 + +这四个槽位彼此独立组合。仅设置角色的会话、带有严格准则的评审者、以及追加自由格式指令的 +会话,使用的都是同一组选项。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, SystemPromptSlots}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + + let role_only = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default().with_role("你是擅长异步编程的资深 Rust 开发者。"), + )) + .build() + .await?; + role_only.close().await; + + let reviewer = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default() + .with_role("你是 Python 代码审查员。") + .with_guidelines("始终检查类型提示,并标记所有 `eval()` 调用。") + .with_response_style("使用要点列表,保持简洁。"), + )) + .build() + .await?; + reviewer.close().await; + + let extra_only = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default().with_extra("每次回复都以“-- A3S”结尾。"), + )) + .build() + .await?; + extra_only.close().await; + + let file_manager = agent + .session_builder("/repo") + .options(SessionOptions::new().with_prompt_slots( + SystemPromptSlots::default() + .with_role("你是极简文件管理员。") + .with_guidelines("只有用户明确要求时才创建文件。"), + )) + .build() + .await?; + let result = file_manager + .send( + "创建 test.txt,写入“prompt slots work”,然后读回内容。", + None, + ) + .await?; + println!("{}", result.text); + + file_manager.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +// 1. Custom role only. +let session = agent.session(workspace, { + role: 'You are a senior Rust developer who specializes in async programming.', +}); + +// 2. Role + guidelines + response style. +session = agent.session(workspace, { + role: 'You are a Python code reviewer.', + guidelines: 'Always check for type hints. Flag any use of `eval()`.', + responseStyle: 'Reply in bullet points. Be concise.', +}); + +// 3. Extra freeform instructions only. +session = agent.session(workspace, { + extra: "Always end your response with '-- A3S'", +}); + +// 无论如何配置插槽,核心工具行为都保持不变。 +session = agent.session(workspace, { + role: 'You are a minimalist file manager.', + guidelines: 'Only create files when explicitly asked.', +}); +const result = await session.send( + "Create a file called test.txt with the content 'prompt slots work'. Then read it back.", +); +``` + + + + +```python +# 1. Custom role only. +opts = SessionOptions() +opts.role = 'You are a senior Rust developer who specializes in async programming.' +session = agent.session(workspace, opts) + +# 2. Role + guidelines + response style. +opts = SessionOptions() +opts.role = 'You are a Python code reviewer.' +opts.guidelines = 'Always check for type hints. Flag any use of `eval()`.' +opts.response_style = 'Reply in bullet points. Be concise.' +session = agent.session(workspace, opts) + +# 3. Extra freeform instructions only. +opts = SessionOptions() +opts.extra = "Always end your response with '-- A3S'" +session = agent.session(workspace, opts) + +# 无论如何配置插槽,核心工具行为都保持不变。 +opts = SessionOptions() +opts.role = 'You are a minimalist file manager.' +opts.guidelines = 'Only create files when explicitly asked.' +session = agent.session(workspace, opts) +result = session.send( + "Create a file called test.txt with the content 'prompt slots work'. Then read it back.", +) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func open( + ctx context.Context, + agent *code.Agent, + slots code.PromptSlots, +) *code.Session { + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + PromptSlots: &slots, + }) + if err != nil { + log.Fatal(err) + } + return session +} + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + roleOnly := open(ctx, agent, code.PromptSlots{ + Role: "你是擅长异步编程的资深 Rust 开发者。", + }) + roleOnly.Close(ctx) + + reviewer := open(ctx, agent, code.PromptSlots{ + Role: "你是 Python 代码审查员。", + Guidelines: "始终检查类型提示,并标记所有 `eval()` 调用。", + ResponseStyle: "使用要点列表,保持简洁。", + }) + reviewer.Close(ctx) + + extraOnly := open(ctx, agent, code.PromptSlots{ + Extra: "每次回复都以“-- A3S”结尾。", + }) + extraOnly.Close(ctx) + + fileManager := open(ctx, agent, code.PromptSlots{ + Role: "你是极简文件管理员。", + Guidelines: "只有用户明确要求时才创建文件。", + }) + defer fileManager.Close(context.Background()) + result, err := fileManager.Run( + ctx, + "创建 test.txt,写入“prompt slots work”,然后读回内容。", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +槽位用于定制角色设定与行为规则;它们不会禁用工具,也不会改变智能体的核心循环。请把 +任务相关的请求保留在 `send` 消息中,而把应在会话每一轮都生效的行为交给这些槽位。 + +可运行版本位于 `sdk/node/examples/skills/test_prompt_slots.ts`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/quick-start.mdx b/website/docs/v8.5.1/zh/guide/examples/quick-start.mdx new file mode 100644 index 00000000..f81e9575 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/quick-start.mdx @@ -0,0 +1,122 @@ +--- +title: '快速开始' +description: '创建一个 agent、打开会话、运行一轮对话并读取结果。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 快速开始 + +最小可用的程序:从 ACL 文件创建一个 agent,在项目目录上打开会话,使用 `send` +运行一轮对话,打印回复文本,并查看运行时为该轮生成的验证摘要。 + + + + +```rust +use a3s_code_core::Agent; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + // Agent::new 接受 .acl 文件路径或内联 ACL 源文本。 + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let result = session.send("列出当前目录中的文件。", None).await?; + println!("{}", result.text); + + // 运行时在生成这个回合时执行的验证。 + println!("{}", result.verification_summary_text()); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +// Agent.create 接受 .acl 文件路径或内联 ACL 字符串。 +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const result = await session.send('List the files in this directory.'); +console.log(result.text); + +// 运行时在生成这个回合时执行的验证。 +console.log(result.verificationSummaryText); + +session.close(); +``` + + + + +```python +from a3s_code import Agent + +# Agent.create 接受 .acl 文件路径或内联 ACL 字符串。 +agent = Agent.create('agent.acl') +session = agent.session('.') + +result = session.send('List the files in this directory.') +print(result.text) + +# 运行时在生成这个回合时执行的验证。 +print(result.verification_summary_text) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Run(ctx, "列出当前目录中的文件。") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) + fmt.Println(result.VerificationSummaryText) +} +``` + + + + +四种 SDK 的 Agent 工厂都接收 `.acl` 文件路径或内联 ACL 源文本。使用完毕后, +请务必关闭 Session,让 Runtime 刷新状态并释放资源。Go 的关闭操作接收 Context, +因为它还需要停止桥接程序持有的原生资源。 + +## 下一步 + +- [流式输出](/guide/examples/streaming) — 在 token 到达时实时读取 +- [会话](/guide/sessions) — 持久化并恢复对话 diff --git a/website/docs/v8.5.1/zh/guide/examples/ripgrep-context.mdx b/website/docs/v8.5.1/zh/guide/examples/ripgrep-context.mdx new file mode 100644 index 00000000..a3bb0820 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/ripgrep-context.mdx @@ -0,0 +1,155 @@ +--- +title: 'Ripgrep 上下文构建器' +description: '在向智能体提问前,使用 grep 和 glob 收集代码上下文。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# Ripgrep 上下文构建器 + +通过 `session.grep` 和 `session.glob` 进行快速代码搜索,可以收集相关的文件和匹配行, +然后将它们注入到提示中 —— 这是在 agent 开始推理之前的一个轻量级检索步骤。当你希望将 +agent 限定在大型代码库的某个特定切片,而不是让它从头开始探索时,可以使用这种方式。 + + + + +```rust +use a3s_code_core::Agent; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let files = session.glob("src/**/*.rs").await?; + let hits = session.grep("create_session").await?; + let context = format!( + "范围内的文件:\n{}\n\n“create_session”的匹配项:\n{}", + files.join("\n"), + hits + ); + let answer = session + .send( + &format!("只使用以下上下文,解释 create_session 如何连接:\n\n{context}"), + None, + ) + .await?; + println!("{}", answer.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +// 1. Find candidate files by glob pattern (returns a list of paths). +const files = await session.glob('src/**/*.ts'); + +// 2. Search the workspace for the symbol we care about (returns ripgrep text). +const hits = await session.grep('createSession'); + +// 3. Build a context string and feed it into a focused prompt. +const context = [ + `Files in scope:\n${files.join('\n')}`, + `Matches for "createSession":\n${hits}`, +].join('\n\n'); + +const answer = await session.run( + `Using only this context, explain how createSession is wired up:\n\n${context}`, +); +console.log(answer); +``` + + + + +```python +from a3s_code import Agent + +agent = Agent.create("agent.acl") +session = agent.session('.') + +# 1. Find candidate files by glob pattern (returns a list of paths). +files = session.glob('src/**/*.ts') + +# 2. Search the workspace for the symbol we care about (returns ripgrep text). +hits = session.grep('createSession') + +# 3. Build a context string and feed it into a focused prompt. +context = '\n\n'.join([ + 'Files in scope:\n' + '\n'.join(files), + 'Matches for "createSession":\n' + hits, +]) + +answer = session.run( + f'Using only this context, explain how createSession is wired up:\n\n{context}' +) +print(answer) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "strings" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + files, err := session.Glob(ctx, "sdk/go/**/*.go") + if err != nil { + log.Fatal(err) + } + hits, err := session.Grep(ctx, "Session") + if err != nil { + log.Fatal(err) + } + scope := "范围内的文件:\n" + strings.Join(files, "\n") + + "\n\n“Session”的匹配项:\n" + hits + answer, err := session.Run( + ctx, + "只使用以下上下文,解释 Session 如何连接:\n\n"+scope, + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(answer.Text) +} +``` + + + + +`glob` 返回匹配的文件路径列表,而 `grep` 则以单个字符串的形式返回原始的 ripgrep 输出。 +两者都在本地运行并能快速返回,因此你可以在花费一次模型调用之前,链式执行多次搜索来低成本地 +组装上下文。当你需要文件的完整内容而不仅仅是匹配行时,可以将它们与 `session.readFile` +搭配使用。 diff --git a/website/docs/v8.5.1/zh/guide/examples/security.mdx b/website/docs/v8.5.1/zh/guide/examples/security.mdx new file mode 100644 index 00000000..38532ed3 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/security.mdx @@ -0,0 +1,435 @@ +--- +title: '安全' +description: '通过权限策略、人工确认流程和安全提供器对特权操作进行管控' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 安全 + +智能体可能产生的每一个副作用——写文件、运行 `bash`、执行 git 推送——都会经过权限策略。 +先从 `ask` 或 `deny` 兜底开始,再列出应被 `allow`(放行)、`deny`(拒绝)或进入 `ask`(询问) +路径的模式。若要引入人工把关,可加上确认策略:`ask` 决策会在 `confirmation_required` +事件处暂停,让你的应用(或人工)对每次调用进行批准或拒绝。只要智能体面向真实仓库运行, +就应使用这套机制。 + + + + +```rust +use a3s_code_core::{ + hitl::{ConfirmationPolicy, TimeoutAction}, + permissions::PermissionPolicy, + Agent, AgentEvent, SessionOptions, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let permission_policy = PermissionPolicy::new() + .allow("read(*)") + .allow("search(*)") + .allow("ls(*)") + .allow("bash(git status:*)") + .deny("write(**/.env*)") + .deny("bash(rm -rf*)") + .ask("write(*)") + .ask("edit(*)") + .ask("bash(git push:*)") + .ask("bash(npm publish:*)") + .ask("bash(*)"); + let confirmation_policy = + ConfirmationPolicy::enabled().with_timeout(120_000, TimeoutAction::Reject); + + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_permission_policy(permission_policy) + .with_confirmation_policy(confirmation_policy), + ) + .build() + .await?; + + let (mut events, lifecycle) = session + .stream("提升版本号并推送发布。", None) + .await?; + while let Some(event) = events.recv().await { + if let AgentEvent::ConfirmationRequired { + tool_id, + tool_name, + args, + .. + } = event + { + println!("[confirm] {tool_name}\n{args:#}"); + session + .confirm_tool_use( + &tool_id, + false, + Some("Rejected by the host review".into()), + ) + .await?; + } + } + let _ = lifecycle.await; + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session(process.cwd(), { + permissionPolicy: { + allow: ['read(*)', 'search(*)', 'ls(*)', 'bash(git status:*)'], + deny: ['write(**/.env*)', 'bash(rm -rf*)'], + ask: [ + 'write(*)', + 'edit(*)', + 'bash(git push:*)', + 'bash(npm publish:*)', + 'bash(*)', + ], + defaultDecision: 'ask', + }, + // 把 `ask` 模式转为人工确认流程。 + confirmationPolicy: { + enabled: true, + defaultTimeoutMs: 120000, + timeoutAction: 'reject', + }, +}); + +// 以流式方式执行,并在确认请求到达时逐个处理。 +const stream = await session.stream('Bump the version and push the release'); +while (true) { + const next = await stream.next(); + if (next.done || !next.value) break; + + const event = next.value; + if (event.type === 'confirmation_required') { + // 查询待处理请求,以显示更完整的信息。 + const [pending] = await session.pendingConfirmations(); + const toolId = pending?.toolId ?? event.toolId; + console.log(`[confirm] ${pending?.toolName ?? event.toolName}`); + console.log(JSON.stringify(pending?.args ?? {}, null, 2)); + + // 实际应用应在这里询问用户。 + const approved = false; // deny risky operations by default + if (toolId) + await session.confirmToolUse(toolId, approved, 'Reviewed by host'); + } +} + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, PermissionPolicy, ConfirmationPolicy + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.permission_policy = PermissionPolicy( + allow=["read(*)", "search(*)", "ls(*)", "bash(git status:*)"], + deny=["write(**/.env*)", "bash(rm -rf*)"], + ask=["write(*)", "edit(*)", "bash(git push:*)", "bash(npm publish:*)", "bash(*)"], + default_decision="ask", + ) + # 把 `ask` 模式转为人工确认流程。 + opts.confirmation_policy = ConfirmationPolicy( + enabled=True, + default_timeout_ms=120_000, + timeout_action="reject", + ) + + session = agent.session(".", opts) + + # 以流式方式执行,并在确认请求到达时逐个处理。 + for event in session.stream("Bump the version and push the release"): + if event.event_type == "confirmation_required": + # 查询待处理请求,以显示更完整的信息。 + pending = session.pending_confirmations() + first = pending[0] if pending else {} + tool_id = first.get("tool_id") or event.tool_id + print(f"[confirm] {first.get('tool_name') or event.tool_name}") + + # 实际应用应在这里询问用户。 + approved = False # deny risky operations by default + if tool_id: + session.confirm_tool_use(tool_id, approved, "Reviewed by host") + + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + enabled := true + timeoutMS := uint64(120_000) + session, err := agent.Session( + ctx, + ".", + &code.SessionOptions{ + PermissionPolicy: &code.PermissionPolicy{ + Allow: []string{"read(*)", "search(*)", "ls(*)", "bash(git status:*)"}, + Deny: []string{"write(**/.env*)", "bash(rm -rf*)"}, + Ask: []string{"write(*)", "edit(*)", "bash(git push:*)", "bash(npm publish:*)", "bash(*)"}, + DefaultDecision: "ask", + }, + ConfirmationPolicy: &code.ConfirmationPolicy{ + Enabled: &enabled, + DefaultTimeoutMS: &timeoutMS, + TimeoutAction: "reject", + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + stream, err := session.Stream(ctx, "提升版本号并推送发布。", nil) + if err != nil { + log.Fatal(err) + } + for event := range stream.Events { + if event.Type != code.EventConfirmationRequired { + continue + } + pending, err := session.PendingConfirmations(ctx) + if err != nil { + log.Fatal(err) + } + if len(pending) == 0 { + continue + } + request := pending[0] + args, _ := json.MarshalIndent(request.Args, "", " ") + fmt.Printf("[confirm] %s\n%s\n", request.ToolName, args) + if _, err = session.ConfirmToolUse( + ctx, + request.ToolID, + false, + "Rejected by the Go host review", + ); err != nil { + log.Fatal(err) + } + } + if err := <-stream.Done; err != nil { + log.Fatal(err) + } +} +``` + + + + +## 添加安全提供器 + +`DefaultSecurityProvider` 会启用输入污点追踪和输出净化,独立于权限策略对工具的输入输出 +进行筛查。通过 `securityProvider`(Node)、`security_provider`(Python)传入,或把 +Go 的 `SessionOptions.DefaultSecurity` 设为 `true`;省略则关闭安全功能。 + + + + +```rust +use a3s_code_core::{permissions::PermissionPolicy, Agent, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options( + SessionOptions::new() + .with_default_security() + .with_permission_policy( + PermissionPolicy::new() + .allow("bash(echo:*)") + .ask("bash(*)"), + ), + ) + .build() + .await?; + + let result = session + .send( + "使用 bash 准确运行:echo screened-by-security-provider", + None, + ) + .await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, DefaultSecurityProvider } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +const session = agent.session(process.cwd(), { + securityProvider: new DefaultSecurityProvider(), + permissionPolicy: { + allow: ['bash(echo:*)'], + ask: ['bash(*)'], + defaultDecision: 'ask', + }, +}); + +// 模型选择的工具调用会同时经过提供程序与策略。 +const result = await session.run( + 'Use bash to run exactly: echo screened-by-security-provider', +); +console.log(result.text); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions, PermissionPolicy, DefaultSecurityProvider + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.security_provider = DefaultSecurityProvider() + opts.permission_policy = PermissionPolicy( + allow=["bash(echo:*)"], + ask=["bash(*)"], + default_decision="ask", + ) + + session = agent.session(".", opts) + + # 模型选择的工具调用会同时经过提供程序与策略。 + result = session.run( + "Use bash to run exactly: echo screened-by-security-provider" + ) + print(result.text) + + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + DefaultSecurity: code.Ptr(true), + PermissionPolicy: &code.PermissionPolicy{ + Allow: []string{"bash(echo:*)"}, + Ask: []string{"bash(*)"}, + DefaultDecision: "ask", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + result, err := session.Run( + ctx, + "Use bash to run exactly: echo screened-by-security-provider", + ) + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + +任意自定义 `SecurityProvider` trait 实现仍是 Rust 原生扩展点;四种 SDK 都暴露 +这里使用的内置提供器。 + + + + +## 说明 + +- `defaultDecision` 是所有未被 `allow` / `deny` / `ask` 匹配到的模式的兜底决策(取值为 + `allow`、`deny` 或 `ask` 之一)。真实仓库优先使用 `ask`,仅对自动化确实需要的部分逐步放开。 +- 设置 `enabled: true` 的 `confirmationPolicy` 才会把 `ask` 决策变成会暂停的 + `confirmation_required` 事件。通过 `session.confirmToolUse(toolId, approved, reason?)` + 逐个处理;若在 `defaultTimeoutMs` 内未收到答复,则由 `timeoutAction`(`reject`)决定结果。 +- 除非最终步骤由受控自动化负责,否则 release 和 publish 操作(`bash(git push*)`、 + `bash(npm publish*)`)应保持在 `ask` 或 `deny` 路径上。 +- `session.tool()`、`session.bash()`、`session.git()` 这类宿主直接调用都是特权操作。 + 它们由你的应用代码发起,应在调用 SDK 前先由宿主授权;上面的 permission policy 管控的是 + `send`、`run` 和 `stream` 内由模型选择的工具调用。 + +可运行的确认循环示例位于 +`sdk/node/examples/streaming/hitl_confirmation_loop.ts` 和 +`sdk/python/examples/hitl_confirmation_loop.py`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/skill-tool.mdx b/website/docs/v8.5.1/zh/guide/examples/skill-tool.mdx new file mode 100644 index 00000000..17ccc2dd --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/skill-tool.mdx @@ -0,0 +1,192 @@ +--- +title: '技能工具' +description: '将已注册的技能作为可调用工具来调用' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 技能工具 + +skills 以两个核心工具的形式暴露给模型:`search_skills`(按意图查找 skill)和 +`Skill`(按名称调用 skill)。tool 类型的 skill 会运行其处理器;instruction 类型的 +skill 会返回其正文供模型应用。你既可以让模型在一次运行中调用这些工具,也可以通过 +`session.tool('Skill', ...)` 直接从 SDK 调用某个 skill。 + +通过 `skillDirs` / `skill_dirs` 注册 skill 目录(包含若干 `SKILL.md` 文件的文件夹), +或通过 session options 传入 inline skills。A3S Code 不再内置默认 skills, +`builtinSkills` / `builtin_skills` 只是兼容标记。注册后的 skills 会通过 +`Skill` 与 `search_skills` 可见。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; +use serde_json::json; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_skill_dirs(["./skills"])) + .build() + .await?; + + println!("{:?}", session.tool_names()); + let run = session + .send("搜索可用技能,然后应用最相关的一个。", None) + .await?; + println!("{}", run.text); + + let result = session + .tool( + "Skill", + json!({ + "skill_name": "release-review", + "prompt": "审查这个发布补丁的阻塞项和验证缺口。" + }), + ) + .await?; + if result.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "Skill".into(), + message: result.output, + }); + } + println!("{}", result.output); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { + skillDirs: ['./skills'], // a folder of SKILL.md files +}); + +// Skill 和 search_skills 是核心工具;先确认它们已出现在工具列表中。 +console.log(session.toolNames()); + +// 方式 A:让模型在运行期间搜索并应用技能。 +const run = await session.run( + 'Search available skills, then apply the most relevant one.', +); +console.log(run.text); + +// 方式 B:把技能作为可调用工具直接执行。 +// 标准参数:{ skill_name, prompt? }。 +const result = await session.tool('Skill', { + skill_name: 'release-review', + prompt: 'Review this release patch for blockers and verification gaps.', +}); +if (result.exitCode !== 0) { + throw new Error(result.output); +} +console.log(result.output); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.skill_dirs = ['./skills'] # a folder of SKILL.md files +session = agent.session('.', opts) + +# Skill 和 search_skills 是核心工具;先确认它们已出现在工具列表中。 +print(session.tool_names()) + +# 方式 A:让模型在运行期间搜索并应用技能。 +run = session.run('Search available skills, then apply the most relevant one.') +print(run.text) + +# 方式 B:把技能作为可调用工具直接执行。 +# 标准参数:{ skill_name, prompt? }。 +result = session.tool('Skill', { + 'skill_name': 'release-review', + 'prompt': 'Review this release patch for blockers and verification gaps.', +}) +if result.exit_code != 0: + raise RuntimeError(result.output) +print(result.output) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + SkillDirs: []string{"./skills"}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + names, err := session.ToolNames(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println(names) + run, err := session.Run(ctx, "搜索可用技能,然后应用最相关的一个。") + if err != nil { + log.Fatal(err) + } + fmt.Println(run.Text) + + result, err := session.Tool(ctx, "Skill", map[string]any{ + "skill_name": "release-review", + "prompt": "审查这个发布补丁的阻塞项和验证缺口。", + }) + if err != nil { + log.Fatal(err) + } + if result.ExitCode != 0 { + log.Fatal(result.Output) + } + fmt.Println(result.Output) +} +``` + + + + +`SKILL.md` 在 frontmatter 中声明其 `kind`(`tool`、`instruction` 或 `agent`)。对于 +tool 类型的 skill,`Skill` 工具会运行该 skill 的处理器并返回其输出;对于 instruction +类型的 skill,则返回正文供模型应用。skill 管理由 SDK 注册、skill 目录或项目文件完成, +而不是通过模型可见的管理工具处理。 + +可运行版本位于 `sdk/node/examples/skills/test_custom_skills_agents.ts`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/skills.mdx b/website/docs/v8.5.1/zh/guide/examples/skills.mdx new file mode 100644 index 00000000..4d26a119 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/skills.mdx @@ -0,0 +1,180 @@ +--- +title: '技能与自定义智能体' +description: '从文件系统约定加载项目技能和自定义子智能体。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 技能与自定义智能体 + +A3S Code 不再内置默认 skills。请从磁盘上的目录加载自己的技能和子智能体来扩展会话。 +使用 `skillDirs` 加载 Markdown skills,使用 `agentDirs` 加载 worker/subagent 定义。 +`registerAgentDir` 可以在 session 创建后继续追加 agent 定义目录;skill 目录在 +session 创建时加载。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; +use std::path::Path; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("/path/to/project") + .options( + SessionOptions::new() + .with_skill_dirs(["./.a3s/skills"]) + .with_agent_dir("./.a3s/agents"), + ) + .build() + .await?; + + session.register_agent_dir(Path::new("./team/shared-agents"))?; + println!("工具:{:?}", session.tool_names()); + println!("技能:{:?}", session.skill_names()); + + let result = session + .send("使用项目约定技能搭建一个新模块。", None) + .await?; + println!("{}", result.text); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); + +// 显式加载项目 skills 和 agents。 +const session = agent.session('/path/to/project', { + skillDirs: ['./.a3s/skills'], + agentDirs: ['./.a3s/agents'], +}); + +// 会话创建后可以继续注册智能体定义目录。 +session.registerAgentDir('./team/shared-agents'); + +// 查看会话已加载的内容。 +console.log('Tools:', session.toolNames()); +console.log('Commands:', session.listCommands()); + +// 智能体现在可以使用项目技能。 +const result = await session.run( + 'Use the project conventions skill to scaffold a new module.', +); +console.log(result.text); + +session.close(); +``` + + + + +```python +from a3s_code import Agent, SessionOptions + +agent = Agent.create("agent.acl") + +# 显式加载项目 skills 和 agents。 +opts = SessionOptions() +opts.skill_dirs = ['./.a3s/skills'] +opts.agent_dirs = ['./.a3s/agents'] + +session = agent.session('/path/to/project', opts) + +# 会话创建后可以继续注册智能体定义目录。 +session.register_agent_dir('./team/shared-agents') + +# 查看会话已加载的内容。 +print('Tools:', session.tool_names()) +print('Commands:', session.list_commands()) + +# 智能体现在可以使用项目技能。 +result = session.run( + 'Use the project conventions skill to scaffold a new module.', +) +print(result.text) + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, "/path/to/project", &code.SessionOptions{ + SkillDirs: []string{"./.a3s/skills"}, + AgentDirs: []string{"./.a3s/agents"}, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + if _, err := session.RegisterAgentDir(ctx, "./team/shared-agents"); err != nil { + log.Fatal(err) + } + tools, err := session.ToolNames(ctx) + if err != nil { + log.Fatal(err) + } + skills, err := session.SkillNames(ctx) + if err != nil { + log.Fatal(err) + } + fmt.Println("工具:", tools) + fmt.Println("技能:", skills) + + result, err := session.Run(ctx, "使用项目约定技能搭建一个新模块。") + if err != nil { + log.Fatal(err) + } + fmt.Println(result.Text) +} +``` + + + + +## 技能注册表行为 + +A3S Code 不再内置默认技能。默认有效技能注册表为空。 +`builtinSkills: true` / `builtin_skills = True` 会被接受以保持兼容,但当前不会 +添加任何默认 skill。请通过 `skillDirs` / `skill_dirs`、inline skills 或显式 +`SkillRegistry` 加载需要的技能。 + +日常项目里,把可复用行为放进 `.a3s/skills` 或配置的 skill 目录。 + +从 `agentDirs` 加载的自定义子智能体可以在 +[`session.parallel(...)`](/guide/examples/orchestration) 和 +[`session.pipeline(...)`](/guide/examples/orchestration) 中按名称引用, +与内置注册表中的智能体(`explore`、`plan`、`general`、`verification`、`review`)一起使用。 + +可运行版本位于 `sdk/node/examples/skills/test_custom_skills_agents.ts`。 diff --git a/website/docs/v8.5.1/zh/guide/examples/streaming.mdx b/website/docs/v8.5.1/zh/guide/examples/streaming.mdx new file mode 100644 index 00000000..cde8729b --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/streaming.mdx @@ -0,0 +1,205 @@ +--- +title: '流式输出' +description: '在回合运行时读取增量 AgentEvent 事件' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 流式输出 + +`session.stream(prompt)` 会在回合运行过程中逐步产出事件,因此你可以在文本到达时即时渲染,并实时响应工具活动。当你需要实时 UI,或希望 CLI 逐 token 打印输出(而不是等待 `send` 或 `run` 返回完整结果)时,请使用它。 + +每个事件都带有稳定的 `version`、`type`、`payload` 和可选 `metadata` 信封字段。常见类型包括 `agent_start`、`text_delta` / `reasoning_delta`、`tool_start` / `tool_end`、`agent_end` 和 `error`。验证信息位于 `agent_end` 的 payload 与便捷字段中。 + + + + +```rust +use a3s_code_core::{Agent, AgentEvent, CodeError, PlanningMode, SessionOptions}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder(".") + .options(SessionOptions::new().with_planning_mode(PlanningMode::Disabled)) + .build() + .await?; + + let (mut events, lifecycle) = session + .stream("使用 bash 工具运行测试,然后总结结果。", None) + .await?; + + while let Some(event) = events.recv().await { + match event { + AgentEvent::TextDelta { text } => print!("{text}"), + AgentEvent::ToolStart { name, .. } => println!("\n[工具开始] {name}"), + AgentEvent::ToolEnd { + name, exit_code, .. + } => println!("\n[工具结束] {name},退出码={exit_code}"), + AgentEvent::End { + verification_summary, + .. + } => println!("\n[验证] {verification_summary:?}"), + AgentEvent::Error { message } => return Err(CodeError::Llm(message)), + _ => {} + } + } + lifecycle + .await + .map_err(|error| CodeError::Internal(error.into()))??; + println!("\n[流式输出完成]"); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session(process.cwd(), { planningMode: 'disabled' }); + +const stream = await session.stream( + 'Use the bash tool to run the tests, then summarize the result.', +); + +while (true) { + const next = await stream.next(); + if (next.done || !next.value) break; + + const event = next.value; + if (event.type === 'text_delta' && event.text) { + process.stdout.write(event.text); + } else if (event.type === 'tool_start') { + console.log(`\n[tool:start] ${event.toolName ?? 'unknown'}`); + } else if (event.type === 'tool_end') { + console.log( + `\n[tool:end] ${event.toolName ?? 'unknown'} exit=${event.exitCode ?? 0}`, + ); + } else if (event.type === 'agent_end') { + console.log(`\n[verification] ${event.verificationSummaryText ?? ''}`); + } else if (event.type === 'error') { + throw new Error(event.error ?? 'stream error'); + } +} + +console.log('\n[stream] complete'); +session.close(); +``` + + + + +```python +import os + +from a3s_code import Agent, SessionOptions + + +def main() -> None: + agent = Agent.create("agent.acl") + + opts = SessionOptions() + opts.planning_mode = "disabled" + session = agent.session(".", opts) + + prompt = "Use the bash tool to run the tests, then summarize the result." + + try: + for event in session.stream(prompt): + if event.type == "text_delta" and event.text: + print(event.text, end="", flush=True) + elif event.type == "tool_start": + print(f"\n[tool:start] {event.tool_name or 'unknown'}") + elif event.type == "tool_end": + print(f"\n[tool:end] {event.tool_name or 'unknown'} exit={event.exit_code or 0}") + elif event.type == "agent_end": + print(f"\n[verification] {event.verification_summary_text or ''}") + elif event.type == "error": + raise RuntimeError(event.error or "stream error") + print("\n[stream] complete") + finally: + session.close() + + +if __name__ == "__main__": + main() +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + + session, err := agent.Session(ctx, ".", &code.SessionOptions{ + PlanningMode: code.PlanningDisabled, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + stream, err := session.Stream( + ctx, + "使用 bash 工具运行测试,然后总结结果。", + nil, + ) + if err != nil { + log.Fatal(err) + } + for event := range stream.Events { + switch event.Type { + case code.EventTextDelta: + var payload struct { + Text string `json:"text"` + } + if err := event.DecodePayload(&payload); err != nil { + log.Fatal(err) + } + fmt.Print(payload.Text) + default: + // 按需把 event.Payload 与 event.Metadata 转发给界面。 + } + } + if err := <-stream.Done; err != nil { + log.Fatal(err) + } +} +``` + + + + +说明: + +- Rust 从 Tokio channel 接收 `AgentEvent`;Node.js 使用 `stream.next()`,Python 使用同步迭代器;Go 必须先持续读取 `stream.Events`,再从 `stream.Done` 读取最终错误。取消 Go Context 也会请求原生 Session 取消当前 Run。 +- 四种 SDK 都公开规范事件类型。Node.js 和 Python 提供便捷投影;Go 将 `Payload` 与 `Metadata` 保留为 `json.RawMessage`,并提供 `DecodePayload`。未来未知事件类型在四种 SDK 中都不会丢失。 +- 当启用确认策略时,流式事件还可能包含人工介入(human-in-the-loop)确认信号(`confirmation_required`、`confirmation_received`、`confirmation_timeout`)。 + +可运行的流式示例位于 `sdk/node/examples/streaming/`。完整的人工确认 +循环在 `sdk/node/examples/streaming/hitl_confirmation_loop.ts`, +对应 Python 版本位于 `sdk/python/examples/`,Go 流式行为由 +`sdk/go/session_test.go` 覆盖。 diff --git a/website/docs/v8.5.1/zh/guide/examples/structured-output.mdx b/website/docs/v8.5.1/zh/guide/examples/structured-output.mdx new file mode 100644 index 00000000..b00ac49c --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/examples/structured-output.mdx @@ -0,0 +1,588 @@ +--- +title: '结构化输出' +description: '使用 generate_object 工具生成通过 JSON 模式校验的对象。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 结构化输出 + +内置的 `generate_object` 工具会让配置的 LLM 生成 JSON 对象,对响应执行你提供的 +JSON Schema 校验,并且只在零退出码结果中返回校验后的对象。当你需要机器可读的结果时 +使用它:抽取、分类、配置生成,或者作为另一个程序的输入。 + +你可以通过 `session.tool('generate_object', ...)` 直接调用它。工具结果会把校验后的对象以 JSON 形式放在 `result.output` 上——解析它并读取 `object` 字段。同一个工具也支持 agent 自主调用,即模型在 `send` 过程中自行决定调用它。 + +## 直接工具调用 + +最简单的方式:直接调用 `generate_object`,先检查工具退出码,再从结果中解析出校验后的对象。 + + + + +```rust +use a3s_code_core::Agent; +use serde_json::{json, Value}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let session = agent.session_builder(".").build().await?; + + let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["name", "age", "skills"], + "properties": { + "name": { "type": "string" }, + "age": { "type": "integer", "minimum": 0 }, + "skills": { + "type": "array", + "items": { "type": "string" }, + "minItems": 1 + } + } + }, + "prompt": "提取:Alice 今年 28 岁,擅长 Rust、TypeScript 和 Python。", + "schema_name": "developer", + "mode": "tool" + }), + ) + .await?; + if result.exit_code != 0 { + return Err(a3s_code_core::CodeError::Tool { + tool: "generate_object".into(), + message: result.output, + }); + } + + let value: Value = serde_json::from_str(&result.output)?; + println!("{}", value["object"]); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('.'); + +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['name', 'age', 'skills'], + properties: { + name: { type: 'string' }, + age: { type: 'integer', minimum: 0 }, + skills: { + type: 'array', + items: { type: 'string' }, + minItems: 1, + }, + }, + }, + prompt: 'Extract: "Alice is 28, skilled in Rust, TypeScript, and Python."', + schema_name: 'developer', + mode: 'tool', +}); + +if (result.exitCode !== 0) { + throw new Error(result.output); +} + +const { object } = JSON.parse(result.output); +console.log(object); +// { name: "Alice", age: 28, skills: ["Rust", "TypeScript", "Python"] } + +session.close(); +``` + + + + +```python +import json +from a3s_code import Agent + +agent = Agent.create('agent.acl') +session = agent.session('.') + +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["name", "age", "skills"], + "properties": { + "name": {"type": "string"}, + "age": {"type": "integer", "minimum": 0}, + "skills": { + "type": "array", + "items": {"type": "string"}, + "minItems": 1, + }, + }, + }, + "prompt": 'Extract: "Alice is 28, skilled in Rust, TypeScript, and Python."', + "schema_name": "developer", + "mode": "tool", +}) + +if result.exit_code != 0: + raise RuntimeError(result.output) + +obj = json.loads(result.output)["object"] +print(obj) +# {"name": "Alice", "age": 28, "skills": ["Rust", "TypeScript", "Python"]} + +session.close() +``` + + + + +```go +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(context.Background()) + session, err := agent.Session(ctx, ".", nil) + if err != nil { + log.Fatal(err) + } + defer session.Close(context.Background()) + + result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"name", "age", "skills"}, + "properties": map[string]any{ + "name": map[string]any{"type": "string"}, + "age": map[string]any{"type": "integer", "minimum": 0}, + "skills": map[string]any{ + "type": "array", "items": map[string]any{"type": "string"}, + "minItems": 1, + }, + }, + }, + "prompt": "提取:Alice 今年 28 岁,擅长 Rust、TypeScript 和 Python。", + "schema_name": "developer", + "mode": "tool", + }) + if err != nil { + log.Fatal(err) + } + if result.ExitCode != 0 { + log.Fatal(result.Output) + } + var value struct { + Object map[string]any `json:"object"` + } + if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + log.Fatal(err) + } + fmt.Println(value.Object) +} +``` + + + + +校验后的值位于解析输出的 `object` 键上。当 `result.exitCode`(Node)/ +`result.exit_code`(Python)为零时,`required` 中声明的字段已经通过运行时校验。 +如果模型在修复重试后仍无法满足 schema,工具会报告非零退出码。 + +## 枚举分类 + +用 `enum` 把字段约束到一个固定集合。这会把自由文本分类变成带 schema 闸门的结果。 + + + + +```rust +use serde_json::{json, Value}; + +let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["sentiment", "confidence"], + "properties": { + "sentiment": { + "type": "string", + "enum": ["positive", "negative", "neutral"] + }, + "confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + } + } + }, + "prompt": "判断情感:这是我用过最差的产品。", + "schema_name": "sentiment" + }), + ) + .await?; +let value: Value = serde_json::from_str(&result.output)?; +println!( + "{} {}", + value["object"]["sentiment"], + value["object"]["confidence"] +); +``` + + + + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['sentiment', 'confidence'], + properties: { + sentiment: { type: 'string', enum: ['positive', 'negative', 'neutral'] }, + confidence: { type: 'number', minimum: 0, maximum: 1 }, + }, + }, + prompt: 'Classify sentiment: "This is the worst product I have ever used."', + schema_name: 'sentiment', +}); + +const { object } = JSON.parse(result.output); +console.log(object.sentiment, object.confidence); // "negative" 0.97 +``` + + + + +```python +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["sentiment", "confidence"], + "properties": { + "sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + }, + }, + "prompt": 'Classify sentiment: "This is the worst product I have ever used."', + "schema_name": "sentiment", +}) + +obj = json.loads(result.output)["object"] +print(obj["sentiment"], obj["confidence"]) # "negative" 0.97 +``` + + + + +```go +result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"sentiment", "confidence"}, + "properties": map[string]any{ + "sentiment": map[string]any{ + "type": "string", "enum": []string{"positive", "negative", "neutral"}, + }, + "confidence": map[string]any{ + "type": "number", "minimum": 0, "maximum": 1, + }, + }, + }, + "prompt": "判断情感:这是我用过最差的产品。", + "schema_name": "sentiment", +}) +if err != nil { + return err +} +var value struct { + Object map[string]any `json:"object"` +} +if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + return err +} +fmt.Println(value.Object["sentiment"], value.Object["confidence"]) +``` + + + + +## 嵌套模式与数组 + +Schema 可以任意深度地嵌套对象和数组,运行时会校验整个结构。这能在一次调用中建模真实的配置文件、清单或 API 载荷。 + + + + +```rust +use serde_json::{json, Value}; + +let result = session + .tool( + "generate_object", + json!({ + "schema": { + "type": "object", + "required": ["items"], + "properties": { + "items": { + "type": "array", + "minItems": 3, + "maxItems": 5, + "items": { + "type": "object", + "required": ["name", "category"], + "properties": { + "name": { "type": "string" }, + "category": { + "type": "string", + "enum": ["fruit", "vegetable", "grain"] + } + } + } + } + } + }, + "prompt": "列出 3 种食物及其类别。", + "schema_name": "food_list" + }), + ) + .await?; +let value: Value = serde_json::from_str(&result.output)?; +let items = value["object"]["items"].as_array().unwrap(); +println!("{} {:?}", items.len(), items); +``` + + + + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['items'], + properties: { + items: { + type: 'array', + minItems: 3, + maxItems: 5, + items: { + type: 'object', + required: ['name', 'category'], + properties: { + name: { type: 'string' }, + category: { type: 'string', enum: ['fruit', 'vegetable', 'grain'] }, + }, + }, + }, + }, + }, + prompt: 'List 3 food items with their categories.', + schema_name: 'food_list', +}); + +const { items } = JSON.parse(result.output).object; +console.log( + items.length, + items.map((i) => i.name), +); +``` + + + + +```python +result = session.tool("generate_object", { + "schema": { + "type": "object", + "required": ["items"], + "properties": { + "items": { + "type": "array", + "minItems": 3, + "maxItems": 5, + "items": { + "type": "object", + "required": ["name", "category"], + "properties": { + "name": {"type": "string"}, + "category": {"type": "string", "enum": ["fruit", "vegetable", "grain"]}, + }, + }, + }, + }, + }, + "prompt": "List 3 food items with their categories.", + "schema_name": "food_list", +}) + +items = json.loads(result.output)["object"]["items"] +print(len(items), [i["name"] for i in items]) +``` + + + + +```go +result, err := session.Tool(ctx, "generate_object", map[string]any{ + "schema": map[string]any{ + "type": "object", + "required": []string{"items"}, + "properties": map[string]any{ + "items": map[string]any{ + "type": "array", "minItems": 3, "maxItems": 5, + "items": map[string]any{ + "type": "object", + "required": []string{"name", "category"}, + "properties": map[string]any{ + "name": map[string]any{"type": "string"}, + "category": map[string]any{ + "type": "string", + "enum": []string{"fruit", "vegetable", "grain"}, + }, + }, + }, + }, + }, + }, + "prompt": "列出 3 种食物及其类别。", + "schema_name": "food_list", +}) +if err != nil { + return err +} +var value struct { + Object struct { + Items []map[string]any `json:"items"` + } `json:"object"` +} +if err := json.Unmarshal([]byte(result.Output), &value); err != nil { + return err +} +fmt.Println(len(value.Object.Items), value.Object.Items) +``` + + + + +## 智能体自主调用 + +你也可以让 agent 自行决定何时使用结构化输出。让它在 `send` 过程中调用 `generate_object`;它会先收集上下文,再输出对象。 + + + + +```rust +let result = session + .send( + "使用 generate_object 从下面内容提取电影标题、年份和类型:\ + 《盗梦空间》于 2010 年上映,是一部科幻惊悚片。", + None, + ) + .await?; + +println!( + "工具调用:{},令牌:{}", + result.tool_calls_count, + result.usage.total_tokens +); +``` + + + + +```ts +const result = await session.send( + 'Use the generate_object tool to extract the following into an object ' + + 'with fields "title" (string), "year" (integer), "genre" (string): ' + + 'The movie "Inception" was released in 2010 and is a sci-fi thriller.', +); + +console.log( + `tool calls: ${result.toolCallsCount}, tokens: ${result.totalTokens}`, +); +``` + + + + +```python +result = session.send( + 'Use the generate_object tool to produce a JSON object with schema ' + '{"type":"object","required":["language","paradigm"],"properties":' + '{"language":{"type":"string"},"paradigm":{"type":"string"}}} ' + 'for: "Rust is a systems programming language with a focus on safety."' +) + +print(f"tool calls: {result.tool_calls_count}, tokens: {result.total_tokens}") +``` + + + + +```go +result, err := session.Run( + ctx, + "使用 generate_object 从下面内容提取电影标题、年份和类型:"+ + "《盗梦空间》于 2010 年上映,是一部科幻惊悚片。", +) +if err != nil { + return err +} +fmt.Printf( + "工具调用:%d,令牌:%d\n", + result.ToolCallsCount, + result.Usage.TotalTokens, +) +``` + + + + +## 模式校验覆盖 + +内置校验器支持: + +- `type`(包括 nullable 数组如 `["string", "null"]`) +- `required`、`properties`、`additionalProperties` +- `enum`、`const` +- `anyOf`、`oneOf` +- `minLength`、`maxLength`、`pattern` +- `minimum`、`maximum`、`exclusiveMinimum`、`exclusiveMaximum` +- `minItems`、`maxItems`、`items` +- 嵌套对象和数组校验 + +## 说明 + +- 校验后的值位于解析后 `result.output` 的 `object` 键上。当前跨 provider 默认路径传 `mode: 'tool'`,仅 prompt 的回退方式传 `mode: 'prompt'`。`auto`、`strict` 和 `json` 在当前运行路径中都会解析为 `tool`。 +- 把你依赖的每个字段都列入 `required`——运行时会强制执行,因此缺失或类型错误的字段会导致校验失败,而不是悄悄返回部分数据。 +- `generate_object` 是独立注册的内置工具,不依赖 built-in skills。 +- 直接 `session.tool(...)` 调用是宿主控制面调用。允许模型在 `send` / `run` / `stream` 中自行选择工具时使用 `permissionPolicy`;直接 SDK 调用前应使用宿主自己的授权逻辑。 + +可运行版本随源码提供,位于 `sdk/node/examples/basic/test_generate_object.ts` 和 `sdk/python/examples/test_generate_object.py`。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-agents.mdx b/website/docs/v8.5.1/zh/guide/filesystem-agents.mdx new file mode 100644 index 00000000..fc577c56 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-agents.mdx @@ -0,0 +1,74 @@ +--- +title: 'agents/ 角色目录' +description: '用 .a3s/agents 定义可被 task 和自动委派调用的工作智能体。' +--- + +# agents/ 角色目录 + +`agents/` 目录存放 worker/subagent 定义。推荐使用 `.a3s/agents/` 作为 A3S 原生位置;迁移项目可以继续读取 `.claude/agents/`,但新文档和新项目应优先使用 `.a3s/agents/`。 + +```text +repo/ +└── .a3s/ + └── agents/ + ├── explorer.md + ├── security-reviewer.md + └── verification-runner.md +``` + +这些文件不是主 Agent 目录。模型可见的 `task` 工具、宿主侧 +`session.task(...)` 与 `session.tasks(...)`,以及 `autoDelegation` 都可以调用它们; +父 session 仍负责最终汇总、验证和权限边界。 + +## 智能体文件格式 + +```md +--- +name: security-reviewer +description: Use for permission, secret, and external side-effect review +tools: Read, Search, Bash(rg *) +disallowedTools: + - Write + - Bash(git push *) +--- + +Review security risks first. Return blockers, evidence paths, and required verification. +``` + +`name` 是调用名,`description` 决定自动委派时是否匹配,正文是该 worker 的角色说明。工具字段收窄 worker 可见能力;不要依赖 worker 自己“承诺不做危险事”。 + +## 手动委派 + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + maxParallelTasks: 4, +}); + +await session.task({ + agent: 'security-reviewer', + description: 'Review release side effects', + prompt: 'Check changed auth, permission, and external API paths.', +}); +``` + +固定流程更适合手动委派或可编程编排;自动委派适合“父 Agent 读到目标后自行选择专家”的场景。 + +## 自动委派 + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, +}); +``` + +自动委派依赖 agent 描述和置信度评分。写 description 时要说清楚“何时使用”,而不是只写角色口号。 + +## 最佳实践 + +- 一个文件只做一个角色,避免“万能 reviewer”。 +- description 面向路由,正文面向执行。 +- worker 输出应包含证据、风险和建议下一步,方便父 session 汇总。 +- 高权限动作留给父 session 或显式工具,不要让子 Agent 默认获得发布、删除、推送权限。 +- 需要动态创建的一次性 worker,用 `workerAgents` 或 `registerWorkerAgent()`,不用落盘。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-config.mdx b/website/docs/v8.5.1/zh/guide/filesystem-config.mdx new file mode 100644 index 00000000..3cdbf031 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-config.mdx @@ -0,0 +1,92 @@ +--- +title: 'agent.acl' +description: '模型、服务提供商、队列、技能目录和工作智能体目录的文件化运行配置。' +--- + +# agent.acl + +`agent.acl` 是文件系统优先形态里的运行配置入口。它负责把模型、provider、并行度、队列、存储、skill 目录和 worker agent 目录固化为可版本化配置。 + +SDK 宿主可以用 `Agent.create("agent.acl")` 显式传入任意 `.acl` 文件。 +`a3s code` TUI 会从 workspace 向上发现 `.a3s/config.acl`,然后读取 +`~/.a3s/config.acl`;它并不要求仓库根部存在 `agent.acl`。AgentDir 内的 +`agent.acl` 服务于该目录的长期 Agent。三者格式一致,但发现方式和作用范围不同。 + +## 基础配置 + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} +``` + +`apiKey` / `api_key` 和 `baseUrl` / `base_url` 都是可接受别名。运行时不硬编码 +模型名;`default_model` 和 session 级 `model` 覆盖必须匹配这里声明的 +`provider/model-id`。 + +把 token 通过环境变量注入,不要写进 `agent.acl`。配置文件进入仓库后,它就是产品行为的一部分,应该像代码一样审查。 + +## 目录发现 + +```acl +skill_dirs = ["./.a3s/skills"] +agent_dirs = ["./.a3s/agents"] +project_doc_max_bytes = 32768 +project_doc_fallback_filenames = ["TEAM_GUIDE.md"] + +auto_delegation { + enabled = true + min_confidence = 0.72 + max_tasks = 4 + auto_parallel = false +} +``` + +`skill_dirs` 指向可复用技能目录;`agent_dirs` 指向 worker/subagent 定义目录。`project_doc_max_bytes` 限制从仓库根到 workspace 的项目指令链总大小;后备文件名会在 `AGENTS.override.md` 和 `AGENTS.md` 之后查找。自动委派只决定是否让模型选择 worker,不会取消父 session 的权限策略、工具可见性或验证要求。 + +## 会话存储 + +```acl +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +当 session 没有显式收到 SDK `sessionStore` 时,`sessions_dir` 是本地文件型 +session persistence 路径。`storage_backend = "memory"` 表示 session 是临时的。 +`storage_url` 会被解析为自定义存储元数据,但它本身不会创建本地 +`FileSessionStore`。 + +## 智能体目录中的配置 + +AgentDir 的 `agent.acl` 可以省略;省略时使用默认配置。存在时,`AgentDir::load` 会把它解析成 `CodeConfig`,并与 `instructions.md`、`skills/`、`tools/`、`schedules/` 一起合成长期 Agent。 + +```text +release-agent/ +├── instructions.md +├── agent.acl +├── skills/ +├── tools/ +└── schedules/ +``` + +适合放在 AgentDir `agent.acl` 的内容包括该 Agent 默认模型、provider、运行限制、队列策略和私有 skill 目录。不要在这里硬编码部署环境差异;用环境变量或宿主注入区分开发、测试和生产。 + +## 配置边界 + +- 配置决定“可以连接什么”和“默认如何运行”,不决定“模型能越过权限门”。 +- 目录路径应相对 workspace 或 AgentDir,避免依赖个人机器绝对路径。 +- 自动委派需要和 `agents/` 的描述质量一起调优;低质量 description 会让运行时错误分派。 +- 高风险工具即使被目录发现,也应继续走 HITL 或 allow-list。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-first.mdx b/website/docs/v8.5.1/zh/guide/filesystem-first.mdx new file mode 100644 index 00000000..ae2a04cf --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-first.mdx @@ -0,0 +1,68 @@ +--- +title: '文件系统优先' +description: '把角色、配置、技能、工具、调度和团队定义沉淀为可审查、可版本化的文件系统约定。' +--- + +# 文件系统优先 + +A3S Code 把文件系统当作 Agent 产品的第一层接口:能长期保存的角色、工具、技能、调度和团队定义都先落成文件,再由运行时按约定加载。这样做的目标不是减少配置项,而是把“Agent 如何工作”变成可以审查、版本化、复用和迁移的工程资产。 + +文件系统优先有两种常用形态: + +```text +repo/ +├── AGENTS.md # 项目长期指令 +├── agent.acl # 模型、provider、队列与委派配置 +└── .a3s/ + ├── agents/ # 可被 task / autoDelegation 调用的 worker agents + └── skills/ # 可复用技能 + +release-agent/ +├── instructions.md # 独立长期 Agent 的角色 slot +├── agent.acl # 该 Agent 的运行配置 +├── skills/ # 该 Agent 私有技能 +├── tools/ # 该 Agent 可声明的 MCP / script tools +└── schedules/ # 周期性 turn +``` + +第一种是仓库工作区约定,适合交互式开发、团队委派和项目知识注入。第二种是 AgentDir 约定,适合长期运行、定时触发和目录级工具声明。 + +## 路径总览 + +| 路径 | 作用 | 什么时候用 | +| ------------------------- | -------------------------------------------- | ---------------------------------------------------------- | +| `AGENTS.md` | 给当前 workspace 注入稳定项目指令 | 代码风格、验证命令、安全边界和发布流程需要长期生效。 | +| `instructions.md` | 定义一个 AgentDir 主 Agent 的角色 slot | 需要把单个目录加载成可长期运行的 Agent。 | +| `agent.acl` | 配置模型、provider、队列、技能目录和委派策略 | 运行时策略需要随仓库或 Agent 一起版本化。 | +| `.a3s/agents/` | 存放 worker/subagent 定义 | 父 agent 需要 `task` 或自动委派。 | +| `.a3s/skills/`、`skills/` | 存放可复用技能 | 多个任务共享一组操作准则、检查清单或领域流程。 | +| `tools/` | 声明 AgentDir 的 MCP 或 script tools | 需要把连接器或受限脚本作为模型可见能力暴露给调度 session。 | +| `schedules/` | 声明周期性 turn | 需要日报、巡检、同步、回归检查等定时 Agent。 | + +这些约定不是新的 prompt 系统。`AGENTS.md`、`instructions.md` 和 skills 都会进入 A3S Code 的上下文组合流程;工具可见性、权限门、HITL、响应契约和验证仍由 harness 控制。 + +## 加载顺序 + +一次典型 session 会先解析 `agent.acl`,再绑定 workspace,随后加载项目指令、skills、agent definitions、direct tools、MCP 连接和运行时策略。AgentDir 的 `serve_agent_dir` 会先用 `instructions.md`、本地 `agent.acl`、`skills/`、`tools/` 和 `schedules/` 合成配置,再为每个 schedule 创建独立 session。 + +路径越靠近具体 Agent,语义越局部:仓库根部的 `AGENTS.md` 描述整个项目,`.a3s/agents/*.md` 描述某个 worker,AgentDir 内的 `instructions.md` 描述该目录的主 Agent。不要把全局规则复制到每个文件里,除非确实需要覆盖上下文边界。 + +## 设计原则 + +- 约定只负责发现和组装,不负责绕过安全。高权限工具仍需要权限策略和确认门。 +- 文件应可被 code review。模型、provider、工具、调度和角色变更都应该能在 diff 中看清楚。 +- secrets 不进入仓库。用环境变量、宿主连接或密钥管理系统注入。 +- 一次性实验可以用 SDK 参数;需要复用、审计或迁移时再固化成文件。 +- `AgentDir` 是主 Agent 目录;`.a3s/agents/` 是 worker/subagent 定义目录,两者不要混用。 + +## 阅读顺序 + +1. [约定大于配置](/guide/convention-over-configuration) +2. [AGENTS.md](/guide/agents-md) +3. [instructions.md](/guide/filesystem-instructions) +4. [agent.acl](/guide/filesystem-config) +5. [Agent 目录](/guide/agent-dir) +6. [agents/ 角色目录](/guide/filesystem-agents) +7. [skills/ 技能目录](/guide/filesystem-skills) +8. [tools/ 工具目录](/guide/filesystem-tools) +9. [schedules/ 调度目录](/guide/filesystem-schedules) diff --git a/website/docs/v8.5.1/zh/guide/filesystem-instructions.mdx b/website/docs/v8.5.1/zh/guide/filesystem-instructions.mdx new file mode 100644 index 00000000..af558d0e --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-instructions.mdx @@ -0,0 +1,50 @@ +--- +title: 'instructions.md' +description: '智能体目录中主智能体的角色插槽,以及它和 AGENTS.md 的边界。' +--- + +# instructions.md + +`instructions.md` 是 AgentDir 主 Agent 的角色文件。它是唯一必需文件,内容会作为 `SystemPromptSlots.role` 注入到每个调度 session 中。 + +它和 `AGENTS.md` 的边界不同: + +| 文件 | 范围 | 适合内容 | +| ----------------- | ---------------------- | -------------------------------------------- | +| `AGENTS.md` | workspace 或子目录 | 项目规则、验证命令、代码风格、安全要求。 | +| `instructions.md` | 一个 AgentDir 主 Agent | 角色身份、工作目标、输出偏好、长期任务边界。 | + +`instructions.md` 是纯 Markdown,不需要 frontmatter。它不会覆盖 harness 的 `BOUNDARIES`、响应格式、工具权限或验证要求。 + +## 推荐写法 + +```md +You are a release-readiness agent for this repository. + +Responsibilities: + +- Track release blockers and risky changes. +- Separate shipped changes from follow-up work. +- Never invent CI status, versions, or owners. + +Output: + +- Start with blockers. +- Include evidence paths. +- End with required verification commands. +``` + +保持短、稳定、可审查。把项目级命令放进 `AGENTS.md`,把可复用流程放进 `skills/`,把周期触发目标放进 `schedules/*.md`。`instructions.md` 只回答“这个长期 Agent 是谁,以及它默认如何工作”。 + +## 会被哪里使用 + +`serve_agent_dir` 会在启动时读取 `instructions.md`,并把它应用到每个已启用 schedule 的 session。配合 `SessionStore` 恢复历史时,历史上下文来自 store,但当前的 `instructions.md` 会重新加载,因此修改角色文件会在下一次重启后生效。 + +交互式 session 如果只需要临时角色,可以直接使用 SDK 的 prompt slots;当角色需要随仓库保存、接受 review 或被多个调度复用时,再把它固化到 `instructions.md`。 + +## 不要放什么 + +- 不要放密钥、token、私有 endpoint 凭据。 +- 不要写“忽略安全规则”或“自动批准所有高风险动作”。 +- 不要复制大型项目手册;用链接和 skills 组织细节。 +- 不要把 schedule prompt 写在这里;每个周期性任务应放在 `schedules/*.md`。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-schedules.mdx b/website/docs/v8.5.1/zh/guide/filesystem-schedules.mdx new file mode 100644 index 00000000..a95ac4f6 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-schedules.mdx @@ -0,0 +1,74 @@ +--- +title: 'schedules/ 调度目录' +description: '用 Markdown 调度文件声明定时回合、独立会话与可恢复上下文。' +--- + +# schedules/ 调度目录 + +`schedules/` 为 AgentDir 声明周期性 turn。每个 Markdown 文件是一条 schedule:frontmatter 描述 cron 元数据,正文是每次触发时发送给 Agent 的 prompt。 + +```text +release-agent/ +└── schedules/ + ├── daily.md + └── weekly-risk-review.md +``` + +## 调度文件 + +```md +--- +cron: '0 9 * * *' +name: daily-release-check +enabled: true +--- + +Summarize merged changes since the last run, inspect release risks, +and report only blockers plus required verification. +``` + +| 字段 | 必填 | 默认值 | 说明 | +| --------- | ---- | ----------- | ----------------------------------- | +| `cron` | 是 | - | 5 字段或 6 字段 cron 表达式。 | +| `name` | 否 | 文件名 stem | 调度名称,也是 session id 后缀。 | +| `enabled` | 否 | `true` | 设为 `false` 可保留文件但暂停调度。 | + +5 字段 cron 会被归一化为 6 字段,前面补 `0` 秒。时间按 UTC 评估。 + +## 独立会话 + +每条 schedule 使用稳定 session id:`schedule:`。同一条 schedule 的多次触发会累积上下文;不同 schedule 彼此隔离。每个 session 都会注入 AgentDir 的 `instructions.md`、`agent.acl`、`skills/` 和 `tools/`。 + +这意味着日报和周报可以共享同一个 AgentDir,但不共享会话历史。需要共享状态时,应使用外部存储、A3S Memory 或显式工具。 + +## `serve` 守护进程 + +```rust +use a3s_code_core::config::AgentDir; +use a3s_code_core::serve::serve_agent_dir; +use a3s_code_core::Agent; +use tokio_util::sync::CancellationToken; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let agent_dir = AgentDir::load("./release-agent")?; + let agent = Agent::from_config(agent_dir.config.clone()).await?; + let cancel = CancellationToken::new(); + + serve_agent_dir(&agent, &agent_dir, "./workspace", None, cancel).await?; + Ok(()) +} +``` + +`serve_agent_dir` 会运行所有 enabled schedules,直到 cancellation token 触发。优雅取消会让进行中的 turn 完成当前迭代后停止。 + +## 恢复与限制 + +传入 `SessionStore` 后,守护进程重启会恢复已有 `schedule:` session 的历史上下文。当前目录里的 `instructions.md`、`skills/` 和 `tools/` 会在每次启动时重新应用。 + +需要注意: + +- 恢复只恢复对话历史,不追补停机期间错过的触发。 +- store 目录是信任边界,它决定恢复后的历史和 workspace。 +- 无效 cron 会在构建调度器时失败,并指出出问题的 schedule。 +- 没有 enabled schedules 的 AgentDir 会立即返回。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-skills.mdx b/website/docs/v8.5.1/zh/guide/filesystem-skills.mdx new file mode 100644 index 00000000..33086c45 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-skills.mdx @@ -0,0 +1,74 @@ +--- +title: 'skills/ 技能目录' +description: '用工作区或智能体目录中的 skills/ 组织可复用技能、检查清单和领域流程。' +--- + +# skills/ 技能目录 + +`skills/` 目录存放可复用技能。技能适合表达稳定流程、检查清单、领域术语和工具使用准则;它不是 worker agent,也不负责启动独立会话。 + +A3S Code 有两类常见路径: + +```text +repo/.a3s/skills/ # workspace 级技能 +release-agent/skills/ # AgentDir 私有技能 +``` + +workspace 技能通过 `skillDirs` 或 `agent.acl` 的 `skill_dirs` 加载;agent 定义使用 +`agentDirs`,不要把 skill 目录放进 `agentDirs`。AgentDir 私有技能会在 +`serve_agent_dir` 为调度 session 创建上下文时注入。 + +## 技能文件 + +```md +--- +name: release-readiness +description: Check whether a repository is ready to release +allowed-tools: read(*), search(*), bash(pnpm test*), bash(cargo test*) +--- + +Always inspect: + +- package or crate version changes +- migration compatibility +- release notes +- required verification commands + +Return blockers first, then risks, then follow-up work. +``` + +frontmatter 帮助发现和筛选,正文描述执行方式。`allowed-tools` 应保持最小集合。 +当 skill 通过 `Skill` 工具被调用时,省略 `allowed-tools` 不会授予任何工具, +因此 invocation 默认 fail-secure。技能只是指导模型,不应该扩大权限。 + +## 何时使用 `skills/` + +适合: + +- 反复出现的 review checklist。 +- 产品、协议、发布、迁移等领域流程。 +- 一组工具调用的推荐顺序。 +- 对多个 worker agents 都有用的共享背景。 + +不适合: + +- 需要独立身份、独立上下文或被 `task` 调用的角色;放进 `agents/`。 +- 需要周期性触发的任务;放进 `schedules/`。 +- 需要连接外部系统的能力声明;放进 `tools/` 或宿主 MCP 配置。 + +## 加载方式 + +```ts +const session = agent.session('/repo', { + skillDirs: ['./.a3s/skills'], +}); +``` + +模型可以通过 `search_skills` 查找相关技能。文件型 skills 和 inline skills 使用同一套 discovery 语义。目录越多,越要保证 `name` 和 `description` 可检索、无重名歧义。 + +## 维护建议 + +- 技能文件应该短而稳定,避免塞入整本文档。 +- 需要示例时给最小可执行片段,不要复制大量日志。 +- 当 skill 只对某个长期 Agent 有效,放在该 AgentDir 的 `skills/` 中。 +- 当 skill 是整个仓库规范,放在 `.a3s/skills/` 中,并在 `AGENTS.md` 中说明它的使用边界。 diff --git a/website/docs/v8.5.1/zh/guide/filesystem-tools.mdx b/website/docs/v8.5.1/zh/guide/filesystem-tools.mdx new file mode 100644 index 00000000..a950ee77 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/filesystem-tools.mdx @@ -0,0 +1,69 @@ +--- +title: 'tools/ 工具目录' +description: '用智能体目录中的 tools/ 声明 MCP 和脚本工具,并保持权限、人工确认与允许列表边界。' +--- + +# tools/ 工具目录 + +`tools/` 是 AgentDir 的目录级工具声明。每个 `tools/.md` 描述一个模型可见能力,当前支持 `kind: mcp` 和 `kind: script` 两类。 + +```text +release-agent/ +└── tools/ + ├── github.md + └── search-auth.md +``` + +工具定义来自文件系统,但工具是否可见、能否执行、是否需要确认,仍由 harness、权限策略和 AgentDir loader 控制。文件存在不等于无限权限。 + +## MCP 工具 + +```md +--- +kind: mcp +name: github +transport: stdio +command: npx +args: ['-y', '@modelcontextprotocol/server-github'] +env: + GITHUB_TOKEN: '${GITHUB_TOKEN}' +--- + +GitHub issues and pull request tools. +``` + +每个已启用的调度 session 会在启动时连接该 MCP server,并获得命名空间化后的 `mcp__github__*` tools。secret 应通过环境变量注入,不要写进工具文件。 + +## 脚本工具 + +```md +--- +kind: script +name: search-auth +path: scripts/search-auth.js +allowed_tools: [grep, glob, read] +limits: + timeoutMs: 30000 + maxToolCalls: 30 + maxOutputBytes: 65536 +--- + +Find authentication-related files and return an evidence list. +``` + +`kind: script` 把一个预先参数化的 QuickJS `program` 调用暴露成模型可见工具。脚本源码 +必须定义 `async function run(ctx, inputs)`。它没有文件系统、网络、进程或环境变量权限, +只能通过 `ctx.tool(...)` 调用允许列表中的工具。 + +## 安全边界 + +- `allowed_tools` 是脚本内部能力边界;只列出最小集合。 +- 未知 `kind`、逃逸 workspace 的路径、重复 tool 名称、非法 limit 都应在加载时失败。 +- 不要让不可信目录声明高权限 MCP server 或脚本工具。 +- 高风险工具应继续走 HITL、allow-list 和审计。 + +## 当前作用域 + +`tools/` 由 `serve_agent_dir` 按调度 session 安装。也就是说,它服务于长期 Agent 和周期性任务。普通交互式 session 应优先使用宿主 direct tools、MCP 连接或 SDK 的 `session.tool(...)` 注册路径。 + +如果一个能力是项目通用连接器,放在宿主配置或 MCP 层更清晰;如果它只属于某个长期 Agent 的定时工作,放进该 AgentDir 的 `tools/`。 diff --git a/website/docs/v8.5.1/zh/guide/hooks.mdx b/website/docs/v8.5.1/zh/guide/hooks.mdx new file mode 100644 index 00000000..2c45d938 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/hooks.mdx @@ -0,0 +1,128 @@ +--- +title: '钩子' +description: '生命周期拦截与策略回调' +--- + +# 钩子 + +钩子用于在会话内注册生命周期回调。管理入口是 `registerHook()`、`hookCount()` 和 +`unregisterHook()`。 + +## 事件 + +Node.js 与 Python 的注册辅助方法接受这些稳定名称: + +```text +pre_tool_use +post_tool_use +generate_start +generate_end +session_start +session_end +skill_load +skill_unload +pre_prompt +post_response +on_error +pre_run_control +post_run_control +``` + +Rust Core 还把 `permission_request`、`pre_compact` 和 `post_compact` 暴露为一等生命 +周期点;Go bridge 可以注册相同的序列化枚举名称。Node.js 与 Python 的字符串解析器 +目前尚不接受这三个名称。因此,策略需要直接拦截权限请求或上下文压缩时,应使用 Rust +或 Go 宿主。 + +Core 还为专用 harness 定义了感知、memory、规划、推理、限流、确认、成功和意图事件。 +只有受保护的运行时路径确实消费某个事件的决策时,才能把它当作策略边界。 + +## 注册示例 + +```ts +session.registerHook( + 'release-publish-observer', + 'pre_tool_use', + { tool: 'bash', commandPattern: 'npm publish|twine upload|cargo publish' }, + { priority: 50, timeoutMs: 1000 }, + () => ({ action: 'continue' }), +); +``` + +处理函数可返回 `{ action: 'continue' }`、`{ action: 'skip' }`、 +`{ action: 'block', reason }`、`{ action: 'retry', reason, delayMs }`,也可返回空值表示 +继续。把钩子用作生产关卡前,应验证产品实际依赖的事件路径。 + +## 生命周期治理 + +真正具有门控语义的事件是 `pre_tool_use`、`permission_request`、`pre_compact`、 +`pre_prompt`、`pre_planning` 和 `pre_run_control`。这些事件的处理函数失败或超时时会关闭放行;其他事件 +属于观察或建议点,处理基础设施失败时继续运行。 + +`pre_tool_use` 可以在工具进入确认或产生副作用前替换参数: + +```ts +() => ({ + action: 'continue', + modified: { + updatedInput: { file_path: 'approved/release.txt', content: 'ready\n' }, + }, +}); +``` + +Core 接受 `updatedInput`、`updated_input`、`args` 或 `modified` 中的直接参数对象,也 +接受 Codex 风格的 `hookSpecificOutput` 包装。改写后的对象会再次通过工具 JSON Schema +校验;非法参数会在任何工具副作用发生前被拒绝。 + +`pre_prompt` 可以返回 `modified.prompt`,以及可选的 +`modified.additionalContext`(或 `additional_context`)。最终 prompt 会包含有边界的 +钩子上下文块,并真正替换发送给模型的用户消息。`permission_request` 可以返回 +`decision: 'allow'` 或 `decision: 'deny'`;`pre_compact` 可以阻止压缩。 +`session_start` 与 `session_end` 为宿主清理和审计提供配对的生命周期观察。 + +`pre_run_control` 会在 typed `steer` 或 `interrupt` 请求进入活动 Run 收件箱前执行 +门控;`post_run_control` 观察持久的 accepted、applied、settled 或 rejected 回执。 +使用相同 Request ID 与载荷重试时,会复用已经记录的准入结果,不会再次触发门控决策; +同一 ID 的冲突载荷会被拒绝。控制后的 Hook 仅用于观察,不能改写已经产生的回执。 + +## 拒绝反馈 + +无法在不改变请求、参数或策略上下文的情况下成功时,返回 `block`。临时条件应返回 +`retry`,并提供原因和建议延迟: + +```ts +() => ({ + action: 'retry', + reason: '策略后端暂时不可用。', + delayMs: 1000, +}); +``` + +Python 回调使用 `delay_ms`;Go 回调返回 +`&code.HookResponse{Action: "retry", Reason: "...", DelayMS: 1000}`。当前调用会被拒绝, +而不是自动重新调度。模型会收到原因和明确的重试指引,直接 SDK 调用者则收到结构化 +工具错误: + +```json +{ + "type": "hook_denied", + "reason": "策略后端暂时不可用。", + "retryable": true, + "retry_after_ms": 1000 +} +``` + +`block` 使用同一错误类型,但 `retryable` 为 `false`、`retry_after_ms` 为 `null`。 +需要保留完整重试说明的 Rust 调用者可使用 `HookEngine::fire_outcome()`;现有 `fire()` +API 保留旧的 `HookResult` 投影。 + +## 传播 + +委派与自动子智能体扇出都经过 `task` 工具。产品依赖钩子跨委派运行传播时,应覆盖对应 +的产品集成路径。 + +## 管理 + +```ts +console.log(session.hookCount()); +session.unregisterHook('release-publish-observer'); +``` diff --git a/website/docs/v8.5.1/zh/guide/index.mdx b/website/docs/v8.5.1/zh/guide/index.mdx new file mode 100644 index 00000000..2777bf5b --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/index.mdx @@ -0,0 +1,320 @@ +--- +title: '概览' +description: 'A3S Code 的安装方式、主要能力和 SDK 入口' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# A3S Code + +A3S Code 是 `a3s code` 终端应用背后的 Rust Runtime,也可以单独接入 IDE、 +Runner 或服务端。它负责 Agent Loop、上下文、工具调用、权限检查、子任务、 +异步 Workspace 检索、持久证据,以及任务的保存和恢复。 + +8.5.1 保持变薄的 `local-code` Harness,并分离工作区搜索平面:精确 `grep` +(可选进程内 trigram 候选裁剪)不会打开持久 zvec FTS,而 `bm25` 仍是排序词法 +路径。会话存储在跨进程 flock 下恢复损坏的 WAL,避免并发写者铸造冲突序号。 + +## 设计规则 + +| 规则 | 含义 | +| ---------------- | -------------------------------------------------------------------------- | +| 默认变薄 | Core `default` = `local-code`。Advanced 评估、server、无头搜索需显式开启。 | +| Grep ≠ zvec | 精确 `grep` 是匹配权威;trigram 裁剪仅失败开放。排序检索使用 `bm25`。 | +| 单一委派路径 | 多条目扇出使用 `task` / `session.tasks`。不要恢复 `parallel_task`。 | +| 仅 Active 记忆 | Durable 服务路径为 `active_recall`。拒绝 Candidate shadow。 | +| Gate 先要证据 | 证据不完整或存在 retention 缺口时,Gate 不可宣称已完成评估。 | +| 宿主拥有产品策略 | 评审 rubric、Cloud 审计与 UI 确认留在 Core 之外。 | + +想直接在终端里使用,安装 [`a3s` CLI](https://github.com/A3S-Lab/a3s)。 +想构建自己的产品,使用 Rust crate、Node.js 包、Python 包,或配合原生桥接程序 +使用 Go module。它们输出同一套事件,因此不同界面不需要各写一套 Agent Loop。 + +## 先选使用方式 + +| 入口 | 什么时候用 | 仓库 | +| -------------------------------- | ------------------------------------------------ | ----------------------------------------------- | +| Rust / Node.js / Python / Go SDK | 把编码 Agent 接入 IDE、Runner、服务端或自己的 UI | [A3S-Lab/Code](https://github.com/A3S-Lab/Code) | +| `a3s code` | 直接在终端里运行编码 Agent | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | +| `a3s-tui` | 构建终端界面,不包含 Agent Runtime | [A3S-Lab/TUI](https://github.com/A3S-Lab/TUI) | +| A3S Flow | 为 `DynamicWorkflowRuntime` 保存和恢复流程 | [A3S-Lab/Flow](https://github.com/A3S-Lab/Flow) | + +## 主要能力 + +| 领域 | 可以做什么 | +| ---------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Agent 会话 | 通过异步 `SessionBuilder` 创建绑定到 Workspace 的 `AgentSession`;支持 `send`、`run`、`stream`、steer、interrupt、取消、保存、恢复和清理。同一会话发生并发冲突时会立即报错。 | +| 终端界面 | [`a3s code`](/guide/tui) 把事件流显示成终端对话,并展示工具调用、确认提示、记忆、文件和会话状态。 | +| 项目约定 | [文件系统优先](/guide/filesystem-first) 介绍如何用 `AGENTS.md`、`agent.acl`、`.a3s/agents/`、`skills/`、`tools/` 和 `schedules/` 管理项目配置。 | +| 工具 | 内置文件、二进制安全的本地下载、搜索、Shell、Git、网页、Batch、结构化输出、QuickJS、Skills、MCP 和子任务工具。模型调用统一经过参数、权限、确认、取消、确定性结果投影与证据记录。 | +| 命令 | [命令](/guide/commands) 说明 TUI 的斜杠命令,以及如何在 SDK 中注册自己的 `/command`。 | +| 子任务 | 通过模型可见的 `task`,或宿主侧 `session.task(...)` 与 `session.tasks(...)` 使用内置或自定义 Agent;单项聚焦执行,多项独立任务并发扇出。 | +| 任务调度 | 一个 [Agent 级优先级调度器](/guide/tasks#agent-wide-priority-scheduler)在 Session、直接工具、detached 子任务和宿主工作流之间共享本地执行容量,并提供 FIFO、老化、取消和占用快照。 | +| 流程编排 | [`session.parallel`](/guide/orchestration)、`session.pipeline`、阶段、Checkpoint、循环上限和预算记录,可用于编写固定且可恢复的工作流。 | +| 定时任务 | [AgentDir](/guide/agent-dir) 可以通过 `serveAgentDir` / `serve_agent_dir` 运行定时 Turn;每个 Schedule 使用稳定的 `schedule:` Session。 | +| 安全 | 权限策略、用户确认、预算、Workspace 路径检查、工具超时、生命周期 Hook 和输出清理都会在执行过程中生效。 | +| Workspace | [Workspace 后端](/guide/workspace-backends) 支持本地文件、应用提供的 Workspace、可选的 S3 兼容存储和 remote-git 服务;原生 Harness 可用 detached Git worktree 隔离会话。 | +| 工作区检索 | [工作区检索](/guide/context#workspace-retrieval)提供异步 Session 文本目录、增量 BM25、可选宿主 Embedding、精确内存向量、Hybrid RRF 与可选确定性 CPU Rerank,不依赖向量数据库。 | +| 事件 | `EventEnvelopeV1` 是 Rust、Node.js、Python、Go 共用的事件格式;未知事件的 Payload 和 Metadata 也会保留。 | +| 保存与恢复 | `SessionSnapshotV1`、Session ID、Auto-save、Run Event、Trace、Artifact、Loop / Workflow Checkpoint 和 Memory Store 用来恢复会话和回放运行过程。 | +| 验证 | [验证](/guide/verification) 支持验证命令、预设、结构化报告、摘要、Artifact、Trace Event 和 Run Replay。 | + +## v8.5.1 新增内容 + +- 默认 `local-code` 的 **`grep` trigram 裁剪**(`CODE-G1`):字面量模式在 + `.a3s-code/grep-trigram` 下构建失败开放的候选缓存,且不会打开持久 zvec FTS。 + 精确正则匹配仍由 Code 拥有。 +- **会话存储 WAL flock 恢复**:并发写者在跨进程 flock 下重读持久最大序号;宿主 + 可隔离损坏的 WAL,并从持久快照继续。 +- 延续 8.4 变薄 Harness(仅 Active 记忆、统一 `task`、`update_plan`、SDK + capabilities v2)以及 8.3 耐久/信任内核。 + +## 早期 v8.4.0 新增内容 + +- 库与 SDK **默认变薄**:`a3s-code-core` 默认 `local-code`(捆绑 zvec FTS); + Node / Python / Go SDK crate 默认 `zvec-rust-fts-bundled`。产品嵌入需要时再显式 + 启用 `advanced-harness`、`server` 和/或 `headless-search`。 +- 模型可见的 **`parallel_task` 已移除**;多条目扇出统一使用 `task` 工具。对应的 + Node / Python / Go 辅助 API 一并删除。 +- Durable Memory 服务路径仅为 **Active**(`active_recall`)。Candidate shadow + 模式已移除;抽取仍可写入 Candidate,直至宿主激活。 +- 内置 **`update_plan`** 清单工具,以及宿主侧 + `set_output_language` / `outputLanguage`(Rust 与各 SDK)。 +- SDK capabilities 发布为 **`a3s-code/sdk-capabilities/v2`**, + `tier: baseline | advanced`。 +- Gate 模式评估在证据不完整时 fail-close;首性原则 E2E 与 Harness 收口手册见 + `manual/FIRST_PRINCIPLES_E2E.md` 与 `manual/HARNESS_CONVERGENCE.md`。 +- 延续 8.3 耐久/信任内核(可协商 Session Store、工具结果信任标签、工作区来源 + 快照、可失败 FFI 初始化,以及宿主侧不可变内容 / Checkpoint 钩子)。 + +## 早期 v8.3.0 新增内容 + +- Session Store 耐久性可协商(KRN-6)。内置 memory / file 适配器会精确声明聚合 + CAS、append-only WAL、写者租约 fencing、可选 AES-256-GCM 静态加密、提交 watch, + 以及引用感知的 Artifact GC。 +- 每个工具结果都带有类型化信任标签(KRN-5):trusted、workspace data 或 + external。Rust 与四种 SDK 都暴露无密钥的 `model_middleware_health` 计数器。 +- 工作区检索把结果绑定到可防篡改的来源快照(KRN-4)。持久 BM25 / zvec 索引在 + Windows 与 Linux 上通过发布资格验证,包括在原生打开前剥离 Windows `\\?\` 路径。 +- Node.js / Python 的 FFI 运行时初始化可失败(KRN-9)。 + `TASK_ADMISSION_AT_CAPACITY` 在各 SDK 上映射一致。 +- Linux arm64 Python Wheel 使用 `manylinux_2_39_aarch64`(glibc 2.39+); + Linux x86_64 仍为 `manylinux_2_28`。 + +## 早期 v8.2.0 新增内容 + +- `steer` 会在活动 Run 的下一个安全点加入更新指令;`interrupt` 会协作式停止 + Provider、工具、工作流和委派任务。幂等回执与可选的预期 Turn 字段可以拒绝重复或 + 过期的界面操作。 +- `pre_run_control` 可以门控控制请求,`post_run_control` 观察每次持久回执变化; + `run_control_applied` 会进入共用事件协议与持久 Run 历史。 +- 默认提示词现在由精简执行循环、运行时权威契约、仓库工具 schema 和安全边界分层 + 组成。文件、工具输出与网页内容都按不可信数据处理;完成声明必须有证据,但提示词 + 不会取代宿主的权限、确认或沙箱。 +- 发布资格测试使用配置中的两个 DeepSeek 模型,真实覆盖工具与 Hook 参数改写、长程 + 编码、SubAgents、Skills、PTC、可重放动态工作流,以及在线 steer/interrupt。 +- 动态工作流的私有 `program` 实现步骤不再重复请求权限,脚本中的实际工具仍完整受 + 治理。QuickJS `ctx.readFile()` 现在返回文件文本;`ctx.read()` 继续保留带行号的 + 工具结果,供审计型脚本使用。 + +## 早期 v8.1 新增内容 + +- `web_search` 使用 `a3s-search` v3.1.0。Google、Baidu、Bing 和 Brave 的浏览器 + 引擎默认使用 Moli;Chrome/Chromium 与 Lightpanda 仍可显式选择。首次使用按 + sidecar、经过校验的用户缓存、系统可执行文件、带 Digest 的 HTTPS 下载顺序发现。 +- Moli 安装采用原子暂存、Receipt 和跨进程锁,默认位置为 + `~/.cache/a3s-code/moli`(也可设置 `A3S_CODE_MOLI_CACHE_DIR`)。第二个 + A3S Code 进程会等待首次安装并复用同一个可执行文件。若宿主必须禁止网络,设置 + `auto_download_moli = false`,缺少运行时时会 fail-closed。 +- Rust、Node.js、Python 和 Go 都公开相同的 `sdk_capabilities` 清单、状态图操作、 + Moli 诊断和类型化搜索配置。使用清单做能力发现,不要解析包内文件来猜测能力。 +- Node 原生平台包和 Python Wheel 都包含对应目标的 Moli sidecar 及来源元数据。 + Linux musl 包会明确写入 `MOLI_UNAVAILABLE`,因为上游 Moli 没有发布 musl + 二进制;这类主机应提供系统 Moli 或选择其他后端。 + +## v7.0 的历史新增内容 + +- Session-owned Workspace Retrieval 会异步构建一个有界文本目录,复用增量 BM25 + Posting,并可发布精确内存向量分区;Session 构建不会等待索引,也不需要向量数据库。 +- 语义检索必须由宿主显式开启。Exact、Glob、BM25、Code Intelligence、RRF 和可选的 + 确定性 Reranker 都在本地 CPU 上运行;需要语义检索时,宿主可以注入进程内 CPU + Embedding 回调。 +- Rust、Node.js、Python 和 Go 统一提供 Typed Line、Fixed-window、Recursive + Chunking、Readiness 与 Batching 指标、可选确定性 Rerank、取消、当前源摘要校验和 + 有界清理。非文本文件在切块和 Embedding 之前就会被排除。 +- 模型边界 Run 证据会绑定实际能力与 Policy Identity、检索 Generation、输入形状、 + 重复 Tool result 上下文和归一化 Usage,同时不会新增保存 Prompt、源文本、向量、 + 凭据或端点明文。 + +Go 使用方必须改用 v7 Module 路径: +`github.com/A3S-Lab/Code/sdk/go/v8`。 + +v6.9 引入了共享优先级/FIFO 调度器、有界个人与项目指令链、受治理生命周期 Hook、 +隔离 Harness Git worktree、确定性 Tool-result 证据和精确 Cognitive-package +Generation Binding。 + +v6.8 只向模型呈现一套紧凑的 `search` 模式,统一 grep、glob 和零依赖 BM25 排序; +同时以一个模型可见的 `task` 模式覆盖聚焦任务与有界并发扇出。旧的宿主直调别名仍可读取, +但不再消耗模型工具 Schema Token。本版本还为 Headless 宿主增加精确、重放安全的 Run +准入,并通过 Node.js、Python 和 Go SDK 返回统一的快照与重放状态。 + +v6.7 的本地 `download` 工具会通过 SSRF 安全的重定向校验,把 HTTP(S) 资源流式写入 +相邻临时文件;它支持自适应 1–4 连接 Range 传输,并可在原子提升前校验 64 位 +`expected_sha256`。这是受权限与 HITL 治理的工作区修改操作,远端工作区后端不会注册。 +详见[工具](/guide/tools#二进制安全的本地下载)。 + +搜索使用结构门控级联:Core 默认 feature 中的 Moli 层、HTTP/RSS 引擎, +然后是原生 API。完整级联仍未达到结构化检索要求时会失败关闭,不会把弱候选当成成功 +证据,也不需要外部语义验证器或重排序 API。委派上下文共享 bulkhead、重试预算和 +相同请求合并,详见[工具](/guide/tools#结构门控的网页搜索)。 + +一次执行大致会经过这些步骤: + +```text +Agent / AgentSession + -> 收集项目上下文 + -> 可选 Plan + -> 模型选择工具或子任务 + -> 检查权限,必要时询问用户 + -> 执行 + -> 发送事件和验证结果 + -> 保存会话 +``` + +## 安装 + +想使用交互式终端应用时,运行对应平台的一键安装脚本: + + + + +```bash +curl --proto '=https' --tlsv1.2 -LsSf \ + https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.sh | sh +``` + + + + +```powershell +[Net.ServicePointManager]::SecurityProtocol = [Net.ServicePointManager]::SecurityProtocol -bor [Net.SecurityProtocolType]::Tls12 +irm https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.ps1 | iex +``` + + + + +脚本会选择当前系统与架构对应的发布包并校验 SHA-256。也可以使用 +`brew install a3s-lab/tap/a3s` 或 `cargo install a3s`。 + +当你要把 A3S Code 嵌入自己的产品时,安装 SDK 包: + +```bash +npm install @a3s-lab/code +pip install a3s-code +cargo add a3s-code-core +go get github.com/A3S-Lab/Code/sdk/go/v8 +``` + +v8.5.1 的 Python 包通过每个平台一个 `cp310-abi3` Wheel 支持 CPython 3.10 至 +3.14。Intel Mac 使用 `macosx_12_0_x86_64` Wheel,最低需要 macOS 12。请参阅 +[Python Wheel 平台](/api/#python-wheel-平台-v820)了解 Bootstrap 流程,以及 +`No module named pip` 的修复命令。 + +## 配置 + +A3S Code 使用 ACL。不要把真实 API Key、私有模型端点、本地配置路径或 +tenant/user 标识提交到仓库;提交只通过环境变量解析凭据的模板。 + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +agent_dirs = ["./.a3s/agents"] +skill_dirs = ["./skills"] +storage_backend = "file" +sessions_dir = ".a3s/sessions" + +search { + headless { + backend = "moli" + auto_download_moli = true + max_tabs = 4 + } +} +``` + +`auto_parallel = false` 只关闭自动并行子智能体扇出。手动 `task` 调用和 SDK +`session.tasks(...)` 扇出仍然可用,除非你单独关闭手动委派。 + +SDK 包会优先选择随包提供的 Moli sidecar。源码构建或精简 Core 可以设置 +`A3S_CODE_MOLI_EXECUTABLE` 指向已校验的可执行文件;否则首次搜索会把固定版本 +下载到共享缓存。 +本地 session 持久化需要把 `storage_backend = "file"` 和 `sessions_dir` 配对; +SDK embedding 场景也可以直接传 typed `FileSessionStore`。 + +## 使用 TUI + +在你希望 agent 检查的 workspace 中运行 `a3s code`: + +```bash +a3s code +a3s code resume +a3s code update +``` + +TUI 会按顺序发现配置:`A3S_CONFIG_FILE`、从当前目录向上查找的 +`.a3s/config.acl`,以及 `~/.a3s/config.acl`。 +顶层 `a3s update` 命令会进入同一个 updater。 + +## 使用 SDK + +```ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/path/to/workspace', { + planningMode: 'auto', + permissionPolicy: { + allow: ['read(*)', 'search(*)'], + ask: ['bash(*)', 'write(*)'], + deny: ['write(**/.env*)', 'bash(rm -rf*)'], + defaultDecision: 'ask', + enabled: true, + }, +}); + +const result = await session.send('Find the authentication entry points.'); +console.log(result.text); +console.log(result.verificationSummaryText); + +session.close(); +await agent.close(); +``` + +## 接着看 + +- [A3S Code TUI](/guide/tui) 说明安装、配置发现、斜杠命令和 Effort。 +- [SDK 与 API](/api/#go-module-与桥接程序)说明 Go 模块和桥接程序的安装方式;后续 Go 示例会与 Node.js、Python 一起出现在各通用章节中。 +- [文件系统优先](/guide/filesystem-first) 介绍 `AGENTS.md`、ACL、AgentDir、Skill、工具和定时任务。 +- [API 契约](/guide/api-contract) 列出经过集成测试的 Node.js API。 +- [会话](/guide/sessions) 介绍创建、流式输出、运行状态、保存、恢复和取消。 +- [工具](/guide/tools) 介绍直接调用、错误类型、结构化输出和 QuickJS Program。 +- [工作区检索](/guide/context#workspace-retrieval)介绍显式语义开关、切块、生命周期、质量指标和 CPU-only 安全默认值。 +- [任务](/guide/tasks) 与 [编排](/guide/orchestration) 介绍子 Agent 和固定工作流。 +- [安全](/guide/security)、[Hook](/guide/hooks) 与 [验证](/guide/verification) 介绍执行前、执行中和执行后的检查。 +- [Memory](/guide/memory) 与 [持久化](/guide/persistence) 介绍跨会话信息和任务恢复。 diff --git a/website/docs/v8.5.1/zh/guide/isolation.mdx b/website/docs/v8.5.1/zh/guide/isolation.mdx new file mode 100644 index 00000000..2bf64e55 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/isolation.mdx @@ -0,0 +1,86 @@ +--- +title: '隔离' +description: '工作区、上下文与委派任务隔离' +--- + +# 隔离 + +隔离从 session 边界开始:每个 session 绑定一个 workspace,委派子运行接收有边界的 +context。宿主直接工具调用是特权操作;暴露给用户前,应在宿主应用内先做权限判断。 + +## 工作区边界 + +```ts +const session = agent.session('/repo'); +``` + +相对文件、搜索、shell 和 Git 操作都从会话工作区求值。Security Provider 与钩子属于 +会话选项;把它们当作生产边界前,应验证依赖的精确策略路径。 + +## 委派上下文 + +`task` 与自动子智能体委派会隔离子运行推理。父 Agent 只接收紧凑结果,而不是完整 +transcript,从而减少 prompt 污染并简化证据审查。 + +## 存储隔离 + +不同产品、租户或测试套件应使用独立的 memory 与会话存储目录: + +```ts +import { FileMemoryStore, FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + +## 原生 Harness Worktree + +当 `a3s code harness` 指向 Git 仓库时,每个获得准入的协议会话都会在源仓库 `HEAD` +创建自己的临时 detached worktree。准入时要求源 worktree 干净,因此宿主本地修改不会 +被静默遗漏。不同会话不共享可变文件;移除 Harness 会话时,其临时 worktree 也会被 +删除,但任何修改都不会自动写回源 checkout。 + +每次运行前后,Core 都会通过隔离的临时 Git index 捕获 tracked 以及未忽略的 untracked +内容。运行进入终态后,它生成一份支持二进制、完整 index 的 unified diff,并把结果 +tree 固定到私有 Git ref。不可变证据只属于该精确运行;冲突的第二次捕获会被拒绝。 +持久化会话恢复时,会把最近一次捕获的结果 tree 恢复到新的 detached worktree。 + +非 Git 工作区仍使用配置的共享路径,不能生成这种 Git changeset 证据。源 Git +worktree 不干净时,隔离准入会失败,而不会回退到共享写入。 + +## 有边界的 Changeset 协议 + +事件流到达 `completed`、`failed` 或 `cancelled` 后,把精确运行身份 POST 到 +`/v1/agent/changes`: + +```json +{ + "schema": "a3s.code.agent-change-set-request.v1", + "identity": { + "schema": "a3s.code.agent-run-identity.v1", + "protocol": "a3s.code.agent.v1", + "agent_release_identity": "sha256:", + "session_id": "conversation-018f4f86", + "run_id": "run-018f4f86-attempt-1" + } +} +``` + +响应使用 `a3s.code.agent-change-set.v1`,包含相同 identity、终态、`base_tree`、 +`result_tree`、`patch_digest`、`patch_bytes`、`observed_at_ms`,以及格式为 +`git_unified_diff_v1`、编码为 `base64` 的 `patch_base64`。原始补丁上限为 4 MiB。 + +变更捕获在 worker 结束后紧接着完成,因此已经进入终态的运行可能短暂返回 +`a3s.code.agent_protocol.change_set_pending`;此时重试同一查询。 +`a3s.code.agent_protocol.change_set_unavailable` 表示工作区不兼容 Git,或捕获结果无法 +满足协议上限。 + +由调用方决定是否以及在何处应用补丁,接口本身不会自动合并。解码或应用前,应校验 +声明的字节数与 `sha256:` 摘要,并保留两个 Git tree 身份。这样既能隔离并发会话写入, +又能向宿主提供确定性的合并或审查制品。 + +## 外部 Harness + +钩子与权限策略是集成点。将其视作生产边界前,应使用自己的 Harness 测试真实组织策略。 diff --git a/website/docs/v8.5.1/zh/guide/lane-queue.mdx b/website/docs/v8.5.1/zh/guide/lane-queue.mdx new file mode 100644 index 00000000..ec4a2bbc --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/lane-queue.mdx @@ -0,0 +1,46 @@ +--- +title: '执行通道队列' +description: '可选外部与混合分发基础设施' +--- + +# 执行通道队列 + +普通 session 默认不启用队列。Lane queue 是显式的高级基础设施,独立于 `task` 和自动 subagent 委派。 + +不要把这层可选分发设施与始终启用的 +[Agent 级任务调度器](/guide/tasks#agent-wide-priority-scheduler)混淆。任务调度器内部使用 `a3s-lane` +优先级队列,在不同 Session 之间共享本地执行容量;Lane queue 则把选中的工作交给 +本地、外部或混合 handler。 + +只有需要外部 worker、混合本地/远程执行、队列指标、死信或多机器分发时才启用 lane queue。 + +```ts +const session = agent.session('/repo', { + queueConfig: { + enableDlq: true, + enableMetrics: true, + }, +}); + +await session.setLaneHandler('execute', { + mode: 'external', + timeoutMs: 300000, +}); +``` + +lane 名称是 `control`、`query`、`execute` 和 `generate`。 + +外部 worker 完成任务时调用: + +```ts +const pending = await session.pendingExternalTasks(); + +if (pending.length > 0) { + await session.completeExternalTask(pending[0].id, { + success: true, + result: { summary: 'worker completed the task' }, + }); +} +``` + +普通本地 agent session 不需要引入队列。 diff --git a/website/docs/v8.5.1/zh/guide/limits.mdx b/website/docs/v8.5.1/zh/guide/limits.mdx new file mode 100644 index 00000000..772da932 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/limits.mdx @@ -0,0 +1,183 @@ +--- +title: '运行限制' +description: '运行时限制、压缩、超时与熔断' +--- + +# 运行限制 + +Limit options 是 session 级控制项,用于长任务、噪声工具输出和 provider +失败场景。 + +## 会话选项 + +```ts +const session = agent.session('/repo', { + maxToolRounds: 24, + maxParseRetries: 3, + toolTimeoutMs: 120000, + circuitBreakerThreshold: 4, + autoCompact: true, + autoCompactThreshold: 0.75, + continuationEnabled: true, + maxContinuationTurns: 3, +}); +``` + +## 选项用途 + +这些 option 的意图: + +- `maxToolRounds` 是单个 turn 的 tool iteration 预算。 +- `maxParseRetries` 是格式错误工具调用的恢复预算。 +- `toolTimeoutMs` 是每个 tool 的超时时间,单位毫秒。 +- `circuitBreakerThreshold` 是连续 provider 失败阈值。 +- `autoCompact` 和 `autoCompactThreshold` 控制 context compaction 行为。 +- `continuationEnabled` 和 `maxContinuationTurns` 控制 continuation injection。 + +## 实用默认值 + +CI、发布和用户可见自动化应使用严格限制;本地探索可以放宽预算,但有副作用的任务仍要显式验证。 + +## 保留上限 + +session 会把 run 历史、trace events 和 subagent task 快照保存在内存里。 +`SessionRetentionLimits` 使用保守的有限默认值,避免长时间会话无限增长。可以覆盖 +任意单项 FIFO 上限;只有明确设置 `unbounded: true` 才会启用无限保留。 + +五个相互独立的上限: + +| 字段 | 触发上限时的效果 | +| ----------------------------- | -------------------------------------------------------------------------------------------------------------------- | +| `max_runs_retained` | 新建 run 超过上限时,**最旧**的 run 及其全部 events 被丢弃。 | +| `max_events_per_run` | run 中最旧的 events 按 FIFO 丢弃。run 快照的 `event_count` **不会**被递减——它保持有史以来记录的累计总数。 | +| `max_event_bytes_per_run` | 丢弃最旧的 run events,直到满足序列化字节上限;单条超大 event 不会被保留。 | +| `max_trace_events` | trace sink 超过上限后,每次新写入都会丢弃最旧的一条 event。 | +| `max_terminal_subagent_tasks` | 超过上限后丢弃最旧的**终态**(completed / failed / cancelled)subagent task 快照。**running 的任务永远不会被丢弃。** | + +所有上限都是软上限:触发时在插入处丢弃最旧条目,绝不返回错误。 + +```ts +const session = agent.session('/repo', { + retentionLimits: { + maxRunsRetained: 100, + maxEventsPerRun: 5000, + maxEventBytesPerRun: 16 * 1024 * 1024, + maxTraceEvents: 20000, + maxTerminalSubagentTasks: 500, + }, +}); +``` + +```python +opts = SessionOptions() +opts.retention_limits = { + 'max_runs_retained': 100, + 'max_events_per_run': 5000, + 'max_event_bytes_per_run': 16 * 1024 * 1024, + 'max_trace_events': 20000, + 'max_terminal_subagent_tasks': 500, +} +session = agent.session('/repo', opts) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + RetentionLimits: &code.RetentionLimits{ + MaxRunsRetained: code.Ptr(uint(100)), + MaxEventsPerRun: code.Ptr(uint(5_000)), + MaxEventBytesPerRun: code.Ptr(uint(16 * 1024 * 1024)), + MaxTraceEvents: code.Ptr(uint(20_000)), + MaxTerminalSubagentTasks: code.Ptr(uint(500)), + }, +}) +``` + +Rust 使用对应的 `SessionRetentionLimits` builder 方法。 + +## 预算守卫 + +`BudgetGuard`(见更新日志 [3.3.0])是一套由宿主提供的成本与配额契约。框架本身 +不强制执行预算——它只定义决策点,并咨询宿主注入的守卫。在 LLM 与工具调用处接入 +三个钩子: + +- `check_before_llm` —— 在每次 LLM 调用之前。 +- `record_after_llm` —— 在每次成功的 LLM 调用之后,带上服务提供商的实际用量, + 以便宿主准确累计消费。 +- `check_before_tool` —— 在每次工具调用之前。 + +每个 `check_*` 返回三种决策之一: + +- `Allow` —— 正常继续,不发出事件。 +- `SoftLimit { resource, consumed, limit, message }` —— 发出 + `AgentEvent::BudgetThresholdHit { kind: "soft" }` 并**继续执行**。会话内的钩子 + 可以据此采取动作,例如自动压缩或下一轮换用更便宜的模型。 +- `Deny { resource, reason }` —— 以 `CodeError::BudgetExhausted` 中止本次调用。 + **会话仍然保持打开**——调用方可以稍后重试,或在宿主重新分配预算后重试。 + +### Node.js:`session.setBudgetGuard({...})` + +每个回调接收**单个 `ctx` 对象**(不是位置参数),并返回一个决策对象(或 `null` / +`{ decision: 'allow' }` 表示放行): + +```ts +session.setBudgetGuard({ + checkBeforeLlm: (ctx) => { + // ctx.sessionId, ctx.estimatedTokens + if (overMonthlyCap(ctx.sessionId)) { + return { + decision: 'deny', + resource: 'llm_tokens', + reason: 'monthly cap', + }; + } + return { decision: 'allow' }; + }, + recordAfterLlm: (ctx) => { + // ctx.sessionId, ctx.usage —— usage 的 key 是 camelCase: + // promptTokens, completionTokens, totalTokens, cacheReadTokens, cacheWriteTokens + addSpend(ctx.sessionId, ctx.usage.totalTokens); + }, + checkBeforeTool: (ctx) => { + // ctx.sessionId, ctx.toolName + return { decision: 'allow' }; + }, + timeoutMs: 5000, // 可选,默认 5000 +}); +``` + +Node.js 桥接层采用**失败即拒绝**策略:`check_*` 回调若未在 `timeoutMs` 内返回, +或返回无法解析的值,都会被当作**拒绝**处理。预算控制绝不能在预算守卫卡住时悄悄 +自我失效(参见更新日志 `[3.3.0]` 中 Node.js 预算守卫的“失败时放行”修复)。 + +回调**绝不能抛出异常。** 受 napi-rs 约束,回调抛出的异常会在返回值转换阶段中止宿主 +进程。请用 `try/catch` 包裹逻辑并返回一个决策(例如拒绝),而不是抛异常。卡住的情况 +由“失败即拒绝”的超时机制安全处理,详见更新日志 [3.3.0] 的已知限制。 + +### Python:会话选项 `budget_guard` + +Python 在 `budget_guard` 这个 `SessionOptions` 字段上提供一个 `BudgetGuard` 形态的 +对象。未定义的方法视为放行且不执行操作。Python 回调使用**位置参数**,框架会捕获 +它们抛出的任何异常(抛异常的 `check_*` 默认视为放行): + +```python +class MyBudgetGuard: + def check_before_llm(self, session_id, est_tokens): + if over_monthly_cap(session_id): + return {'decision': 'deny', 'resource': 'llm_tokens', 'reason': 'monthly cap'} + return {'decision': 'allow'} + + def record_after_llm(self, session_id, usage): + # usage 是一个 dict,key 为 snake_case: + # total_tokens, cache_read_tokens(以及 prompt_tokens, completion_tokens, cache_write_tokens) + add_spend(session_id, usage['total_tokens']) + + def check_before_tool(self, session_id, tool_name): + return {'decision': 'allow'} + +opts = SessionOptions() +opts.budget_guard = MyBudgetGuard() +session = agent.session('/repo', opts) +``` + +决策返回字典 `{"decision": "deny", "resource": ..., "reason": ...}`(以及 +`"soft"` / `"allow"`)在两个 SDK 上具有相同结构。 diff --git a/website/docs/v8.5.1/zh/guide/mcp.mdx b/website/docs/v8.5.1/zh/guide/mcp.mdx new file mode 100644 index 00000000..97567a55 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/mcp.mdx @@ -0,0 +1,241 @@ +--- +title: 'MCP' +description: '模型上下文协议服务集成' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# MCP + +MCP 把 A3S Code session 连接到外部工具服务器。stdio server 可以加入 live +session、检查状态、通过注册工具调用,并在不需要时移除。server 的工具会注册到 +session 中,名称格式为 `mcp____`。 + +## 添加服务器 + +新代码优先使用 object-shaped `addMcp(...)` API。它直接对应 core 的 +`McpServerConfig` 形状,避免 positional overload 的参数顺序问题。 + + + + +```rust +use a3s_code_core::mcp::{McpServerConfig, McpTransportConfig}; +use std::collections::HashMap; + +let count = session + .add_mcp_server(McpServerConfig { + name: "echo".into(), + transport: McpTransportConfig::Stdio { + command: "node".into(), + args: vec![ + "tools/mcp_echo_server.mjs".into(), + "example-value".into(), + ], + }, + enabled: true, + env: HashMap::new(), + oauth: None, + tool_timeout_secs: 30, + }) + .await?; + +println!("已注册工具数:{count}"); +``` + + + + +```ts +const count = await session.addMcp({ + name: 'echo', + transport: { + type: 'stdio', + command: process.execPath, + args: ['tools/mcp_echo_server.mjs', 'example-value'], + }, + timeoutMs: 30000, +}); + +console.log('registered tools:', count); +``` + + + + +```python +count = session.add_mcp({ + 'name': 'echo', + 'transport': { + 'type': 'stdio', + 'command': 'python', + 'args': ['tools/mcp_echo_server.py', 'example-value'], + }, + 'timeout_ms': 30000, +}) +print('registered tools:', count) +``` + + + + +```go +count, err := session.AddMCPServer(ctx, code.MCPServerConfig{ + Name: "echo", + Transport: code.MCPTransport{ + Type: "stdio", + Command: "node", + Args: []string{"tools/mcp_echo_server.mjs", "example-value"}, + }, + ToolTimeoutSecs: 30, +}) +if err != nil { + return err +} +fmt.Println("已注册工具:", count) +``` + + + + +Python 通过 `session.add_mcp({...})` 暴露相同的对象式 API。 +`addMcpServer(...)` / `add_mcp_server(...)` 和 +`addMcpServerConfig(...)` / `add_mcp_server_config(...)` 仍作为兼容别名保留。 + +远程 server 使用嵌套 transport object,`type` 取 `'http'` 或 +`'streamable-http'`。凭据应由宿主环境或 secret manager 注入;在生产文档或发布说明依赖前,先针对你的 server 做集成测试。 + +## 检查和移除 + + + + +```rust +use serde_json::json; + +println!("{:#?}", session.mcp_status().await); +println!( + "{:#?}", + session + .tool_names() + .into_iter() + .filter(|name| name.starts_with("mcp__")) + .collect::>() +); +let result = session + .tool("mcp__echo__echo", json!({ "message": "文档 MCP 正常" })) + .await?; +println!("{}", result.output); +session.remove_mcp_server("echo").await?; +``` + + + + +```ts +console.log(await session.mcpStatus()); +console.log(session.toolNames().filter((name) => name.startsWith('mcp__'))); +await session.tool('mcp__echo__echo', { message: 'docs mcp ok' }); +await session.removeMcpServer('echo'); +``` + + + + +```python +print(session.mcp_status()) +print([name for name in session.tool_names() if name.startswith('mcp__')]) +session.tool('mcp__echo__echo', {'message': 'docs mcp ok'}) +session.remove_mcp_server('echo') +``` + + + + +```go +status, err := session.MCPStatus(ctx) +if err != nil { + return err +} +names, err := session.ToolNames(ctx) +if err != nil { + return err +} +result, err := session.Tool(ctx, "mcp__echo__echo", map[string]any{ + "message": "docs mcp ok", +}) +if err != nil { + return err +} +fmt.Println(status, names, result.Output) +if err := session.RemoveMCPServer(ctx, "echo"); err != nil { + return err +} +``` + + + + +## 所有权与会话隔离 + +MCP 能力发现与实时修改分属不同所有者: + +- 智能体全局管理器持有从全局配置加载的服务器。 +- 通过 Rust `SessionOptions` 传入的管理器是该会话继承的只读能力来源。 +- 每个已构建会话都新建一个私有管理器,专门承接实时 `addMcp` 与 + `removeMcpServer` 操作。 + +会话组装会读取继承来源的工具定义,但不会把会话配置合并回这些管理器。本地新增的 +服务器可以在当前会话内遮蔽同名的继承全限定工具;移除本地服务器后,继承工具会 +重新出现。这个操作不会注销、断开或改变全局或宿主持有的来源,同级会话也不会受影响。 + +委派的子智能体会收到有序能力来源,从而调用同一批 MCP 工具,但不会取得所有者身份。 +`session.close()` 只断开私有管理器的服务器;`agent.close()` 和 +`disconnectIdleMcp(...)` 仍负责智能体全局管理器。 + +Rust 中,宿主传入的 MCP 来源需要异步构建会话,使发现过程不会阻塞: + +```rust +let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_mcp(manager)) + .build() + .await?; +``` + +## 闲置断开 + +已通过标准输入输出连接的 MCP 服务器即使处于空闲状态也会占用资源,包括文件描述符 +和一个后台工作进程。在长期运行的集群会话中,这些安静的服务器会越积越多。 +`disconnectIdleMcp`(更新日志 `[3.3.0]` 的 “MCP idle disconnect”)会按需回收它们: +断开最后活动时间早于阈值的全局 MCP 服务器,释放文件描述符和工作进程, +**同时保留每台服务器已经注册的配置**,让后续工具调用可以按需重连。 + +这是一个智能体级方法,只作用于智能体的全局 MCP 管理器,底层由 +`McpManager::disconnect_idle(threshold_ms)` 支持。它返回被断开的服务器名称列表。 + +```ts +// 回收空闲超过 5 分钟的服务器,返回被断开的名称。 +const dropped = await agent.disconnectIdleMcp(5 * 60 * 1000); +console.log('disconnected:', dropped); +``` + +```python +# 回收空闲超过 5 分钟的服务器,返回被断开的名称。 +dropped = agent.disconnect_idle_mcp(5 * 60 * 1000) +print('disconnected:', dropped) +``` + +当前 Go 桥接层提供会话本地的添加、状态、调用和移除,以及智能体级 +`RefreshMCPTools`;它还没有暴露全局闲置连接清扫器。 + +运行成千上万长期会话的宿主应通过清扫器周期性调用它,例如每 60 秒调用一次、阈值设为 +5 分钟。对已断开服务器的后续工具调用会根据保留的配置重新建立连接,无需重新注册。 + +断开操作还会清除孤儿时间戳:只调用过 `touch()`、从未真正连接过的服务器条目,会在 +每次 `disconnect_idle` 调用时被清扫,使活动映射在管理器的长期生命周期内不会无限 +增长(更新日志 `[3.3.0]` 修复了 “MCP timestamp leak”)。 + +## 安全 + +外部工具服务器应视为特权集成,凭据只放在环境变量或宿主密钥管理器中。 diff --git a/website/docs/v8.5.1/zh/guide/memory.mdx b/website/docs/v8.5.1/zh/guide/memory.mdx new file mode 100644 index 00000000..f2ca3b23 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/memory.mdx @@ -0,0 +1,84 @@ +--- +title: '记忆' +description: '会话记忆存储与显式召回' +--- + +# 记忆 + +记忆用于记录可复用事实,帮助驾驭层复用经验。 + +## 默认存储 + +每个会话默认都有记忆存储。普通 SDK 会话若不传入自定义存储,会使用 +`/.a3s/memory` 下的文件存储。`a3s code` TUI 使用和 `/memory` 面板相同的 +`memory_dir`;默认是 `~/.a3s/memory`,所以不同项目会共享这份长期记忆,除非你配置 +项目本地路径。可以通过配置中的 `memory_dir` 移动默认位置,也可以给单个会话 +传入带类型的存储对象来覆盖默认后端。若文件存储初始化失败,会话会降级到 +内存存储,并通过初始化警告暴露该问题。 + +## 大语言模型抽取 + +有可用记忆存储时,默认启用大语言模型记忆抽取。每个已完成且非空的回合都会提交给 +当前模型。是否存在能改变未来回答或行动的信息,由模型判断,而不是由关键词列表、 +工具类型或回合长度启发式规则判断。没有符合条件的内容时,模型必须返回空的 +`items` 数组。工具调用及其结果只作为判断上下文,绝不会被机械地复制进长期记忆。 + +系统只接受持久的 `semantic`(语义)或 `procedural`(过程)记忆,并要求来源、 +重要度、置信度、作用域以及未来价值理由均通过校验。运行时会拒绝格式错误的输出和 +明显的凭据;这些是结构与安全检查,不负责判断语义价值。 + +模型会收到一组有数量上限的相关记忆,并可把给定标识标记为 `supersedes` +(取代)或 `conflicts_with`(冲突)。运行时会校验这些关系、移除已确认被取代的 +条目并保留冲突。流式宿主会在发布最终事件前注册抽取任务,按先进先出顺序处理回合, +并在优雅关闭时等待已经受理的抽取任务。 + +`rememberSuccess` 和 `rememberFailure` 仍是显式 SDK 操作。普通智能体运行时不会 +因为工具结果而自动调用它们。 + +## 存储清理 + +默认存储只会合并规范化后完全重复的内容。标点仍有意义;表述不同但相关的内容会保持 +独立,除非大语言模型通过 `supersedes` 显式合并。存储层不会根据关键词重叠推断 +语义等价。 + +配置自动清理后,系统会删除陈旧且重要度低的记忆;但精选记忆会受到强保护: +固定或受保护的条目、频繁召回的条目、已合并的记忆,以及带 `supersedes` 或 +`conflicts_with` 关系元数据的条目,都不会因为容量上限被直接裁掉。 + +可以在 `config.acl` 中调整抽取上限: + +```acl +memory { + llmExtraction = true + llmExtractionMaxItems = 5 + llmExtractionMaxInputChars = 8000 +} +``` + +## 手动写入与召回 + +```ts +import { FileMemoryStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + memoryStore: new FileMemoryStore('./.a3s/memory'), +}); + +await session.rememberSuccess( + 'release preflight', + ['bash', 'grep'], + 'provider 验证通过', +); +const related = await session.recallSimilar('release provider verification', 5); +const recent = await session.memoryRecent(10); +const tagged = await session.recallByTags(['grep'], 10); +``` + +Memory 是辅助上下文,当前任务的验证证据仍来自当前命令和 trace。 + +## 持久记忆模式 + +生产环境的持久记忆绑定必须使用 **Active recall** +(`DurableMemorySession::active_recall`)。仅候选的 **shadow** 模式 +(`ShadowCandidates` / `::shadow`)已**移除**(`HARNESS-CONV4` / `CAP-GA1`)。 +抽取仍可写入有证据支撑的 Candidate 节点;宿主需显式激活后,recall 才会准入。 diff --git a/website/docs/v8.5.1/zh/guide/multi-machine.mdx b/website/docs/v8.5.1/zh/guide/multi-machine.mdx new file mode 100644 index 00000000..157b6995 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/multi-machine.mdx @@ -0,0 +1,242 @@ +--- +title: '多机器' +description: '通过 AgentExecutor 扩展边界将编排步骤分布到多台机器' +--- + +# 多机器 + +A3S Code 把多智能体编排表达为代码中的一套*语法*(grammar),再把由此产生的步骤 +放到你希望它运行的任何地方。这一划分沿着**框架 / 宿主边界**展开,于 +`[3.4.0]` 引入: + +- **框架**拥有编排语法和可序列化的数据契约。它从不决定步骤在哪里运行。 +- **宿主**拥有放置(placement)、传输和调度——哪个节点执行某个步骤、step 规格如何 + 送达、以及并发如何映射到集群。 + +两者之间唯一的接触点是一个 trait:`AgentExecutor`。 + +## 框架 / 宿主边界 + +框架的契约是两个可序列化类型: + +- `AgentStepSpec`——_运行什么_,与*在哪里运行*无关:`task_id`、`agent`、 + `description`、`prompt`,以及可选的 `max_steps`、`parent_session_id`、 + `output_schema`。 +- `StepOutcome`——运行一个 spec 的结果:`task_id`、`session_id`、`agent`、 + `output`、`success`,以及可选的 `structured`。 + +两者都能干净地序列化,因此宿主可以把一个 spec 送到另一个节点,并把 outcome 持久化 +到 checkpoint。组合这些 spec 的 combinators 完全针对 `AgentExecutor` trait 编写, +从不观察步骤实际在哪里运行——所以同一套编排可以从单进程扩展到集群而无需改动。 + +## `AgentExecutor` 扩展边界 + +`AgentExecutor` 是语法与宿主之间的边界: + +```text +combinators (parallel / pipeline / resumable) + -> AgentExecutor::execute_step(spec) -> StepOutcome + ├─ 内置 TaskExecutor:把步骤作为本地子 agent 运行 + └─ 宿主 executor:把步骤放到远程节点 +``` + +内置的 `TaskExecutor` 把每个步骤作为子 agent 在本地运行——在进程内、基于 Tokio—— +继承会话的 agent 注册表、LLM 客户端、工作区、MCP 工具和 subagent 跟踪器。诸如 +集群运行时这样的宿主用自己的 executor 替换它,把步骤分布到集群;combinators 不受影响。 + +`concurrency_hint()` 是**建议性的,不是本地硬上限**。本地默认返回会话的 +`max_parallel_tasks`;由调度器支撑的宿主可以返回其集群范围的目标值。因为它是 hint +而非 ceiling,编排得以扩展到单进程之外。 + +会话直接暴露内置 seam: + +- `AgentSession::agent_executor()` 返回一个由会话支撑的 `AgentExecutor`。 +- `AgentSession::session_store()` 返回会话的 store(在配置了的情况下),可恢复的 + combinator 需要它来记录进度。 + +下面的 SDK 语法会替你调用 `agent_executor()`;只有在实现或替换自定义 executor 时 +你才需要直接使用它们。 + +## 并行:带屏障的扇出 + +`execute_steps_parallel` 把 `specs` 在 executor 上 fan-out 并等待全部完成(一个 +屏障 / barrier)。结果保持输入顺序,panic 的分支会变成一个失败的 `StepOutcome` +而不会拖垮整批,并且并发受 executor 的 concurrency hint 限制。 + +```ts +const outcomes = await session.parallel([ + { + taskId: 'a', + agent: 'explore', + description: 'survey', + prompt: 'Map the auth module', + }, + { + taskId: 'b', + agent: 'review', + description: 'audit', + prompt: 'Review error handling', + }, +]); + +for (const o of outcomes) { + console.log(o.taskId, o.success, o.output); +} +``` + +```python +outcomes = session.parallel([ + {"task_id": "a", "agent": "explore", "description": "survey", "prompt": "Map the auth module"}, + {"task_id": "b", "agent": "review", "description": "audit", "prompt": "Review error handling"}, +]) + +for o in outcomes: + print(o["task_id"], o["success"], o["output"]) +``` + +## 流水线:逐项串联,阶段之间无屏障 + +`execute_pipeline` 让每个 item 独立地流经一条由 `PipelineStage` 组成的链路。 +**stage 之间没有屏障**——item A 可以处于 stage 3,而 item B 仍在 stage 1——因此 +墙钟时间取决于最慢的单条链路,而不是每个 stage 最慢步骤之和。 + +一个 stage 是 `(ctx) => spec | null` 回调,其中 `ctx` 携带上一步的 outcome 和原始 +item。返回一个 spec 以运行下一步,或返回 `null` 提前停止该 item 的链路;当某步失败 +时链路也会停止(后续 stage 只会建立在失败结果之上)。这些桥接**fail closed**:一个 +挂起、返回 `null` 或抛出的 stage 只会停止它自己的链路。 + +> Node 的 stage 回调**绝不能 throw**——把逻辑包在 `try`/`catch` 中并在出错时返回 +> `null`(与 `setBudgetGuard` 相同的约束)。挂起超过超时的 stage 会对该链路 fail +> closed。Python 的 stage 若 raise 会被捕获并当作 `null` 处理。 + +```ts +const outcomes = await session.pipeline( + ['src/auth', 'src/api'], + [ + (ctx) => ({ + taskId: 's1', + agent: 'explore', + description: 'survey', + prompt: `Survey ${ctx.item}`, + }), + (ctx) => + ctx.previous?.success + ? { + taskId: 's2', + agent: 'review', + description: 'review', + prompt: `Review: ${ctx.previous.output}`, + } + : null, + ], +); +``` + +```python +def survey(ctx): + return {"task_id": "s1", "agent": "explore", "description": "survey", + "prompt": f"Survey {ctx['item']}"} + +def review(ctx): + prev = ctx["previous"] + if prev and prev["success"]: + return {"task_id": "s2", "agent": "review", "description": "review", + "prompt": f"Review: {prev['output']}"} + return None + +outcomes = session.pipeline(["src/auth", "src/api"], [survey, review]) +``` + +## 可恢复执行:跨节点恢复 + +`execute_steps_parallel_resumable` 是 `parallel` 加上一份日志。在每个步骤边界,它把 +一个 `WorkflowCheckpoint` 以 `workflowId` 为键写入 `SessionStore`。恢复时它跳过已 +完成的步骤,重新派发其余步骤。它**只记录成功的步骤**,因此失败的步骤会在恢复时重试 +——它的效果尚未完成。 + +需要一个 `SessionStore`。当未配置时,Node 桥接以 +`"parallelResumable requires a sessionStore"` 拒绝;Python 抛出等价异常。 + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +const outcomes = await session.parallelResumable( + [ + { + taskId: 'a', + agent: 'explore', + description: 'survey', + prompt: 'Map the auth module', + }, + { + taskId: 'b', + agent: 'review', + description: 'audit', + prompt: 'Review error handling', + }, + ], + 'release-audit', +); +``` + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session("/repo", opts) + +outcomes = session.parallel_resumable([ + {"task_id": "a", "agent": "explore", "description": "survey", "prompt": "Map the auth module"}, + {"task_id": "b", "agent": "review", "description": "audit", "prompt": "Review error handling"}, +], "release-audit") +``` + +因为 checkpoint 是可序列化的,且 executor 是一个参数,宿主可以通过传入另一个节点的 +executor,在**不同的节点**上恢复被中断的工作流——框架迁移*运行什么*,宿主提供 +_在哪里运行_。 + +## 模式约束的步骤输出 + +携带 `output_schema`(Node 中为 `outputSchema`)的 spec 必须返回一个符合该 JSON +Schema 的值;经校验的对象落入 `StepOutcome.structured`。executor 复用结构化输出的 +coercion 与 repair 机制。coercion 失败会**把该步骤降级为不成功**,因此调用方永远不会 +把未经校验的文本当作所承诺的对象。 + +```ts +const [finding] = await session.parallel([ + { + taskId: 'classify', + agent: 'review', + description: 'classify', + prompt: 'Classify this defect', + outputSchema: { + type: 'object', + properties: { severity: { type: 'string' }, summary: { type: 'string' } }, + required: ['severity', 'summary'], + }, + }, +]); + +if (finding.success && finding.structured) { + console.log(finding.structured.severity); +} +``` + +## 跨机器放置编排步骤 + +lane queue 仍然是一种有效的传输——但它现在是**`AgentExecutor` seam 背后的一个选项, +而不是唯一的集成点**。要分布工作,宿主针对任何适合其平台的传输(HTTP、消息队列、 +任务系统、lane queue、内部 RPC)实现 `AgentExecutor::execute_step`,把 +`concurrency_hint()` 接到其集群目标,再把该 executor 交给 combinators。语法—— +parallel、pipeline、resumable——保持不变;移动的只有放置。 + +结果是:coordinator 会话拥有对话、最终合成和发布决策,而它的编排步骤在宿主放置的 +任何地方执行,跨节点恢复由可序列化的 checkpoint 承载。 + +参见 [Orchestration](/guide/orchestration) 深入了解 combinator 语法,以及 +[Persistence](/guide/persistence) 了解 checkpoint store。 diff --git a/website/docs/v8.5.1/zh/guide/orchestration.mdx b/website/docs/v8.5.1/zh/guide/orchestration.mdx new file mode 100644 index 00000000..e65eb481 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/orchestration.mdx @@ -0,0 +1,438 @@ +--- +title: '编排' +description: '可编程、确定性的多智能体编排:并行扇出、无屏障流水线与可恢复工作流' +--- + +# 编排 + +编排是**高级**宿主编排面——模型驱动委派的*可编程*姊妹能力。日常编码 Agent 扇出请优先使用统一的 [`task`](/guide/tasks)。在 [Tasks](/guide/tasks) 与 +[Teams](/guide/teams) 中,由 LLM 在运行时决定是否调用含一个或多个 `tasks[]` 项的 +`task`——扇出的形状取决于模型的选择。编排把这个决策搬进你的代码:你 +用一套语法表达扇出、流水线与验证面板,因此形状是可复现、可测试、受预算约束 +且可恢复的——与模型的临场选择无关。 + +当工作的*结构*是宿主提前已知的(并行跑这三个 reviewer;让每个候选项依次经过 +explore → verify → review;崩溃后恢复这一批),就用编排。当你希望由模型决定是否 +以及如何委派时,就用 Tasks/Teams。 + +## 框架 / 宿主边界(接缝) + +本层的一切都围绕单一接缝 `AgentExecutor` 编写:"运行这个 step,把结果给我。"这 +条接缝把职责切得很干净: + +- **框架**拥有*语法*——有哪些 step、如何组合、并发*提示*,以及可序列化契约 + `AgentStepSpec` / `StepOutcome`。 +- **宿主**拥有*放置*——传输、调度,以及 step 实际在哪里运行。 + +内置的默认 executor(`TaskExecutor`)在本地、进程内、基于 tokio 运行每个 step。 +宿主可以替换为自己的 `AgentExecutor`,把 step 放置到集群各处;组合子从不观察 step +在哪里运行,因此同一套编排无需改动即可从单进程扩展到集群。 + +`concurrency_hint()` 是**建议性**的,而非硬性本地上限——正是它让编排能扩展到单进 +程之外(由调度器支撑的宿主会返回其集群级目标,而不是本地上限)。 + +SDK 已为你接好这一切:`AgentSession::agent_executor()` 返回由 session 支撑的 +executor(它把每个 step 作为子 agent 在本节点运行,继承 session 的 agent registry、 +LLM client、workspace 与 MCP 工具),`session_store()` 返回 session 的 store。 +`parallel` / `pipeline` / `parallelResumable` 方法会替你调用它们。 + +## 步骤契约 + +一个 step 由 `AgentStepSpec` 描述,并解析为 `StepOutcome`。两者都刻意保持可序列 +化:宿主可以把 spec 发送到另一个节点,可恢复组合子也会把 outcome 持久化进 +checkpoint。 + +`AgentStepSpec` 字段: + +- `task_id`——step 的稳定 id(由你指定);会流入生命周期事件与 checkpoint。 +- `agent`——要运行的 agent 的 registry key(例如 `explore`、`review`)。 +- `description`——用于展示/追踪的简短人类标签。 +- `prompt`——交给子 agent 的指令。 +- `max_steps`_(可选)_——每个 step 的 tool-round 上限。 +- `parent_session_id`_(可选)_——用于事件关联的父 session id。 +- `output_schema`_(可选)_——设置后,步骤必须返回符合此 JSON Schema 的值(见 + [强制模式约束的步骤输出](#schema-forced-step-output))。 + +`StepOutcome` 字段: + +- `task_id`——产生该结果的 step id。 +- `session_id`——子运行的 session id(失败的 step 仍可寻址)。 +- `agent`——运行的 agent。 +- `output`——step 的文本输出。 +- `success`——失败或 panic 的 step 为 `false`(绝不会丢弃兄弟 step)。 +- `structured`_(可选)_——经模式校验的对象,仅当任务规格携带 `output_schema` + 时存在。 + +key 的大小写风格因 SDK 而异: + +| 概念 | Node(camelCase) | Python(snake_case) | +| --------------- | ----------------- | -------------------- | +| step id | `taskId` | `task_id` | +| tool-round 上限 | `maxSteps` | `max_steps` | +| 父 session | `parentSessionId` | `parent_session_id` | +| 强制模式约束 | `outputSchema` | `output_schema` | + +`agent`、`description`、`prompt`、`output`、`success` 与 `structured` 在两个 SDK +中拼写相同。 + +## `parallel`:带屏障的扇出 [#parallel--barrier-fan-out] + +`session.parallel(specs)` 把每个任务规格作为扇出分支运行,并按**输入顺序**解析出每 +个 spec 对应的一个 `StepOutcome`。它映射到核心组合子 `execute_steps_parallel`。它 +是一个**屏障**:在返回前会等待每个 step。 + +每个分支彼此隔离——失败*或 panic* 的 step 会变成 `success: false`,绝不会丢弃兄弟 +step。并发受 executor 的并发提示约束(默认即 session 配置的并行度)。 + +```ts +const outcomes = await session.parallel([ + { + taskId: 'explore', + agent: 'explore', + description: 'Risky changes', + prompt: 'Find risky changed files in this diff.', + }, + { + taskId: 'verify', + agent: 'verification', + description: 'Test gaps', + prompt: 'Identify missing or weak verification.', + }, + { + taskId: 'review', + agent: 'review', + description: 'Correctness', + prompt: 'Review the diff for correctness risks.', + }, +]); + +for (const outcome of outcomes) { + if (outcome.success) { + console.log(outcome.taskId, outcome.output); + } else { + console.warn('failed:', outcome.taskId, outcome.output); + } +} +``` + +```python +outcomes = session.parallel([ + {"task_id": "explore", "agent": "explore", "description": "Risky changes", + "prompt": "Find risky changed files in this diff."}, + {"task_id": "verify", "agent": "verification", "description": "Test gaps", + "prompt": "Identify missing or weak verification."}, + {"task_id": "review", "agent": "review", "description": "Correctness", + "prompt": "Review the diff for correctness risks."}, +]) + +for outcome in outcomes: + if outcome["success"]: + print(outcome["task_id"], outcome["output"]) + else: + print("failed:", outcome["task_id"], outcome["output"]) +``` + +## `pipeline`:阶段之间无屏障 + +`session.pipeline(items, stages)` 让每个 item **独立地**流经一连串阶段——阶段之间 +没有屏障,因此 item A 可以处于 stage 3,而 item B 还在 stage 1。墙钟时间是最慢的 +_单条链_,而不是逐阶段屏障会带来的"每阶段最慢之和"。 + +阶段是 spec 构造器,而非 spec:每个阶段接收上一个 outcome 和原始 item,返回要运行 +的下一个 step,或返回 `null` / `None` 提前停止该 item 的链。失败的 step 同样会停止 +链(后续阶段只会基于失败结果继续)。回调形状: + +- Node:`(ctx) => spec | null`,其中 `ctx = { previous: StepOutcome | null, item }` +- Python:`stage(ctx) -> spec | None`,其中 `ctx = {"previous": , "item": }` + +阶段可以基于上一个 outcome 分支——例如"验证 review 阶段产出的发现"。 + +约束(来自源码): + +- 流水线阶段**不支持**逐阶段 `output_schema`——需要模式校验的步骤请用 + [`parallel`](#parallel--barrier-fan-out)。 +- **Node:** 阶段回调**绝不能 throw**——一次 throw 会中止进程(与 `setBudgetGuard` + 相同的约束)。请把逻辑包进 `try/catch`,出错时 `return null`。 +- **Node:** 超过 `timeoutMs`(第 3 个参数,默认 `30000`)仍挂起的阶段会 fail + closed——被当作 `null` 处理,仅停止该条链。 +- **Python:** 抛异常的阶段 callable 会被捕获并当作 `None`(仅停止该条链)。 + +> Node pipeline 阶段回调**绝不能 throw。** 在当前 napi 版本中,返回值转换时的 JS +> throw 会中止进程(与 `setBudgetGuard` 相同的 fail-closed 约束)。务必把阶段逻辑 +> 包进 `try/catch` 并在出错时 `return null`。 + +```ts +const outcomes = await session.pipeline( + ['src/auth.ts', 'src/payments.ts'], + [ + (ctx) => ({ + taskId: `explore-${ctx.item}`, + agent: 'explore', + description: 'Inspect file', + prompt: `Summarize the responsibilities and risks of ${ctx.item}.`, + }), + (ctx) => { + try { + if (!ctx.previous) return null; + return { + taskId: `review-${ctx.item}`, + agent: 'review', + description: 'Review of prior finding', + prompt: `Review this summary for correctness risks:\n${ctx.previous.output}`, + }; + } catch { + return null; // stages must not throw + } + }, + ], +); +``` + +```python +def explore_stage(ctx): + item = ctx["item"] + return { + "task_id": f"explore-{item}", + "agent": "explore", + "description": "Inspect file", + "prompt": f"Summarize the responsibilities and risks of {item}.", + } + +def review_stage(ctx): + prev = ctx["previous"] + if prev is None: + return None + item = ctx["item"] + return { + "task_id": f"review-{item}", + "agent": "review", + "description": "Review of prior finding", + "prompt": f"Review this summary for correctness risks:\n{prev['output']}", + } + +outcomes = session.pipeline( + ["src/auth.ts", "src/payments.ts"], + [explore_stage, review_stage], +) +``` + +## 可恢复 / 可迁移工作流 + +`session.parallelResumable(specs, workflowId)`(Node)/ +`session.parallel_resumable(specs, workflow_id)`(Python)是 `parallel` 加上一份 +日志。它映射到 `execute_steps_parallel_resumable`。 + +它在每个 step 边界把一份 `WorkflowCheckpoint` 写入 session store。恢复时它跳过已完 +成的 step(复用其缓存的 outcome),只重新派发其余 step。它**只记录成功的 step**—— +失败的 step 不入日志,因此恢复时会重试。完全成功后 checkpoint 会被删除;只有崩溃才 +会留下一份供恢复。 + +由于 checkpoint 可序列化、executor 是一个参数,宿主可以通过传入另一个节点的 +executor,在**另一个节点**上恢复被中断的工作流(迁移)。 + +该组合子**要求已配置 session store**——两个 SDK 方法在缺少 store 时都会 reject/ +raise(Node 的错误信息是 `parallelResumable requires a sessionStore`)。 + +`WorkflowCheckpoint` 的模式字段为 `schema_version` / `workflow_id` / `steps` / +`checkpoint_ms`。由*未来的*、不兼容的 `schema_version` 写入的检查点在加载时 +会被拒绝(`ensure_loadable`)。该失败是 fail-safe 而非致命的:不可读的 checkpoint +会记录一条 warning,工作流从头重跑,而不是从它无法解释的状态恢复。 + +store 相关见 [Persistence](/guide/persistence),迁移路径见 +[Multi-Machine](/guide/multi-machine)。 + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); + +// 第一次尝试可能会在中途被打断。 +let outcomes = await session.parallelResumable(specs, 'release-batch-42'); + +// 崩溃或重启后:使用同一个 workflowId 恢复,并跳过已完成的步骤。 +outcomes = await session.parallelResumable(specs, 'release-batch-42'); +``` + +```python +from a3s_code import Agent, FileSessionStore, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.session("/repo", opts) + +# 第一次尝试可能会在中途被打断。 +outcomes = session.parallel_resumable(specs, "release-batch-42") + +# 崩溃或重启后:使用同一个 workflow_id 恢复,并跳过已完成的步骤。 +outcomes = session.parallel_resumable(specs, "release-batch-42") +``` + +## 跨扇出任务的共享预算 [#shared-budget] + +默认情况下,每个子代理各自统计自己的 LLM 成本。给 `parallel` 传入一个 token +预算后,所有子智能体改为汇入**同一个账本**——为整个扇出设一个统一上限。它映射 +到核心的 `WorkflowBudget`:一个安装到每个子运行上的、聚合型 `BudgetGuard`。 + +预算是一个**可选参数**,因此向后兼容: + +- 不传预算时,`parallel(specs)` 返回原来的结果数组,与之前完全一致。 +- 传入预算时,`parallel(specs, budgetTokens)` 解析为 `{ outcomes, budget }`, + 其中 `budget` 是账本快照(`consumedTokens` / `limitTokens`)。 + +一旦达到上限,之后*启动*的 step 会被拒绝——其结果为 `success: false`,并带有预算 +耗尽信息。它是一个**软上限**:由于用量是在每次 LLM 调用*之后*记账的,宽扇出 +可能在账本追上之前冲过上限几个在飞回合。框架**绝不**强杀进行中的扇出;预算耗尽 +只是拒绝*下一次* LLM 调用。 + +```ts +// 不传预算 → 原样返回结果数组(行为不变)。 +const outcomes = await session.parallel(specs); + +// 传入预算 → { outcomes, budget }。所有子代理共享一个账本。 +const { outcomes: out, budget } = await session.parallel(specs, 500_000); +console.log(budget.consumedTokens, budget.limitTokens); // 例如 48213, 500000 +// TS 返回类型是联合类型:Array | { outcomes, budget }。 +``` + +```python +# 不传预算 → 原样返回列表(行为不变)。 +outcomes = session.parallel(specs) + +# 传入预算 → {"outcomes": [...], "budget": {"consumed_tokens", "limit_tokens"}}。 +res = session.parallel(specs, budget_tokens=500_000) +print(res["budget"]["consumed_tokens"], res["budget"]["limit_tokens"]) +``` + +## 循环直到完成(`execute_loop`)[#looping] + +对于长度未知、需要迭代至收敛的工作(循环直至没有剩余项、反复打磨直到满意),核心语法 +提供了 `execute_loop`。每一轮都是一个屏障(`execute_steps_parallel`);宿主提供的 +谓词看到本轮的结果,返回 `LoopDecision::Continue(next_specs)` 或 +`LoopDecision::Stop`。必填的 `max_iterations` 是一个**硬上限**——一旦达到,即使谓词 +还想继续也会停止,从而让 LLM 驱动的循环永远不会失控。 + +```rust +use a3s_code_core::orchestration::{execute_loop, AgentStepSpec, LoopDecision}; + +let outcomes = execute_loop(executor, initial_specs, /* max_iterations */ 5, None, |round| { + // 本轮没有新发现就停止;否则扇出后续 step。 + let follow_ups = derive_follow_ups(round); + if follow_ups.is_empty() { + LoopDecision::Stop + } else { + LoopDecision::Continue(follow_ups) + } +}) +.await; +``` + +> 从宿主 SDK 你并不需要专门的 `loop` 动词——直接用你自己语言里的 `while`/`for` +> 围绕 `parallel` 写循环,根据结果决定下一轮即可。`execute_loop` 是为 Rust 语法 +> 而存在,并为循环提供一个单一、强制的终止守卫。 + +## 工作流外观接口(Rust / 嵌入)[#workflow-facade] + +`session.workflow()` 返回一个可廉价克隆的 `Workflow`,它预先接好了会话的 +executor、持久化 store、逐 step 事件流,以及一个稳定的、由会话派生的 root id。它是 +把以上能力打包起来的可编程句柄;控制流就是普通 Rust——`await` 一个动词、查看结果、 +决定下一步运行什么。 + +- **动词**——`agent`(单步)、`parallel`(屏障式扇出)、`phase`(*命名*的、 + 可恢复的屏障,并发出里程碑)、`pipeline`(按条目的链),以及不会失败的 `log`。 + 每个动词都只委派给一个 combinator。 +- **Phase 与事件**——`phase(name, specs)` 派生确定性 checkpoint id + (`{root}/{index}:{name}`),在配置了 store 时走可恢复屏障,并在一个广播上发出 + `WorkflowEvent::PhaseStart` / `PhaseEnd`,你可用 `subscribe()` 读取。`log()` + 发出 `WorkflowEvent::Log`。 +- **预算**——`session.workflow_with_token_budget(Some(limit))` 安装一个共享的 + `WorkflowBudget`;`budget_snapshot()` 读取账本,达到上限时会触发 + `WorkflowEvent::BudgetExhausted`。 + +```rust +let wf = session.workflow(); // 或 session.workflow_with_token_budget(Some(500_000)) +let mut events = wf.subscribe(); + +// 先跑一步,再根据其结果计算出一个*可变*数量的扇出——这正是“动态”所在: +// 形状在运行时决定,而非提前声明。 +let plan = wf.agent(AgentStepSpec::new("plan", "plan", "plan", goal)).await; +let specs = derive_specs(&plan); // 你的代码 +let done = wf.phase("implement", specs).await; // 可恢复屏障 + 里程碑 +let reviews = wf.phase("review", to_review(&done)).await; // 预算在各 phase 间共享 + +if let Some(b) = wf.budget_snapshot() { + println!("spent {} / {:?} tokens", b.consumed_tokens, b.limit_tokens); +} +``` + +SDK 暴露的是扁平的 `parallel` / `pipeline` / `parallelResumable` 动词(以及上面 +`parallel` 的预算重载);完整的 `Workflow` 句柄——phases、事件订阅、loop +combinator——属于 Rust / 嵌入层 API。 + +## 受模式约束的步骤输出 [#schema-forced-step-output] + +携带 `output_schema`(Node 中为 `outputSchema`)的任务规格会强制步骤返回符合该 JSON +Schema 的值;经校验的对象落在 `StepOutcome.structured` 中。这复用了与 A3S Code 其余 +部分相同的结构化输出强转 + 修复机制。强转失败会把该 step **降级为不成功** +(`success: false`),因此调用方绝不会把未经校验的文本当作承诺的对象。 + +强制模式约束仅适用于 `parallel` / `parallelResumable` 的任务规格——**不**适用于 +pipeline 阶段。 + +```ts +const [outcome] = await session.parallel([ + { + taskId: 'triage', + agent: 'review', + description: 'Structured triage', + prompt: 'Triage this diff.', + outputSchema: { + type: 'object', + properties: { + severity: { type: 'string', enum: ['low', 'medium', 'high'] }, + summary: { type: 'string' }, + }, + required: ['severity', 'summary'], + }, + }, +]); + +if (outcome.success) { + console.log(outcome.structured.severity, outcome.structured.summary); +} +``` + +```python +outcomes = session.parallel([ + { + "task_id": "triage", + "agent": "review", + "description": "Structured triage", + "prompt": "Triage this diff.", + "output_schema": { + "type": "object", + "properties": { + "severity": {"type": "string", "enum": ["low", "medium", "high"]}, + "summary": {"type": "string"}, + }, + "required": ["severity", "summary"], + }, + }, +]) + +outcome = outcomes[0] +if outcome["success"]: + print(outcome["structured"]["severity"], outcome["structured"]["summary"]) +``` + +## 成本治理与生命周期 + +编排 step 走的是同一个 session,因此 session 的各项控制对它们直接生效。 +`setBudgetGuard`(Node)/ `budget_guard`(Python)约束每个 step 的 LLM 成本; +`close()` 会连同 session 的其余工作一起取消进行中的 step;宿主提供的身份标签 +(`tenant_id`、`principal`、`agent_template_id`、`correlation_id`)会贯穿每个 step, +供宿主侧聚合与计费使用。这些控制的细节见 [Sessions](/guide/sessions) 与 +[Limits](/guide/limits)。 diff --git a/website/docs/v8.5.1/zh/guide/persistence.mdx b/website/docs/v8.5.1/zh/guide/persistence.mdx new file mode 100644 index 00000000..10111efb --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/persistence.mdx @@ -0,0 +1,277 @@ +--- +title: '持久化' +description: '保存与恢复会话' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 持久化 + +持久化让会话可以跨进程恢复,也让产品界面拥有稳定的会话标识。 + +恢复后的会话可以重新填充任务列表、执行记录、制品和交付摘要,不必重放已经完成的运行。 + +A3S Code 会持久化三种相关但不同的对象: + +| 对象 | 写入方 | 恢复入口 | 用途 | +| ------------------- | -------------------------------- | -------------------------------------- | -------------------------------------------------------------------------------------------- | +| `SessionSnapshotV1` | `session.save()` 或 `autoSave` | `agent.resumeSession(id, options)` | 恢复一个带版本号的完整代次,其中包含对话、制品、追踪、运行记录、验证报告与子智能体任务快照。 | +| 循环检查点 | 智能体循环运行期间 | `session.resumeRun(runId)` | 从上一个完成的工具回合边界继续被中断的运行。进程内正常完成的运行会删除这个检查点。 | +| 工作流检查点 | `parallelResumable` / 工作流阶段 | `parallelResumable(specs, workflowId)` | 进程重启后跳过已经完成的编排步骤。 | + +## 文件会话存储 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +let options = SessionOptions::new() + .with_session_id("release-review") + .with_file_session_store("./.a3s/sessions") + .with_auto_save(true); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; + +session.send("审查发布就绪情况", None).await?; +session.save().await?; +``` + + + + +```ts title=persistence.ts +import { FileSessionStore } from '@a3s-lab/code'; + +const session = agent.session('/repo', { + // !mark(1:3) 快照配置 + sessionId: 'release-review', + sessionStore: new FileSessionStore('./.a3s/sessions'), + autoSave: true, +}); + +await session.send('检查发布就绪情况'); +await session.save(); +``` + + + + +```python +from a3s_code import FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_id = 'release-review' +opts.session_store = FileSessionStore('./.a3s/sessions') +opts.auto_save = True + +session = agent.session('/repo', opts) +session.send('检查发布就绪情况') +session.save() +``` + + + + +```go +options := &code.SessionOptions{ + SessionID: "release-review", + FileSessionStoreDir: ".a3s/sessions", + AutoSave: code.Ptr(true), +} +session, err := agent.Session(ctx, "/repo", options) +if err != nil { + return err +} +if _, err := session.Run(ctx, "检查发布就绪情况"); err != nil { + return err +} +if err := session.Save(ctx); err != nil { + return err +} +``` + + + + +Go 通过 `FileSessionStoreDir` 选择内置文件存储;自定义 `SessionStore` trait +实现仍属于 Rust 嵌入能力。 + +## 原子快照代次 + +`session.save()` 会把当前持久化状态收集成一个 `SessionSnapshotV1`,并且只调用一次 +`SessionStore::save_snapshot`。信封包含: + +- `schema_version` 与 `SessionData` +- 工具制品 +- 追踪事件与运行记录 +- 验证报告 +- 委派的子智能体任务快照 + +文件存储会把完整 JSON 信封写入并同步临时文件,然后原子替换 +`.json`。因此读取方看到的是上一代或下一代,不会读到“新对话搭配旧 +运行或追踪分片”的组合。内存存储在同一把锁下发布同一个聚合快照。两者都报告 +`SessionStoreCapabilities { atomic_session_snapshots: true }`。 + +历史文件仍可读取。裸 `SessionData` 会在加载时与旧制品、追踪、运行、验证和 +子智能体分片位置合并,再通过 v1 内存形状恢复。保存新聚合快照后,单个信封会成为 +权威代次。已经具有聚合外形、但模式损坏或版本不支持的文档会直接被拒绝,不会重新 +解释成旧式数据。 + +自定义存储必须显式实现 `save_snapshot`。默认实现返回错误,不会把聚合快照拆成 +多次独立写入,也不会把空操作当成成功。默认 `load_snapshot` 只用于尽力组装旧式 +数据;宿主可以通过 `capabilities()` 区分这种行为与原子后端。 + +## 可协商的耐久性(v8.5.1) + +`SessionStoreCapabilities` 现在也会声明可选的 KRN-6 保证。内置适配器仅在真正 +兑现语义时置位;宿主必须先协商再依赖: + +| 能力 | 含义 | +| ----------------------------- | --------------------------------------------------------------------------------------- | +| `aggregate_cas` | `save_snapshot_cas` 仅在期望 digest 匹配(或省略)时提交。 | +| `append_only_event_log` | 文件存储在每次原子替换前后追加仅含 digest 的 Intent/Committed WAL 记录。 | +| `lease_fencing` | `acquire_writer_lease` 发布耐久 epoch;接管后旧持有者在提交时 fail-closed。 | +| `encrypted_at_rest` | `FileSessionStore::with_encryption_key` 用 AES-256-GCM 密封文档;错误密钥 fail-closed。 | +| `watch` | `watch_commits` 在耐久提交后投递仅含 digest 的 `SessionStoreCommitEventV1`。 | +| `reference_aware_artifact_gc` | Artifact 保留会钉住宿主 URI 根,避免限额驱逐删除仍可达内容。 | + +即使启用静态加密,仅含 digest 的 WAL 也保持未加密。memory / file 适配器继续声明 +它们已经兑现的原子快照;新标志默认不声明,直到完成配置。 + +## 恢复 + + + + +```rust +use a3s_code_core::SessionOptions; + +let resumed = agent + .resume_session_async( + "release-review", + SessionOptions::new().with_file_session_store("./.a3s/sessions"), + ) + .await?; +``` + + + + +```ts +const resumed = agent.resumeSession('release-review', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + + + + +```python +opts = SessionOptions() +opts.session_store = FileSessionStore('./.a3s/sessions') +resumed = agent.resume_session('release-review', opts) +``` + + + + +```go +resumed, err := agent.ResumeSession(ctx, "release-review", &code.SessionOptions{ + FileSessionStoreDir: ".a3s/sessions", +}) +``` + + + + +`resumeSession` 恢复的是已保存的会话快照。它不同于 `resumeRun`:用户继续一个 +已保存对话时用 `resumeSession`;只有存在中断运行的检查点时,才用 +`resumeRun(runId)`。恢复操作会在恢复任何历史或运行时证据前先校验快照模式。 + +## 记忆与会话 + +会话持久化保存对话和可回放证据;记忆保存可复用任务事实。当你需要既可 +恢复、又能从重复任务中学习的工作流时,两者一起使用。 + +## 循环检查点与运行恢复 + +配置了 `SessionStore` 后,智能体循环会在每个完成的工具回合之后持久化一个 +`LoopCheckpoint`。边界策略很严格:检查点**只**在工具回合之间产生,绝不在工具 +执行中途产生。如果进程在某个工具执行时崩溃,那一轮的工作会在恢复时丢失,由 +大语言模型从上一个检查点重新推演——在错误边界一侧重跑非幂等工具(写入、命令行) +比让模型重新思考更糟。 + +`session.resumeRun(runId)`(Node)/ `session.resume_run(run_id)`(Python)/ +`session.ResumeRun(ctx, runID)`(Go)—— +对应核心的 `AgentSession::resume_run(checkpoint_run_id)`——会加载该运行标识下 +最新的检查点,并从最后一个边界回放循环。由于检查点存放在共享存储中,恢复可以 +发生在**任何**共享该存储的节点上。累计计量会延续而不是从零重启:`total_usage` +和 `tool_calls_count` 从检查点继续累加。恢复出的工作会分配新的运行标识;新旧运行 +的关系由宿主通过元数据表达,框架不予解释。 + +已正常完成的运行不通过 `resumeRun` 恢复;它们的最终状态应通过 `runs()`、 +`runEvents(runId)`、制品、验证报告和会话快照查看。 + +```ts +const result = await session.resumeRun('run-abc123'); +console.log(result.totalTokens); +``` + +```python +result = session.resume_run('run-abc123') +print(result.total_tokens) +``` + +```go +result, err := session.ResumeRun(ctx, "run-abc123") +if err != nil { + return err +} +fmt.Println(result.Usage.TotalTokens) +``` + +Go 也通过 `ParallelResumable(ctx, specs, workflowID)` 暴露工作流检查点编排。 + +当会话没有配置 `sessionStore`(或给定标识下不存在检查点)时, +`resume_run` 会拒绝。`SessionStore` 新增了 `save_loop_checkpoint` / +`load_loop_checkpoint` / `delete_loop_checkpoint`;文件存储采用崩溃安全的原子 +写入。`LoopCheckpoint::ensure_loadable()` 在反序列化之后立即被调用,会拒绝 +来自未来且不兼容的检查点模式版本,因此 `resume_run` 和实时运行的接收端都不会 +对无法读取的检查点采取行动。 + +参见 CHANGELOG `[3.3.0]`——"Loop checkpoints + run resumption"——以及 +`[3.4.0]` 的 "LoopCheckpoint::ensure_loadable()"。 + +## 工作流检查点 + +`WorkflowCheckpoint` 是工具回合 `LoopCheckpoint` 在上一层的步骤边界对应物: +它把已完成的编排步骤记入日志,使被中断的工作流从最长的已完成前缀恢复。它的 +字段是 `schema_version`、`workflow_id`、`steps` 和 `checkpoint_ms`,模式由 +`WORKFLOW_CHECKPOINT_SCHEMA_VERSION` 常量固定。恢复的运行会跳过已记录的步骤, +只重新派发其余的。 + +`SessionStore` 新增了 `save_workflow_checkpoint` / `load_workflow_checkpoint` / +`delete_workflow_checkpoint`(默认为空操作;文件存储采用崩溃安全的原子写入)。 +来自未来且不兼容的模式版本会在加载时通过 +`WorkflowCheckpoint::ensure_loadable()` 被拒绝。 + +这与编排语法配套使用——参见 +[编排](/guide/orchestration)和 +[多机执行](/guide/multi-machine)。 + +参见 CHANGELOG `[3.4.0]`——"WorkflowCheckpoint"。 + +## 在其他节点恢复 + +两种检查点类型都是可序列化的。配合共享的 `SessionStore` 和可插拔执行器,宿主 +就能在与启动节点**不同**的节点上恢复被中断的运行或工作流——框架持有可序列化 +契约,宿主负责放置与传输。 + +## 运维注意事项 + +包含私有提示词、工具输出或路径的存储不应公开提交。 diff --git a/website/docs/v8.5.1/zh/guide/providers.mdx b/website/docs/v8.5.1/zh/guide/providers.mdx new file mode 100644 index 00000000..ab98b9a1 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/providers.mdx @@ -0,0 +1,119 @@ +--- +title: '模型服务提供商配置' +description: 'ACL 模型服务提供商配置与环境变量注入' +--- + +# 模型服务提供商配置 + +A3S Code 从 ACL 读取运行时配置。配置源可以是 `.acl` 文件路径,也可以是 +内联 ACL 字符串。JSON 和旧 HCL 配置不属于当前配置面。 + +## 基本结构 + +```acl +default_model = "provider/model-id" +max_parallel_tasks = 4 +auto_parallel = false + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + name = "Human readable model name" + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} + +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +`apiKey` 与 `api_key` 都可用,`baseUrl` 与 `base_url` 都可用。仓库里只提交模板 +和非敏感默认值;真实 API Key、私有端点和账号相关模型名应放在环境变量或密钥管理系统中。 + +## 服务提供商类型 + +内置工厂覆盖三条路径: + +| 服务提供商名称 | 客户端路径 | 说明 | +| ---------------------------- | ------------------- | ---------------------------------------------- | +| `anthropic` / `claude` | Anthropic 客户端 | 使用配置中的模型标识和可选服务提供商基础 URL。 | +| `openai` / `gpt` | OpenAI 兼容客户端 | 用于兼容 OpenAI Chat Completions 的端点。 | +| `glm` / `zhipu` / `bigmodel` | 智谱兼容客户端 | 密钥和基础 URL 仍应从环境变量注入。 | +| 其它服务提供商名称 | OpenAI 兼容兜底路径 | 适合私有或自托管的 OpenAI 兼容服务。 | + +运行时不硬编码模型名。`default_model` 和每个会话的 `model` 都是 +`provider/model-id` 形式的标识符,必须能匹配你在服务提供商配置块里声明的模型。 + +## 委派控制 + +```acl +agent_dirs = ["./.a3s/agents"] + +auto_delegation { + enabled = true + auto_parallel = false + allow_manual_delegation = true + min_confidence = 0.72 + max_tasks = 4 +} +``` + +`max_parallel_tasks` 限制有边界的同级任务扇出。`auto_delegation.enabled` +开启自动子智能体委派。顶层 `auto_parallel = false` 会覆盖 +`auto_delegation.auto_parallel`,只关闭自动并行子智能体扇出;手动 `task` 扇出仍可用。 +设置 `allow_manual_delegation = false` 时,模型可见的 `task` 不会注册。 + +## 存储 + +```acl +storage_backend = "memory" +storage_backend = "file" +sessions_dir = ".a3s/sessions" +``` + +短测试使用内存;通过 ACL 加载的可恢复本地会话使用 +`storage_backend = "file"` 和 `sessions_dir`。`storage_url` 会被解析为自定义 +存储元数据,但它本身不会创建文件型会话存储。SDK 宿主也可以直接传 +`sessionStore: new FileSessionStore(...)` / +`opts.session_store = FileSessionStore(...)`。 + +## 私有服务提供商检查 + +真实服务提供商的冒烟测试应通过 `A3S_CONFIG_FILE` 指向本地且被 Git 忽略的 ACL 文件。 +不要把 provider 值复制进命令、日志、文档、PR 或已提交 fixture。 + +```bash +A3S_CONFIG_FILE=/path/to/local/config.acl \ + scripts/real_config_env_integration.sh +``` + +SDK 对齐由单独的真实 provider 检查脚本覆盖: + +```bash +A3S_CONFIG_FILE=/path/to/local/config.acl \ + scripts/sdk_real_config_env_integration.sh +``` + +这两个 Wrapper 只会重写 `providers "openai"` Block 中的凭据。若配置使用 +`providers "deepseek"` 等原生 OpenAI-compatible Provider Name,应让专项 Runner +直接加载 ACL,不要经过 Wrapper: + +```bash +A3S_CONFIG_FILE=/absolute/path/to/.a3s/config.acl \ + cargo test -p a3s-code-core --test test_deepseek_adversarial_e2e -- \ + --ignored --test-threads=1 --nocapture +``` + +完整的 Node.js、Python 与 Go DeepSeek Retrieval Matrix 使用 +`A3S_REAL_EVAL_ROOT=/absolute/path/to/a3s`;各平台命令与 Gate 见 +[Cross-SDK Evaluation Contract](https://github.com/A3S-Lab/Code/blob/main/sdk/evaluation/README.md#real-deepseek-matrix)。 + +包含 Secret 的原始 Provider Evidence 应留在发布记录或 CI Artifact,不写入公开文档。 +只有不包含 Endpoint、Header、环境变量名、Credential、Prompt 或源文本的聚合脱敏指标 +才可发布。完整本地 API 表面见 [API Contract](/guide/api-contract)。 diff --git a/website/docs/v8.5.1/zh/guide/rfcs/_meta.json b/website/docs/v8.5.1/zh/guide/rfcs/_meta.json new file mode 100644 index 00000000..04371f2a --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/rfcs/_meta.json @@ -0,0 +1 @@ +["workspace-remote-git"] diff --git a/website/docs/v8.5.1/zh/guide/rfcs/workspace-remote-git.mdx b/website/docs/v8.5.1/zh/guide/rfcs/workspace-remote-git.mdx new file mode 100644 index 00000000..f16eba30 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/rfcs/workspace-remote-git.mdx @@ -0,0 +1,500 @@ +--- +title: 'RFC:远端 Workspace Git 后端' +description: '为非本地 workspace 后端提供 git 操作的 HTTP/JSON 协议' +--- + +# RFC:远端 Workspace Git 后端 + +| 字段 | 取值 | +| ---- | ------------------------------------------------------- | +| 状态 | **已实现** v2.6.x(`core/src/workspace/remote_git.rs`) | +| 归属 | `crates/code` workspace 子系统 | +| 关联 | S3 / 非本地 workspace 加固 Phase 4.1 | +| 相关 | `S3WorkspaceBackend`、`WorkspaceGit*` trait 族 | + +本文档作为协议规范保留——讲此协议的客户端与服务端应当严格匹配下方的 +线协议形态。后续修订应作为独立的 amendment 文档发布,而非原地修改这 +份 RFC,以便"实际 ship 出去的是什么"这条历史线索可审计。 + +本 RFC 提出一套协议和 Rust 客户端,让 `a3s-code` 的内置 `git` 工具能够 +跑在非本地文件系统的 workspace 之上——首先服务 S3,将来覆盖容器 / DFS +等无法承载 `.git` 目录的后端。 + +## 1. 动机 + +Workspace 抽象层已经通过 `WorkspaceFileSystem` 把内置工具与具体文件系统 +解耦。对 `bash`、`grep`、`glob`、`git` 这四个工具,我们使用**能力门控**: +后端无法提供的能力对应的工具就不会被注册,模型根本看不到它。 + +这对 `bash` 是合理的(对象存储确实没法跑 shell),对 `grep` 是可接受的 +(我们后来加了 `LIST + GET + regex` 的降级路径)。但对 `git` 是**痛点**: +很多 a3s-code 工作流期望查询分支/提交状态、diff 工作区、创建分支、stash +变更。如果只对本地会话开放 `git`,云端 workspace 就成了二等公民。 + +本方案是引入一个**远端 `WorkspaceGit` 后端**,把这些操作通过网络委托 +给宿主运维的外部服务。该服务持有真实的工作区(或基于 libgit2 的实现), +对外暴露一小套 HTTP API。a3s-code 客户端讲这套 API,对工具层来说它 +就是另一个 `WorkspaceGit` provider。 + +```text + ┌──────────────────┐ + │ a3s-code │ + 模型 │ │ + ──────► │ git 工具 │ HTTP/JSON + │ │ │ ┌────────────────┐ + │ ▼ │ │ │ + │ WorkspaceGit ───┼──►│ gitserver │ + │ (RemoteGit…) │ │ (libgit2 / sh) │ + │ │ │ │ + │ WorkspaceFs ────┼──►│ (S3, etc.) │ + └──────────────────┘ └────────────────┘ +``` + +## 2. 非目标 + +- **托管 gitserver。** 本 RFC 只定义客户端与协议,服务端的实现 + (libgit2 封装、shell 外调、Gitea API 适配等)超出范围。 +- **替换本地 git。** 当 `WorkspaceFs` 是 `LocalWorkspaceBackend` 时, + 现有的 `LocalWorkspaceBackend` 的 `WorkspaceGit` 实现仍是默认。 + 宿主按会话自由组合。 +- **push / pull。** `WorkspaceGit` trait 对远程是只读的 + (`list_remotes` 返回已配置的远程;不暴露 `git push` / `git fetch`)。 + 如果有工具需要,未来 RFC 再加。 +- **Worktrees。** Worktree 是本地文件系统概念;远端后端**不**实现 + `WorkspaceGitWorktreeProvider`。详见 §8。 + +## 3. 协议选择 — HTTP/JSON + +**结论:** HTTP/JSON,不用 gRPC。 + +| 评估维度 | HTTP/JSON | gRPC | +| ---------------- | ------------------------ | ------------------------------ | +| 现有依赖 | `reqwest` 已在树中 | `tonic` + proto 工具链全是新增 | +| schema 严格度 | 手写 `serde` 类型 | `.proto` 强类型 | +| 测试 mock 难度 | `wiremock` 就够 | 需要 gRPC mock 基础设施 | +| 运维可调试性 | `curl` 直接打 | `grpcurl`(普及度低) | +| 流式 | chunked / SSE | 原生双向流 | +| 操作数量 | ~12 个 op,全部请求/响应 | 数量同上;流式罕用 | +| 服务端实现自由度 | 直接包 `git` 或 `gitea` | 必须实现 gRPC server | + +操作面 ~12 个 RPC、请求响应都是扁平结构。流式只对 `log` 和 `diff` 有 +点用,而这两个客户端已经有上限。HTTP/JSON 在 mock 与运维调试上的优势 +明显压过 gRPC 的 schema 优势。如果操作面剧增或流式变成关键路径, +再回头考虑 gRPC。 + +我们对**所有操作都用 `POST`**:git 操作本质是命令式而非资源 CRUD, +请求体让加字段(兼容方式)变得轻松——`GET` 强迫所有信息塞 query string。 + +## 4. 仓库标识 + +客户端知道自己在操作哪个仓库,服务端需要路由。仓库标识写在 **URL 路径** +里: + +``` +POST /v1/repos/{repo_id}/git/ +``` + +`repo_id` 是宿主与 gitserver 运维方协商的、不透明的、URL 安全字符串。 +典型值有: + +- `users/{user_id}/sessions/{session_id}`(与 S3 工作区前缀一一对应) +- UUID +- 服务端的工作区路径 + +客户端把 `repo_id` 当不透明字符串;服务端负责映射到实际的工作区/裸仓。 + +## 5. 端点参考 + +全部 `POST`,全部 `application/json` 收发。字段命名 `snake_case`,匹配 +Rust serde 默认值。 + +### 5.1 状态 — `WorkspaceGit::status` + +``` +POST /v1/repos/{repo_id}/git/status +→ 200 { + "branch": "main", + "commit": "abc123...", + "is_worktree": false, + "is_dirty": true, + "dirty_count": 3 + } +→ 404 {"error":{"code":"REPO_NOT_FOUND", ...}} +``` + +### 5.2 日志 — `WorkspaceGit::log` + +``` +POST /v1/repos/{repo_id}/git/log +{"max_count": 10} +→ 200 { + "commits": [ + {"id":"abc...", "message":"feat: ...", "author":"Alice ", "date":"2026-05-19T..."} + ] + } +``` + +### 5.3 列出分支 — `WorkspaceGit::list_branches` + +``` +POST /v1/repos/{repo_id}/git/branches +→ 200 {"branches":[{"name":"main", "is_current":true}, ...]} +``` + +### 5.4 创建分支 — `WorkspaceGit::create_branch` + +``` +POST /v1/repos/{repo_id}/git/branches/create +{"name":"feat/x", "base":"main"} +→ 201 {} +→ 409 {"error":{"code":"BRANCH_EXISTS", ...}} +→ 404 {"error":{"code":"BASE_NOT_FOUND", ...}} +``` + +### 5.5 检出 — `WorkspaceGit::checkout` + +``` +POST /v1/repos/{repo_id}/git/checkout +{"refspec":"feat/x", "force":false} +→ 200 {"stdout":"Switched to branch 'feat/x'"} +→ 409 {"error":{"code":"WORKING_TREE_DIRTY", ...}} +``` + +### 5.6 差异 — `WorkspaceGit::diff` + +``` +POST /v1/repos/{repo_id}/git/diff +{"target": null} // null 表示工作区与暂存区的差异 +{"target": "main"} // 与指定引用比较 +→ 200 {"diff":"<统一差异文本>", "truncated": false} +``` + +`truncated` 为 `true` 表示服务端截断了响应正文——见 §9。客户端会在 +`git diff` 工具结果里把这个信号透出去。 + +### 5.7 列出远端 — `WorkspaceGit::list_remotes` + +``` +POST /v1/repos/{repo_id}/git/remotes +→ 200 {"remotes":[{"name":"origin", "url":"git@github.com:...", "direction":"fetch"}]} +``` + +### 5.8 是否为仓库 — `WorkspaceGit::is_repository` + +``` +POST /v1/repos/{repo_id}/git/exists +→ 200 {"is_repository": true} +``` + +与“仓库不存在”(404)有别:服务端可能允许 `repo_id` 映射到非 Git +目录。`is_repository` 让客户端不依赖 404 就能探测意图。 + +### 5.9 列出贮藏 — `WorkspaceGitStashProvider::list_stashes` + +``` +POST /v1/repos/{repo_id}/git/stashes +→ 200 {"stashes":[{"index":0, "message":"WIP on main: ..."}]} +``` + +### 5.10 创建贮藏 — `WorkspaceGitStashProvider::stash` + +``` +POST /v1/repos/{repo_id}/git/stashes/create +{"message":"wip", "include_untracked":true} +→ 201 {} +→ 409 {"error":{"code":"NOTHING_TO_STASH", ...}} +``` + +## 6. 认证 + +客户端支持两种传输认证模式,按会话配置: + +1. **Bearer token(默认)。** `Authorization: Bearer `。Token + 下发是宿主的责任(例如同一身份层签发的短期 JWT,复用 S3 访问的 + 门禁)。 +2. **mTLS。** 在 backend config 上设置 `client_cert_pem` 和 + `client_key_pem` 两个路径。客户端在构造时读两个文件,拼接后交给 + `reqwest::Identity::from_pem`。`rustls-tls` 后端要求密钥是 PKCS#8 + PEM 格式。只设一边会在构造期 fail-closed 报错。 + +两者可以同时启用——深度防御的部署直接两个都设。 + +无认证模式(本地开发场景)通过把 token 设为空、不设 mTLS 来开启; +客户端在构造时会 `tracing::warn!` 一条以让这个状态可见。 + +## 7. 错误模型 + +HTTP 状态码是**传输信号**。错误**类别**在 JSON body 里: + +```json +{ + "error": { + "code": "BRANCH_EXISTS", + "message": "branch 'feat/x' already exists" + } +} +``` + +客户端映射: + +| HTTP | 默认行为 | +| ------- | ------------------------------------------- | +| 200/201 | `Ok(...)` | +| 400 | `Err(anyhow!("bad request: {message}"))` | +| 401/403 | `Err(anyhow!("auth failed: {message}"))` | +| 404 | `Err(anyhow!("not found: {message}"))` | +| 409 | 类型化 conflict — 见下 | +| 5xx | `Err(anyhow!("gitserver internal: {...}"))` | + +客户端为可恢复 conflict 引入一个类型化错误: + +```rust +#[derive(Debug, Clone, thiserror::Error)] +#[error("remote git conflict: {code}: {message}")] +pub struct RemoteGitConflict { + pub code: String, + pub message: String, +} +``` + +希望从 `BRANCH_EXISTS` / `WORKING_TREE_DIRTY` / `NOTHING_TO_STASH` +恢复的工具用 `anyhow::Error::downcast_ref::()` +取出来——这与 `edit` / `patch` 在 S3 CAS 路径上用 `WorkspaceVersionConflict` +的模式相同。 + +协议正式定义的错误码(可扩展): + +| Code | 来源 | +| -------------------- | --------------------------------------------- | +| `REPO_NOT_FOUND` | `repo_id` 未注册 | +| `NOT_A_REPOSITORY` | 路径存在但不是 git 仓库 | +| `BRANCH_EXISTS` | `create_branch` 同名冲突 | +| `BRANCH_NOT_FOUND` | checkout / diff 目标缺失 | +| `BASE_NOT_FOUND` | `create_branch` 的 base ref 缺失 | +| `WORKING_TREE_DIRTY` | `checkout` 会丢工作区改动(且 `force=false`) | +| `NOTHING_TO_STASH` | 干净工作区上的 stash | +| `RATE_LIMITED` | 服务端限流(自定义阈值) | + +## 8. 可选 Trait + +`WorkspaceGit` 完整实现。 +`WorkspaceGitStashProvider` 实现。 +`WorkspaceGitWorktreeProvider` **故意不实现**。Worktree 是本地文件系统 +概念,对远端服务映射不干净: + +- "在路径 X 创建一个 worktree"——客户端没有路径概念;服务端的路径 + 布局对客户端是不透明的。 +- 工具里用 worktree 隔离的工作流(多个 agent 并行跑在隔离副本上), + 在云端的更优形式是**多个会话,每个会话有自己的 `repo_id`**,而不是 + 在客户端模拟一个本地文件系统特性。 + +依赖 `WorkspaceGitWorktreeProvider` 的工具会从 `services.git_worktree()` +拿到 `None`,并报"worktrees unavailable on remote git workspaces"。 + +## 9. 大小与成本上限 + +沿用 S3 后端的设防方式,远端 git 客户端强制几个客户端侧上限,避免 +模型触发无界响应: + +| 配置项 | 默认值 | 作用对象 | +| ------------------------ | ------ | ----------------------------- | +| `max_diff_bytes` | 1 MiB | `diff` 响应 body | +| `max_log_entries` | 200 | `log` 的 `max_count` 上限 | +| `request_timeout` | 30 s | 每次 HTTP 调用 | +| `operation_timeout` (WS) | 60 s | 叠加在 `WorkspaceServices` 层 | + +期望服务端也尊重这些上限——客户端会把相关上限传到请求里(例如 +`max_log_entries`),并把服务端返回的 `diff.truncated` 透出去,让 +工具能提示"diff 太大被截断,请缩小目标"。 + +## 10. Rust 客户端设计 + +```rust +// core/src/workspace/remote_git.rs + +#[derive(Debug, Clone)] +pub struct RemoteGitBackendConfig { + pub base_url: String, // https://git.example.invalid + pub repo_id: String, // path-segment 安全 + pub bearer_token: Option, + pub client_cert_pem: Option, // mTLS + pub client_key_pem: Option, + pub request_timeout: Option, // 默认 30s + pub max_diff_bytes: Option, // 默认 1 MiB + pub max_log_entries: Option, // 默认 200 +} + +#[derive(Debug, Clone)] +pub struct RemoteGitBackend { + http: reqwest::Client, + base_url: String, + repo_id: String, + max_diff_bytes: u64, + max_log_entries: usize, +} + +#[async_trait] +impl WorkspaceGit for RemoteGitBackend { /* 见 §5 */ } + +#[async_trait] +impl WorkspaceGitStashProvider for RemoteGitBackend { /* 见 §5 */ } +``` + +组合工厂沿用 S3 的形式: + +```rust +impl WorkspaceServices { + /// 在已有文件系统后端之上挂一个远端 git provider。 + pub fn with_remote_git( + self: Arc, + cfg: RemoteGitBackendConfig, + ) -> Arc { ... } +} +``` + +或者作为 S3 + 远端 git 工作区的顶层便利方法: + +```rust +pub fn s3_with_remote_git( + s3: S3BackendConfig, + git: RemoteGitBackendConfig, +) -> Arc { ... } +``` + +接线沿用现有的 builder 模式: + +```rust +let backend = Arc::new(RemoteGitBackend::new(cfg)); +let git: Arc = backend.clone(); +let stash: Arc = backend; + +WorkspaceServices::builder(workspace_ref, fs) + .file_system_ext(fs_ext) // S3 ETag CAS + .git(git) + .git_stash(stash) + // 不设 git_worktree — 见 §8 + .operation_timeout(Duration::from_secs(60)) + .build() +``` + +之后能力门控会自动注册 `git` 工具。 + +## 11. 每调用可观测性 + +每次 HTTP 调用发一个 `tracing::debug!` 事件,字段形态与 +`S3WorkspaceBackend::emit_s3_call_event` 一致(见 +`core/src/workspace/s3.rs`): + +| 字段 | 示例 | +| ------------- | ----------------------------- | +| `op` | `git.status`、`git.diff`、... | +| `repo_id` | `sessions/example` | +| `outcome` | `ok` \| `error` | +| `status` | HTTP 状态码 | +| `bytes` | 响应 body 长度 | +| `duration_ms` | 墙钟耗时 | + +已经 meter S3 成本的宿主可以把同一个 subscriber 接上来 meter gitserver +成本——不引入新依赖,也不要新接口。 + +## 12. 组合示例 + +### S3 workspace + 远端 git + +```rust +let ws = WorkspaceServices::s3_with_remote_git( + S3BackendConfig::new("workspace", "sessions/example", access_key, secret_key) + .endpoint("https://s3.example.invalid") + .force_path_style(true) + .enable_search(true), + RemoteGitBackendConfig::new( + "https://git.example.invalid", + "sessions/example", + ) + .bearer_token(token), +); + +let session = agent + .session_builder("s3://workspace/sessions/example") + .options(SessionOptions::new().with_workspace_backend(ws)) + .build() + .await?; +``` + +该会话注册的工具:`read`、`write`、`edit`、`patch`、`ls`、`grep`、`glob`、 +`git`。(`bash` 仍然隐藏——对象存储跑不了 shell。) + +### 本地文件系统 + 远端 git(混合) + +CI 跑在本地 checkout 上,但宿主想把 git 操作路由进沙箱化服务 +(例如做审计或限流)时有用。 + +```rust +let local = WorkspaceServices::local("/workspaces/repo"); +let ws = local.with_remote_git(remote_cfg); // 覆盖本地的 git provider +``` + +## 13. 未决问题 + +这些应在 Phase 4.2(实现)开始前敲定。 + +1. **Diff 方言。** 本地后端返回原始 `git diff` 的 stdout。远端 API 是否 + 应该强制具体 diff 方言(POSIX `diff -u`?libgit2 的变体?),还是 + 透传服务端产出?建议:透传,文档要求服务端必须生成 unified diff。 + +2. **并发操作。** 两个客户端同时访问同一 `repo_id`(一个跑 `checkout`、 + 一个跑 `diff`),服务端是否要串行?建议:服务端按 `repo_id` + 做串行;写进文档;客户端不在冲突上重试。 + +3. **长任务。** 大树上的 `checkout` 可能超过 `request_timeout`。协议是 + 否要支持轮询 / 异步 job 模式?建议:先不做。设合理的超时;我们 + 面向的(每会话沙箱)工作区不大。 + +4. **Hooks。** 服务端可能装了 pre-commit / pre-push hooks。响应里要 + 不要给一个"hook 输出"通道?建议:当非空时把 hook 的 stderr 塞进 + `checkout` / `stash` 的响应;hook 失败映射到 HTTP 422 + + `HOOK_FAILED`。 + +5. **schema 版本化。** 第一版端点放 `/v1/`。什么时候升 `/v2/`? + 建议:只在请求/响应 shape 出现不兼容变更时;新增字段保留在 + `/v1/`(客户端要忽略未知字段)。 + +6. **参考实现。** 是否要在本仓库放一个最小参考 gitserver + (比如 libgit2 + Rust)?建议:仓库外。客户端 + 协议足够;参考 + 服务归运维方。 + +## 14. 范围外(未来 RFC) + +- **push / fetch 到上游。** 让"agent 改完代码 push 回去"的工作流成立, + 涉及凭证下发。 +- **稀疏 / 局部 checkout。** 大型 monorepo 必要;当前面假定整个仓库 + 全量落地到服务端。 +- **流式 `log` / `diff`。** 当响应稳定超出 `max_diff_bytes` 时必要, + 会让 gRPC 的判断重新被讨论。 +- **Hooks 管理。** 通过客户端列举 / 配置服务端 hook。 + +## 15. 实现笔记(v2.6.x 已发布) + +最初草案给的是 8 步实施提纲;实际发布出去的(Phase 4.2 + 后续 Phase 5.x)如下: + +1. `RemoteGitBackend` / `RemoteGitBackendConfig` / `RemoteGitConflict` 位于 + `core/src/workspace/remote_git.rs`。最终没有加 + `remote-git` cargo feature——`reqwest` 已经在依赖里,模块无条件编译。 +2. `WorkspaceGit` 与 `WorkspaceGitStashProvider` 完整实现; + `WorkspaceGitWorktreeProvider` 故意不实现(见 §8)。 +3. `WorkspaceServices::with_remote_git` 是公开挂载点。内部 helper + `with_git_provider`(v2.6.x 一个后续提交里加上)用 struct literal + 显式拷贝字段——确保 `WorkspaceServices` 未来加字段时装饰器不会静默丢失。 +4. 除 bearer token 外支持 mTLS(`client_cert_pem` + `client_key_pem`), + 在 Phase 5.2 一次后续提交里 ship;同一个提交把上方 §6 也更新到了 + 已实现状态。 +5. `diff` 客户端侧防 OOM:HTTP body 流式累积,硬上限 + `max_diff_bytes * 4`(Phase 6.2)。soft `max_diff_bytes` 显示截断 + 在 JSON decode 之后照旧生效。 +6. 测试面:`remote_git.rs` 里 25+ 个 wiremock 单元测试 + 1 个端到端 + 驱动内置 `git` 工具的集成测试;workspace 通用 conformance suite + 在 Phase 6.3 加入。 +7. README 与 CHANGELOG 已更新;TLS 后端选择与 AWS SDK 一致用 + rustls-tls。 +8. SDK 暴露在 Phase 5.1 完成(Node + Python)——草案曾说"trait 稳定之 + 后再补",结果是 Phase 4.2 完成两个提交后就上了。 diff --git a/website/docs/v8.5.1/zh/guide/security.mdx b/website/docs/v8.5.1/zh/guide/security.mdx new file mode 100644 index 00000000..0d90585b --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/security.mdx @@ -0,0 +1,65 @@ +--- +title: '安全' +description: '权限策略、HITL、hook 与验证关卡' +--- + +# 安全 + +A3S Code 在创建 session 时暴露安全控制,并通过 session hook 暴露生命周期回调。宿主直接调用 `session.tool()`、`session.bash()` 等 API 时,应把它们视为宿主侧特权操作;是否把这些能力暴露给用户,由宿主应用自己判断。 + +委派子运行会继承所选 subagent 或 worker spec 的有边界权限。使用 `confirmationInheritance` 控制子运行的 Ask 决策,并把高风险发布命令放在显式 policy 后面。 + +权限确认界面应该向用户说明操作、原因、影响范围和风险,而不只是放两个按钮。 + +## 安全下载 + +`download` 是有边界的工作区修改操作,不是任意网络或文件系统原语。它只会在会话拥有 +可写本地工作区时注册。模型选择的调用与其他修改一样经过权限策略、HITL 确认、Hook、 +超时、取消和工作区检查。直接调用 `session.tool('download', ...)` 仍是宿主特权操作, +嵌入应用必须先完成授权。 + +网络边界只接受 HTTP(S),拒绝 URL 用户信息和非公网目标,并逐跳重新校验有限次数的 +重定向。直接连接会拒绝包含任何私网或保留地址的 DNS 结果,并把该跳验证后的公网地址 +固定给请求。跨源重定向会移除凭据和资源校验器。显式代理模式由已配置代理解析域名, +但仍保留字面主机与重定向检查。 + +对象存储和发行系统经常用签名查询参数授权,所以实际请求会保留它们;诊断信息和 +`source_anchors` 会移除这些参数,成功或失败的工具元数据不会暴露签名。 + +目标路径必须位于本地工作区内,且不能跨越符号链接。内容在字节数和时间上限内流式写入 +相邻临时文件;严格 Range 校验、可选 `expected_sha256`、先同步后提升与原子替换会阻止 +不完整或未验证内容进入最终路径。取消或失败会清理临时文件,`overwrite` 默认 `false`。 + +完整参数契约见[工具](/guide/tools#二进制安全的本地下载)。 + +## 权限策略与 Hook + +```ts title=permissions.ts +const session = agent.session('/repo', { + // !callout(1:7) 执行前生效 + permissionPolicy: { + deny: ['bash(rm -rf*)', 'write(**/.env*)'], + ask: ['bash(git push*)', 'bash(npm publish*)'], + allow: ['read(*)', 'search(*)', 'bash(cargo test*)'], + defaultDecision: 'ask', + enabled: true, + }, +}); +``` + +Hook API 的管理面包括注册、计数和卸载: + +```ts +session.registerHook( + 'observe-secret-read', + 'pre_tool_use', + { pathPattern: '**/.env*' }, + { priority: 100 }, + () => ({ action: 'continue' }), +); + +console.log(session.hookCount()); +session.unregisterHook('observe-secret-read'); +``` + +发布流程应要求测试、包检查、CI 和 provider 验证证据。 diff --git a/website/docs/v8.5.1/zh/guide/sessions.mdx b/website/docs/v8.5.1/zh/guide/sessions.mdx new file mode 100644 index 00000000..2dfbb4f4 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/sessions.mdx @@ -0,0 +1,555 @@ +--- +title: '会话' +description: '创建、流式输出、恢复和保存工作区绑定的会话' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 会话 + +`Agent` 持有配置和服务提供商状态。`Session` 把这个智能体绑定到一个工作区和一次 +对话生命周期。 + +界面通常从这里开始接入:订阅 Session 产生的 `AgentEvent`,再把事件映射为进度、 +工具调用、权限确认和结果。 + +```ts title=session.ts +import { Agent } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +// !focus(1:7) +const session = agent.session('/repo', { + model: 'provider/model-id', + planningMode: 'enabled', + goalTracking: true, + autoDelegation: { enabled: true, maxTasks: 4 }, + autoParallel: false, +}); +``` + +## Rust 构建路径 + +Rust 会话构建以异步为先,因为默认内存存储、文件型存储、队列、轨迹记录与 +MCP 发现都可能需要 I/O: + +```rust +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +`session_async`、`resume_session_async`、`session_for_agent_async` 与 +`session_for_worker_async` 是直接的异步入口。Node 和 Python 保留现有工厂 +命名,由原生绑定在内部委托给同一个异步构建内核。 + +同步 Rust `Agent::session` 是严格兼容路径:只接受已经显式初始化好的资源,绝不 +启动或阻塞异步运行时。仍需初始化的默认或文件型内存存储、文件会话存储、队列、 +轨迹记录器,以及 `SessionOptions` 中的任何宿主 MCP 管理器都会返回 +`CodeError::AsyncSessionBuildRequired`;会话选项中的 MCP 能力发现始终是异步的。 +应改用构建器,不要捕获错误后悄悄更换后端。 +同步路径只能继承智能体初始化时已经缓存的全局 MCP 工具。 + +`planningMode` 是显式三态:`'auto'` 使用默认结构化预分析,`'enabled'` +强制规划,`'disabled'` 在低延迟调用中关闭规划。旧的布尔 `planning` 选项仍保留兼容。 + +规划会写入运行级状态。宿主应用可以把这些状态渲染成任务列表,并随着运行事件更新 +每一项,而不是从文本令牌中猜测进度。 + +## 单任务操作约束 + +同一个会话同时只准入一个会影响对话记录的操作。`send`、`stream`、它们的附件变体、 +斜杠命令与 `resumeRun` 共用快速失败准入门。重叠调用会在读取历史或派发命令前返回 +`CodeError::SessionBusy`,不会排队 +等待当前操作。 + +开始下一次对话操作前,应等待活动结果、把事件流消费到结束,或先取消它。 +即使公开的事件流句柄被丢弃或中止,运行时也会在生产者真正停止前继续 +持有租约。直接调用宿主工具的辅助方法不改变对话记录,因此不占用这把租约。 + +Node 与 Python 的事件流迭代器会在终止边界等待这段生命周期清理。迭代器完整消费 +并报告结束后,立即开始下一次对话操作不会继承上一个事件流留下的过期忙碌状态。 + +## 发送消息 + + + + +```rust +let result = session + .send("审查这个仓库并列出发布阻塞项", None) + .await?; + +println!("{}", result.text); +println!("{}", result.usage.total_tokens); +println!("{:?}", result.verification_summary().status); +``` + + + + +```ts +const result = await session.send('检查仓库并列出发布阻塞项'); + +console.log(result.text); +console.log(result.totalTokens); +console.log(result.verificationStatus); +``` + + + + +```python +result = session.send("审查这个仓库并列出发布阻塞项") + +print(result.text) +print(result.total_tokens) +print(result.verification_status) +``` + + + + +```go +result, err := session.Run(ctx, "检查仓库并列出发布阻塞项") +if err != nil { + return err +} + +fmt.Println(result.Text) +fmt.Println(result.Usage.TotalTokens) +fmt.Println(result.VerificationSummary.Status) +``` + + + + +## 流式输出 + + + + +```rust +use a3s_code_core::{AgentEvent, CodeError}; + +let (mut events, lifecycle) = session + .stream("运行相关测试并解释失败原因", None) + .await?; + +while let Some(event) = events.recv().await { + match event { + AgentEvent::TextDelta { text } => print!("{text}"), + AgentEvent::ToolStart { name, .. } => println!("\n工具:{name}"), + AgentEvent::End { .. } => break, + AgentEvent::Error { message } => return Err(CodeError::Llm(message)), + _ => {} + } +} +lifecycle + .await + .map_err(|error| CodeError::Internal(error.into()))??; +``` + + + + +```ts +const stream = await session.stream('运行聚焦测试并解释失败'); + +while (true) { + const { value: event, done } = await stream.next(); + if (done) break; + if (!event) continue; + + if (event.text) process.stdout.write(event.text); + if (event.toolName) console.log('tool:', event.toolName); +} +``` + + + + +```python +for event in session.stream("运行相关测试并解释失败原因"): + if event.type == "text_delta" and event.text: + print(event.text, end="", flush=True) + elif event.type == "tool_start": + print(f"\n工具:{event.tool_name or '未知'}") + elif event.type == "error": + raise RuntimeError(event.error or "流式执行出错") +``` + + + + +```go +stream, err := session.Stream(ctx, "运行聚焦测试并解释失败", nil) +if err != nil { + return err +} + +for event := range stream.Events { + if event.Type != code.EventTextDelta { + continue + } + var payload struct { + Text string `json:"text"` + } + if err := event.DecodePayload(&payload); err != nil { + return err + } + fmt.Print(payload.Text) +} +if err := <-stream.Done; err != nil { + return err +} +``` + + + + +每个 SDK 事件都是 `EventEnvelopeV1` 投影,包含 `version === 1`、开放的 `type` +字符串、完整 `payload` 与可选 `metadata`。`text`、`toolName` 等便捷字段 +由信封统一派生。消费端应保留默认分支,并为未来的事件类型保存原始载荷。 + +## 调整或中断活动 Run + +Run Control 修改正在执行的操作,不会启动第二个对话 Turn。`steer` 把更新后的用户指令 +排入下一个安全点;`interrupt` 协作式取消当前 Provider 与工具工作,并等待受监督清理 +完成后再把 Run 置为 `cancelled`。 + + + + +```rust +use a3s_code_core::{InterruptRequest, SteerRequest}; + +let state = session.run_control_snapshot().await; +let receipt = session + .steer(SteerRequest::new("优先处理失败的测试")) + .await?; +println!("{:?}", receipt.state); + +session + .interrupt(InterruptRequest::new().with_reason("用户停止运行")) + .await?; +``` + + + + +```ts +const state = await session.runControlSnapshot(); +const receipt = await session.steer('优先处理失败的测试', { + runId: state?.runId, + expectedTurnId: state?.turnId, + expectedTurnRevision: state?.turnRevision, +}); +console.log(receipt.state); + +await session.interrupt({ reason: '用户停止运行' }); +``` + + + + +```python +state = await session.run_control_snapshot_async() +options = {} +if state: + options["run_id"] = state["run_id"] + if state.get("turn_id") is not None: + options["expected_turn_id"] = state["turn_id"] + options["expected_turn_revision"] = state["turn_revision"] +receipt = await session.steer_async( + "优先处理失败的测试", + options, +) +print(receipt["state"]) + +await session.interrupt_async({"reason": "用户停止运行"}) +``` + + + + +```go +state, err := session.RunControlSnapshot(ctx) +if err != nil { + return err +} +options := &code.SteerOptions{} +if state != nil { + options.RunID = &state.RunID + options.ExpectedTurnID = state.TurnID + options.ExpectedTurnRevision = &state.TurnRevision +} +receipt, err := session.Steer(ctx, "优先处理失败的测试", options) +if err != nil { + return err +} +fmt.Println(receipt.State) +_, err = session.Interrupt(ctx, &code.InterruptOptions{}) +``` + + + + +调用方使用相同 Request ID 和相同载荷重试时,请求具有幂等性。可选的 Run 与预期 Turn +字段会让过期界面操作以失败为默认。已接受不等于已经应用;需要区分时,应观察 +`run_control_applied` 或查询持久 Run 事件。两种操作都不会改变模型、权限、沙箱、预算 +或确认策略。 + +## 临时提问 + +SDK 没有专用的临时提问辅助方法。要提出临时问题,可以先快照当前历史,再把它 +显式传给 `send` 或 `stream`。显式历史只服务这一次调用,不会把答案写回 +会话历史。 + +```ts +const snapshot = session.history(); +const answer = await session.send('这个 session 已经看过哪些文件?', snapshot); + +console.log(answer.text); +console.log(session.history().length === snapshot.length); +``` + +## 恢复会话 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions}; + +let session = agent + .resume_session_async( + "release-review", + SessionOptions::new().with_file_session_store("./.a3s/sessions"), + ) + .await?; +``` + + + + +```ts +import { Agent, FileSessionStore } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.resumeSession('release-review', { + sessionStore: new FileSessionStore('./.a3s/sessions'), +}); +``` + + + + +```python +from a3s_code import FileSessionStore, SessionOptions + +opts = SessionOptions() +opts.session_store = FileSessionStore("./.a3s/sessions") +session = agent.resume_session("release-review", opts) +``` + + + + +```go +options := &code.SessionOptions{ + FileSessionStoreDir: ".a3s/sessions", +} +session, err := agent.ResumeSession(ctx, "release-review", options) +``` + + + + +使用会话存储时设置 `autoSave: true`,或显式调用 `await session.save()`。 +这里恢复的是已保存的会话快照;中断运行的检查点通过 +`session.resumeRun(runId)` 恢复。Go 使用 `Save`、`ResumeSession` 和 +`ResumeRun(ctx, runID)` 对应同样的两层持久化。参见 +[持久化](/guide/persistence)。 + +## 生命周期与关闭 + +`session.close()` 是一次完整的优雅停止。首次调用会把会话切换到 +**已关闭**状态——之后的 `send`/`stream` 调用会以 `CodeError::SessionClosed` +快速失败,而不会启动新运行——然后取消正在执行的运行、所有正在进行的委派子 +智能体任务,以及所有挂起的人工确认。后续调用不会重复操作,且保证 +不会触发 panic。用 `session.isClosed()`(Node)、`session.is_closed()`(Python) +或 `session.IsClosed(ctx)`(Go)查询关闭状态。 + +```ts +session.close(); +if (session.isClosed()) { + // send/stream 现在会以 CodeError::SessionClosed 拒绝 +} +``` + +```python +session.close() +if session.is_closed(): + # send/stream 现在会以 CodeError::SessionClosed 拒绝 + pass +``` + +```go +if err := session.Close(ctx); err != nil { + return err +} +closed, err := session.IsClosed(ctx) +``` + +### 取消令牌 + +每个运行都通过 `child_token()` 从同一个会话级父令牌派生出自己的 +逐操作取消令牌,因此 `close()` 会一次性级联到所有正在进行的工作。需要原始 +令牌的嵌入方——例如把它接入宿主侧的 `select!`,或者绕过 `close()` 的 +运行存储和钩子副作用直接中止会话——可以通过 +`AgentSession::session_cancel_token()` 克隆它。 + +### 智能体侧会话注册表 + +所属的 `Agent` 通过 `Weak` 引用跟踪它的存活会话(惰性回收),这样控制面 +就能在不持有会话句柄的情况下驱动生命周期: + +- `Agent::list_sessions()` 返回存活的会话 ID(已排序,稳定)。 +- `Agent::close_session(id)` 按 ID 关闭单个会话——与 + `AgentSession::close()` 相同的清理流程,从带外调用。 +- `Agent::close()` 关闭每个存活会话并拆除智能体持有的后台资源(同时 + 断开全局 MCP 连接)。返回后,新的 `session` / `resumeSession` 调用会以 + `CodeError::SessionClosed` 快速失败。 +- `Agent::is_closed()` 报告智能体自身是否已被关闭。 + +```ts +const ids = await agent.listSessions(); +await agent.closeSession(ids[0]); +await agent.close(); // 关闭所有剩余会话和全局 MCP +console.log(agent.isClosed()); +``` + +```python +ids = agent.list_sessions() +agent.close_session(ids[0]) +agent.close() # 关闭所有剩余会话和全局 MCP +print(agent.is_closed()) +``` + +```go +ids, err := agent.ListSessions(ctx) +if err == nil && len(ids) > 0 { + _, err = agent.CloseSession(ctx, ids[0]) +} +err = agent.Close(ctx) +``` + +参见更新日志 `[3.3.0]` 中的“会话与智能体生命周期控制”。 + +## 宿主身份标签 + +`SessionOptions` 携带四个不透明的身份字段,宿主可以在创建会话时附加。 +框架只负责传递它们——从不解释或强制执行。它们会被传播进 `SessionData`、 +钩子和追踪,并在恢复时还原,因此宿主可以据此驱动多租户聚合、计费 +和分布式追踪: + +| Node(驼峰命名) | Python(蛇形命名) | Go | 含义 | +| ----------------- | ------------------- | ----------------- | ---------------------------------- | +| `tenantId` | `tenant_id` | `TenantID` | 多租户标签 | +| `principal` | `principal` | `Principal` | 触发会话的用户或服务 | +| `agentTemplateId` | `agent_template_id` | `AgentTemplateID` | 会话实例化所基于的智能体模板或定义 | +| `correlationId` | `correlation_id` | `CorrelationID` | 分布式追踪关联标识 | + +```ts +const session = agent.session('/repo', { + tenantId: 'tenant-example', + principal: 'principal-example', + agentTemplateId: 'agent-template-example', + correlationId: 'trace-example', +}); +``` + +```python +opts = SessionOptions() +opts.tenant_id = 'tenant-example' +opts.principal = 'principal-example' +opts.agent_template_id = 'agent-template-example' +opts.correlation_id = 'trace-example' +session = agent.session('/repo', opts) +print(session.tenant_id, session.principal) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + TenantID: "tenant-example", + Principal: "principal-example", + AgentTemplateID: "agent-template-example", + CorrelationID: "trace-example", +}) +``` + +参见更新日志 `[3.3.0]` 中的“宿主提供的身份标签”。 + +## 运行记录与回放 + +每次 `send()` 或 `stream()` 都会创建运行记录。应用可以用这些记录实现界面状态、 +审计、回放、取消和测试断言: + +```ts +const runs = await session.runs(); +const latest = runs.at(-1); + +if (latest) { + console.log(await session.runSnapshot(latest.id)); + console.log(await session.runEvents(latest.id)); +} +``` + +```go +runs, err := session.Runs(ctx) +if err == nil && len(runs) > 0 { + latest := runs[len(runs)-1] + snapshot, snapshotErr := session.RunSnapshot(ctx, latest.ID) + events, eventsErr := session.RunEvents(ctx, latest.ID) + _, _, _ = snapshot, snapshotErr, eventsErr + _ = events +} +``` + +`currentRun()` 用来读取调用当下的当前运行。`send()` 或 `stream()` 仍在 +执行时,可以把它的 `id` 传给 `cancelRun(id)` 请求取消。空闲时,`currentRun()` +可能返回 `null`,也可能保留一个运行快照;已完成历史应使用 `runs()`, +取消前必须检查 `status`: + +```ts +const current = await session.currentRun(); +if (current?.id && current.status === 'running') { + await session.cancelRun(current.id); +} +``` + +## 智能体定义 + +`sessionForAgent()` 应用一个命名的智能体定义,来源是内置智能体、 +`.a3s/agents` 或配置的 `agentDirs`。 + +```ts +const session = agent.sessionForAgent('/repo', 'explore', ['./agents'], { + planningMode: 'auto', +}); +``` + +```go +session, err := agent.SessionForAgent( + ctx, + "/repo", + "explore", + []string{"./agents"}, + &code.SessionOptions{PlanningMode: code.PlanningAuto}, +) +``` + +对于通过值定义的一次性 worker,可使用 `SessionForWorker`;两种方法都返回通用的 +Go `Session` API。 diff --git a/website/docs/v8.5.1/zh/guide/skills.mdx b/website/docs/v8.5.1/zh/guide/skills.mdx new file mode 100644 index 00000000..41ee9133 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/skills.mdx @@ -0,0 +1,47 @@ +--- +title: '技能' +description: '提示时技能、内联技能与技能工具' +--- + +# 技能 + +Skills 是可复用指令。文件型 skill 目录、inline skills、显式 registry 和 +`search_skills` 使用同一套 discovery 路径。A3S Code 不再内置默认 skills; +SDK 仍接受 `builtinSkills: true`,但它只是兼容 no-op。 + +```ts +const session = agent.session('/repo', { + builtinSkills: true, + skillDirs: ['/repo/.a3s/skills'], + inlineSkills: [ + { + name: 'strict-release-review', + kind: 'instruction', + content: '始终先列 blocker,再列可选改进。', + }, + ], +}); +``` + +`builtinSkills` 不会添加默认 skills。可复用行为应放进项目 skill 文件、 +inline skills 或 agent 定义。Skill 文件是带 frontmatter 的 Markdown。使用 +`skillDirs` 加载技能文件,使用 `agentDirs` 加载 worker/subagent 定义。可用 +`allowed-tools` 限定 Skill 调用期间的工具,例如 `read(*), search(*), bash(cargo test*)`。 + +`search_skills` 用于查找相关 skill: + +```ts +await session.tool('search_skills', { + query: 'release blockers', + limit: 5, +}); +``` + +带 `allowed-tools` 前置元数据的 Markdown 技能文件和内联技能都能被发现。技能管理 +现在由 SDK 或文件系统完成,不再作为模型可见的管理工具暴露。 + +当 skill 通过 `Skill` 工具被调用时,`allowed-tools` 是 fail-secure 的:省略 +frontmatter 不会授予任何工具。请显式声明最小可用工具集。普通 session 的工具调用 +默认不会被 active skill 限制,除非显式打开 +`enforceActiveSkillToolRestrictions` / `enforce_active_skill_tool_restrictions` +兼容旧行为。 diff --git a/website/docs/v8.5.1/zh/guide/tasks.mdx b/website/docs/v8.5.1/zh/guide/tasks.mdx new file mode 100644 index 00000000..b795cffc --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/tasks.mdx @@ -0,0 +1,334 @@ +--- +title: '任务' +description: '使用统一 task 工具与子智能体进行手动和自动委派' +--- + +# 任务 + +常规多智能体路径只有一个模型可见的 `task` 工具。它的 `tasks` 数组既可提交一个 +聚焦子任务,也可提交多个相互独立的子任务并发扇出。子运行上下文相互隔离,只向父 +智能体返回紧凑结果,而不是完整对话记录。 + +同一个委派核心也驱动自动子智能体委派。需要运行时主动为高置信工作启动专用 +子智能体时启用它;如果只想让自动委派串行执行,可以用 +`autoParallel: false` 关闭自动并行扇出。 + +Web 界面可以把这些状态分别展示为计划列表和子智能体运行列表,并用任务标识关联两者。 + +## 内置子智能体 + +| 智能体 | 适用场景 | +| ----------------------------- | -------------------------------------------------------- | +| `explore` | 只读代码搜索、文件检查和结构发现。 | +| `plan` | 只读实现计划和架构分析。 | +| `general` / `general-purpose` | 多步骤实现工作,可读写并执行命令。 | +| `verification` | 聚焦检查、复现、回归验证和对抗测试。 | +| `review` | 以发现为先的代码审查,关注正确性、回归、安全与可维护性。 | + +可以显式提及它们,例如 `@review`、`@agent-plan`、`使用 verification 子智能体` +或 `委派给 general-purpose`。 + +## 手动委派 + +让父智能体委派一个有边界的子任务: + +```text +Use task to ask an explore agent to inspect the auth module. +Return files inspected, findings, risks, and confidence. +``` + +宿主已经知道任务边界时,可以直接调用 SDK 辅助方法: + +```ts +const task = await session.task({ + agent: 'explore', + description: 'Inspect auth module', + prompt: 'Return files inspected, findings, risks, and confidence.', +}); +if (task.exitCode !== 0) throw new Error(task.output); +console.log(task.output); +``` + +子智能体应返回紧凑契约: + +- 摘要 +- 已检查或修改的文件 +- 证据引用 +- 风险和未知项 +- 置信度 + +父智能体不应吞入完整的子对话记录。 + +## 并行委派 + +当工作彼此独立时,在一次 `task` 调用中提交多个 `tasks` 项,或使用 +`session.tasks(...)` 并发执行: + +```text +Run one task call with three independent tasks: +1. inspect provider config parsing +2. inspect Node SDK declarations +3. inspect release scripts + +Merge the results into one release-readiness report. +``` + +```ts +const batch = await session.tasks([ + { + agent: 'explore', + description: 'Inspect config', + prompt: 'Check provider parsing.', + }, + { + agent: 'verification', + description: 'Verify SDK', + prompt: 'Check SDK declarations.', + }, +]); +if (batch.exitCode !== 0) throw new Error(batch.output); +console.log(batch.output); +``` + +统一 `task` 调用接受 1–32 个任务。只有单任务调用可以设置 `background`;多任务调用 +会收集所有分支,因此拒绝 `background: true`。默认要求所有分支成功;仅在允许不完整 +证据的场景使用 `allow_partial_failure`,且 `min_success_count` 只能在该模式下设置。 + +`session.task(...)` 和 `session.tasks(...)` 都返回来自 `task` 工具的 `ToolResult`。 +读取 `output` 获取紧凑摘要,并在信任结果前检查 `exitCode`。会话选项中的 +`maxParallelTasks` 与 ACL 中的 `max_parallel_tasks` 会限制同级任务扇出。 + +所有扇出都应使用 `task` / `session.tasks`。模型可见与 SDK 的 `parallel_task` / +`parallelTask` 助手已**移除**(`HARNESS-CONV4`),请改用多条目 `task`。 + + + +## Agent 级优先级调度器 + +每个 `Agent` 都拥有一个由其所有 Session 共享的调度器。它限制可同时执行的独立操作 +数量,并在有槽位释放时决定哪个等待操作先进入。调度器基于 `a3s-lane` 优先级队列, +无需额外启用。 + +这是准入边界,不是抢占式执行器:已经持有槽位的工作会运行到完成或取消;优先级只 +决定槽位空闲后哪个等待项先启动。 + +### 哪些操作共享边界 + +同一个 `max_active` 容量覆盖: + +- 通过 send、run 或 stream 启动的对话运行 +- 宿主发起的可信或受治理直接工具调用 +- detached 后台子任务 +- 宿主启动的工作流 + +这样多个 Session 不会各自获得一份互不相关的并发预算;繁忙的后台 Session 也不能 +通过另一套执行 API 绕过交互任务。 + +三个相邻控制项解决不同问题: + +| 控制项 | 作用域 | +| ------------------------------ | ------------------------------------ | +| `task_scheduler.max_active` | 一个 `Agent` 所有 Session 的全局准入 | +| `max_parallel_tasks` | 一次委派任务或工作流内的同级扇出 | +| [Lane 队列](/guide/lane-queue) | 可选的外部或混合 worker 分发 | + +Session 的单任务规则也相互独立:同一 Session 中两个会改变 transcript 的调用会立即 +失败,不会进入这个调度器等待。 + +### 配置容量与老化 + +```acl +task_scheduler { + max_active = 4 + aging_interval_ms = 30000 +} +``` + +两个值都必须大于零。默认允许 4 个活动操作,老化间隔为 30 秒。 + +### 选择优先级 + +| 优先级 | 适用场景 | 老化规则 | +| ------------- | -------------------------------- | -------------------- | +| `urgent` | 必须下一个运行的显式宿主控制工作 | 永不老化 | +| `interactive` | 面向用户的交互轮次 | 默认值,也是老化上限 | +| `foreground` | 可见但不直接阻塞交互的工作 | 向 interactive 提升 | +| `background` | detached 或异步工作 | 向 interactive 提升 | +| `maintenance` | 最低优先级的维护工作 | 向 interactive 提升 | + +低等级在高等级之后运行;相同有效优先级保持 FIFO。非 urgent 工作每等待满一个 +`aging_interval_ms` 就提升一级,最高到 `interactive`,因此持续的交互流量不会永久 +饿死 background 或 maintenance 工作。`urgent` 始终保留在老化任务之上。 + +在创建 Session 时指定优先级: + +```rust +use a3s_code_core::{SessionOptions, TaskPriority}; + +let options = SessionOptions::new() + .with_task_priority(TaskPriority::Background); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + +```ts +const session = await agent.sessionAsync('/repo', { + taskPriority: 'background', +}); +``` + +```python +from a3s_code import SessionOptions + +options = SessionOptions() +options.task_priority = "background" +session = agent.session("/repo", options) +``` + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + TaskPriority: code.TaskPriorityBackground, +}) +``` + +有效名称为 `urgent`、`interactive`、`foreground`、`background` 和 +`maintenance`;非法名称会在 Session 选项校验阶段失败。 + +### 观察占用情况 + +宿主可以从 `Agent` 或它的任意 Session 读取同一份即时快照: + +```rust +let stats = agent.task_scheduler_stats().await?; +let same_scheduler = session.task_scheduler_stats().await?; +println!("active={} pending={}", stats.active, stats.pending); +``` + +```ts +const stats = await agent.taskSchedulerStats(); +const sameScheduler = await session.taskSchedulerStats(); +console.log(stats.active, stats.pendingByPriority.background); +``` + +```python +stats = agent.task_scheduler_stats() +same_scheduler = session.task_scheduler_stats() +print(stats["active"], stats["pendingByPriority"]["background"]) +``` + +```go +stats, err := agent.TaskSchedulerStats(ctx) +sameScheduler, err := session.TaskSchedulerStats(ctx) +fmt.Println(stats.Active, stats.PendingByPriority.Background) +``` + +| 字段 | 含义 | +| ------------------- | -------------------------- | +| `maxActive` | 配置的全局容量 | +| `active` | 正在持有槽位的操作数 | +| `pending` | 等待准入的操作数 | +| `activeByPriority` | 按请求优先级分组的活动操作 | +| `pendingByPriority` | 按请求优先级分组的等待操作 | +| `closed` | 调度器是否正在关闭 | + +Rust 使用 snake_case struct 字段;Node.js 和 Python dict 使用 camelCase wire 名称; +Go 使用导出的 struct 字段。这是诊断快照,不是容量预留,读取后数值可能立即变化。 + +### 取消与关闭 + +取消会在等待工作获得槽位前把它移除。取消活动工作时,它会在结算后释放槽位。关闭 +`Agent` 会拒绝等待中和新提交的准入请求,再等待已经获得准入的工作完成,最后结束 +调度器。 + +## 自动委派 + +自动委派默认需要显式启用。运行时会把当前请求与内置或自定义智能体描述进行评分, +并在置信度足够时启动最多 `maxTasks` 个子运行。 + +```ts +const session = agent.session('/repo', { + autoDelegation: { enabled: true, minConfidence: 0.72, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, +}); +``` + +```acl +auto_delegation { + enabled = true + auto_parallel = false + min_confidence = 0.72 + max_tasks = 4 +} +``` + +`autoParallel: false` / `auto_parallel = false` 是自动并行子智能体扇出的全局开关。 +手动 `task` 扇出和 `session.tasks(...)` 仍然可用。 + +## 智能体目录 + +通过 `agentDirs`、`agent_dirs` 或 A3S 内置目录加载自定义智能体定义: + +```ts +const session = agent.session('/repo', { agentDirs: ['./.a3s/agents'] }); +const loaded = session.registerAgentDir('./more-agents'); +``` + +A3S 会扫描配置的 `agent_dirs`、项目/用户 `.a3s/agents`,以及 Claude 兼容的 `.claude/agents` 迁移路径。新项目优先使用 `.a3s/agents`。 + +Markdown agent 文件支持 frontmatter: + +```markdown +--- +name: docs-auditor +description: Use proactively after documentation changes +tools: Read, Grep, Glob +disallowedTools: + - Write + - Bash(rm:*) +--- + +Audit docs for drift, broken examples, and unclear migration notes. +``` + +`tools` 字段是 allowlist。`disallowedTools` 是 denylist,且优先级高于 allowlist。模型路由字段不属于这个兼容层。 + +## 工作智能体 + +通过 `workerAgents` 或 `registerWorkerAgent()` 注册一次性 worker agents: + +```ts +const session = agent.session('/repo', { + workerAgents: [ + { + name: 'frontend-worker', + description: 'Small verified frontend fixes', + kind: 'implementer', + model: 'provider/model-id', + maxSteps: 24, + confirmationInheritance: 'auto_approve', + }, + ], +}); +``` + +### 确认继承 + +通过 `confirmationInheritance` 控制子运行如何处理 Ask 决策: + +- `'auto_approve'`(默认):子运行自动批准所有 Ask 决策 +- `'deny_on_ask'`:子运行遇到 Ask 时立即失败 +- `'inherit_parent'`:子运行继承父级的确认策略 + +旧生命周期控制面 API 已移除;需要 UI 状态时,应用应消费 streaming events、run replay,以及 Node `cancelRun(runId)`。 + +## 可编程编排 + +本页的所有内容都是模型驱动的:`task`、`session.task(...)` / +`session.tasks(...)` 以及自动委派让 LLM 决定何时以及如何扇出。当宿主已经知道工作的 +形态并希望它确定可复现时,改用 `session.parallel(...)`、`session.pipeline(...)` 和 +`session.parallelResumable(...)` 以编程方式表达。开发者定义的扇出、无屏障流水线以及 +可恢复或可迁移工作流见[编排](/guide/orchestration)。 diff --git a/website/docs/v8.5.1/zh/guide/teams.mdx b/website/docs/v8.5.1/zh/guide/teams.mdx new file mode 100644 index 00000000..8754a31e --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/teams.mdx @@ -0,0 +1,127 @@ +--- +title: '团队' +description: '用智能体定义、task 工具和自动委派构建团队' +--- + +# 团队 + +团队是一种驾驭层模式:命名智能体定义加上统一的 `task` 委派核心。 +A3S Code 不再暴露单独的团队运行器 API;父会话仍负责合成、策略和最终验证。 + +## 推荐形态 + +1. 把长期角色放在 `.a3s/agents`,或配置额外 `agentDirs`。 +2. 每个角色保持聚焦:description、prompt、allowed tools、denied tools。 +3. 需要运行时选择高置信 subagent 时启用 `autoDelegation`。 +4. 宿主已知 lane 时直接调用 `session.task(...)` 或 `session.tasks(...)`。 +5. 父 agent 合并子摘要、证据引用和风险。 + +```ts +const session = agent.session('/repo', { + agentDirs: ['./.a3s/agents'], + planningMode: 'auto', + autoDelegation: { enabled: true, maxTasks: 4 }, + maxParallelTasks: 8, + autoParallel: false, +}); + +await session.send(` +Build a release-readiness team: +- explorer: find risky changed areas +- tester: identify missing verification +- security: review side-effect paths + +Use independent subagents where useful and return a single prioritized report. +`); +``` + +## 自定义智能体文件 + +Markdown agent 文件使用 Claude 兼容 frontmatter,但 A3S 原生位置是 `.a3s/agents`: + +```markdown +--- +name: release-reviewer +description: Use proactively after release or CI changes +tools: Read, Search, Bash(cargo test*) +disallowedTools: + - Write + - Bash(git push*) +--- + +Review release blockers first, then risks, then follow-up work. +``` + +A3S 也读取 `.claude/agents` 作为迁移来源。新项目优先使用 `.a3s/agents`。 + +## 内置团队角色 + +无需创建文件即可使用: + +- `explore`:只读仓库探索 +- `plan`:只读实现计划 +- `general` / `general-purpose`:多步骤实现 +- `verification`:检查、复现和回归验证 +- `review`:findings-first 代码审查 + +## 手动执行通道 + +宿主已经知道 lane 时,可直接调用 SDK helper: + +```ts +const result = await session.tasks([ + { + agent: 'explore', + description: 'Changed files', + prompt: 'Find risky changed files.', + }, + { + agent: 'verification', + description: 'Test gaps', + prompt: 'Find missing verification.', + }, + { + agent: 'review', + description: 'Regression review', + prompt: 'Review correctness risks.', + }, +]); + +if (result.exitCode !== 0) throw new Error(result.output); +console.log(result.output); +``` + +`session.tasks(...)` 是 `task` 的宿主侧多项封装;它返回 `ToolResult`, +而不是 `StepOutcome[]`。当你需要每条 lane 都有独立结构化 outcome 时,使用 +`session.parallel(...)`。 + +当团队的 lane 结构固定、应当可复现且可恢复而非由模型选择时,使用 [编排](/guide/orchestration) 中的可编程组合子(`session.parallel` / `session.pipeline` / `session.parallelResumable`)。 + +## 工作智能体 + +当角色由宿主动态构造而不是存储在磁盘上时,通过 `workerAgents` 或 `registerWorkerAgent()` 注册一次性 worker agents: + +```ts +const session = agent.session('/repo', { + workerAgents: [ + { + name: 'frontend-worker', + description: 'Small verified frontend fixes', + kind: 'implementer', + model: 'provider/model-id', + maxSteps: 24, + confirmationInheritance: 'auto_approve', + }, + ], +}); +``` + +通过 `confirmationInheritance` 控制子运行如何处理 Ask 决策: + +- `'auto_approve'`(默认):子运行自动批准所有 Ask 决策 +- `'deny_on_ask'`:子运行遇到 Ask 时立即失败 +- `'inherit_parent'`:子运行继承父级的确认策略 + +## 运行状态 + +应用应通过 streaming events 和 run replay 快照展示状态,Node 侧需要取消时使用 `cancelRun(runId)`。 diff --git a/website/docs/v8.5.1/zh/guide/telemetry.mdx b/website/docs/v8.5.1/zh/guide/telemetry.mdx new file mode 100644 index 00000000..083a8691 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/telemetry.mdx @@ -0,0 +1,132 @@ +--- +title: '遥测与可观测性' +description: '追踪事件、验证摘要与运行时可观测性' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 遥测与可观测性 + +A3S Code 通过事件、追踪、验证报告和结果元数据提供运行证据。 + +## 结果元数据 + + + + +```rust +let result = session.send("运行发布检查", None).await?; + +println!("{}", result.usage.prompt_tokens); +println!("{}", result.usage.completion_tokens); +println!("{}", result.usage.total_tokens); +println!("{}", result.tool_calls_count); +println!("{:?}", result.verification_summary().status); +println!("{}", result.verification_summary_text()); +``` + + + + +```ts +const result = await session.send('运行发布检查'); + +console.log(result.promptTokens); +console.log(result.completionTokens); +console.log(result.totalTokens); +console.log(result.toolCallsCount); +console.log(result.verificationStatus); +console.log(result.verificationSummaryText); +``` + + + + +```python +result = session.send('运行发布检查') + +print(result.prompt_tokens) +print(result.completion_tokens) +print(result.total_tokens) +print(result.tool_calls_count) +print(result.verification_status) +print(result.verification_summary_text) +``` + + + + +```go +result, err := session.Run(ctx, "运行发布检查") +if err != nil { + return err +} + +fmt.Println(result.Usage.PromptTokens) +fmt.Println(result.Usage.CompletionTokens) +fmt.Println(result.Usage.TotalTokens) +fmt.Println(result.ToolCallsCount) +fmt.Println(result.VerificationSummary.Status) +fmt.Println(result.VerificationSummaryText) +``` + + + + +## 追踪与验证 + + + + +```rust +let trace = session.trace_events(); +let reports = session.verification_reports(); +let summary = session.verification_summary(); +let text = session.verification_summary_text(); +``` + + + + +```ts +const trace = session.traceEvents(); +const reports = session.verificationReports(); +const summary = session.verificationSummary(); +const text = session.verificationSummaryText(); +``` + + + + +```python +trace = session.trace_events() +reports = session.verification_reports() +summary = session.verification_summary() +text = session.verification_summary_text() +``` + + + + +```go +trace, traceErr := session.TraceEvents(ctx) +reports, reportsErr := session.VerificationReports(ctx) +summary, summaryErr := session.VerificationSummary(ctx) +text, textErr := session.VerificationSummaryText(ctx) +``` + + + + +Go 方法需要跨越原生桥接边界,因此会返回 `error`;Rust、Node.js 与 Python 的 +这些进程内观察读取器是同步接口。 + +## 流式事件 + +流式输出返回带版本号的 `AgentEvent` 信封。`version`、`type`、`payload` 和可选 +的 `metadata` 是稳定且无损的字段;文本、工具名、工具输出、错误、令牌总数和 +验证摘要则是便捷投影。未来新增的未知 `type` 及其载荷仍会完整提供给宿主。 + +## 日志 + +产品遥测应优先使用结构化追踪事件和验证报告,不要解析控制台输出。 diff --git a/website/docs/v8.5.1/zh/guide/tools.mdx b/website/docs/v8.5.1/zh/guide/tools.mdx new file mode 100644 index 00000000..c7f7a704 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/tools.mdx @@ -0,0 +1,559 @@ +--- +title: '工具' +description: '工具选择、直接工具调用与验证证据' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 工具 + +A3S Code 会保留工具注册表。`toolNames()` 返回当前会话的工具表面。这个表面由 +工作区能力与会话级集成共同组装;因此非本地工作区可以有意隐藏它无法提供服务的工具。 + +工具活动通过会话事件流传给客户端。界面可以直接展示开始、输出、错误和完成状态, +不需要解析终端文本。 + +## 工具表面 + +| 层级 | 工具 | 注册规则 | +| -------------------------- | --------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 受工作区能力限制的内置工具 | `read`、`write`、`edit`、`patch`、`download`、`search`、`ls`、`bash`、`git` | 仅在 `WorkspaceServices` 声明具备所需能力时注册;`search` 提供 grep/glob 模式,并在可读取文件时增加 BM25;`download` 还要求可写的本地工作区。 | +| 运行时内置工具 | `web_fetch`、`web_search`、`batch`、`program` | 由核心工具执行器注册。`batch` 和 `program` 接收当前有作用域的调用器,因此内部工具既受会话表面限制,也继承调用方的治理作用域。 | +| 会话启动工具 | `task`、`generate_object`、`search_skills`、`Skill` | 构建 `AgentSession` 时加入。`task` 是唯一模型可见的委派 schema(含多条目扇出)。`parallel_task` 别名已移除(`HARNESS-CONV4`)。关闭手动委派后会移除委派能力。 | +| TUI 工作流工具 | `dynamic_workflow`,以及 `/login` 后可选的宿主运行时工具 | 由 `a3s code` 宿主注册。`dynamic_workflow` 通过 A3S Flow 回放支撑 `ultracode` 和 DeepResearch;宿主运行时工具属于登录后集成。 | +| 动态集成 | `mcp____` 与宿主注册工具 | 从 MCP 管理器或宿主代码发现后加入。 | + +当某个工作流依赖特定模型工具可见时,在测试或应用诊断中使用 `toolNames()` / +`toolDefinitions()` 验证。工具可见性不是安全授权。`send`、`run`、`stream` +中的模型选择工具调用会经过当前技能限制、权限策略、确认、钩子、预算、队列与超时、 +取消、递归调用保护、输出净化、制品上限与工作区路径检查。`session.tool(...)` +这类直接 SDK 调用使用另一条显式策略,见下文。`activeTools()` 回答的是另一个问题: +当前操作中有哪些工具调用正在运行。 + +## 统一调用内核 + +运行时会为每次调用标记来源: + +| 来源 | 示例 | 权限与确认策略 | +| -------------------- | ------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | +| 智能体 | 模型在 `send` 或 `stream` 中发出工具调用 | 应用完整的模型侧策略与人工确认。 | +| 受治理嵌套调用 | 模型拥有的 `batch`、`program`、工作流或公开 `InvocationRuntime` 的内部调用 | 再次进入模型侧策略,并继承调用栈、取消、预算、钩子和沙箱;环境中的宿主直接调用上下文不能改变这个来源。 | +| 运行时内部调用 | `dynamic_workflow` 启动私有 `program` 执行引擎 | 只跳过已经由 `dynamic_workflow` 完成的重复权限与人工确认;Hook、证据、预算、取消、递归检查和脚本中的实际工具仍受治理。 | +| 宿主直接调用 | `session.tool(...)`、带类型的读写及 Git 辅助方法、`session.program(...)` 与直接任务辅助方法 | 可信控制面:宿主已经选择该调用,因此跳过模型侧权限与人工确认。 | +| 可信宿主直接嵌套调用 | 宿主直接调用内置 `batch`、`program` 或动态工作流后,由它执行宿主选定的子调用 | 仅为该内置嵌套操作保留可信控制面权限;第三方工具不能通过公开 API 构造此来源。 | + +宿主直接调用的 Skill、Task 或自定义工具所创建的模型子运行会重新从“智能体”来源开始。 +公开自定义工具只能通过 `InvocationRuntime` 发起嵌套调用,而这条路径始终创建受治理 +嵌套来源;一次直接调用不会变成扩展代码可复用的授权令牌。 + +所有来源都经过同一个调用内核。前置钩子可以阻止调用;预算检查在产生副作用前执行; +队列与超时、取消、递归保护、后置钩子以及安全提供程序的输出净化仍然有效。安装 +有作用域的调用器后,嵌套调用不能回退到原始注册表。 + +宿主直接调用策略不是面向终端用户的授权系统。嵌入式应用必须先完成用户身份认证和 +授权,再把请求翻译成直接 SDK 辅助方法。关闭会话会通过会话取消作用域中止正在执行的 +宿主直接调用工具。 + +## 有边界的工具契约 + +受治理的智能体、嵌套调用和会话调用会先根据工具缓存的 JSON Schema 校验参数,再进入 +确认或副作用阶段。工具还会声明每次调用的调度能力,包括只读、幂等、可恢复、取消安全、 +分页、最大并行度和输出类型。 + +`read`、`ls`、支持分页的 `search` 模式、Git 日志/列表/差异和 `web_fetch` 会返回明确的续传游标或偏移量。 +Shell 会对两个输出流执行字节上限,同时保留开头、结尾和精确计数;沙箱宿主仍能分别 +读取 stdout 与 stderr。命令截止时间同时覆盖输出排空与 `child.wait()`,关闭输出管道 +不能绕过超时。超时和取消会终止完整的 Unix 进程组。大文件修改不会把完整前后内容塞入 +事件,而是返回有边界的预览、统一差异、哈希、大小与制品引用。 + +### 确定性的工具结果投影 + +每个会话会固定一份 `a3s.code.tool-result-transform-policy.v1` 策略。默认的保守策略保留 +开头 100 KiB,不折叠或采样内容。上下文高效预设会保留 UTF-8 安全的 64 KiB 开头与 +32 KiB 结尾,折叠至少三行完全相同的重复内容,并从超大的顶层 JSON 数组中采样最多 +32 项。 + + + + +```rust +use a3s_code_core::{tools::ToolResultTransformPolicyV1, SessionOptions}; + +let options = SessionOptions::new().with_tool_result_transform_policy( + ToolResultTransformPolicyV1::context_efficient(), +); +let session = agent + .session_builder("/repo") + .options(options) + .build() + .await?; +``` + + + + +```ts +const session = await agent.sessionAsync('/repo', { + toolResultTransformPolicy: { + schema: 'a3s.code.tool-result-transform-policy.v1', + maxOutputBytes: 100 * 1024, + headBytes: 64 * 1024, + tailBytes: 32 * 1024, + foldRepeatedLines: true, + repeatedLineThreshold: 3, + structuredSampleItems: 32, + }, +}); +``` + + + + +```python +from a3s_code import SessionOptions, ToolResultTransformPolicy + +options = SessionOptions() +options.tool_result_transform_policy = ToolResultTransformPolicy.context_efficient() +session = agent.session("/repo", options) +``` + + + + +```go +session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + ToolResultTransformPolicy: &code.ToolResultTransformPolicy{ + Schema: "a3s.code.tool-result-transform-policy.v1", + MaxOutputBytes: 100 * 1024, + HeadBytes: 64 * 1024, + TailBytes: 32 * 1024, + FoldRepeatedLines: true, + RepeatedLineThreshold: 3, + StructuredSampleItems: 32, + }, +}) +``` + + + + +投影顺序是确定的:Core 先对超大 JSON 数组采样,再折叠完全重复的行;结果仍过大时, +最后执行 UTF-8 安全的开头/结尾裁剪。`max_output_bytes` 可设为 1–100 KiB;兼容配置以外 +的策略必须在该上限内为转换标记预留 512 字节。 + +策略会写入会话快照。恢复时,未指定策略会继承快照值;显式传入不同策略则会被拒绝, +因此回放不会静默改变模型当时观察到的内容。 + +每个工具结果都包含 `metadata.a3s_tool_result_evidence`,schema 为 +`a3s.code.tool-result-evidence.v1`。证据记录 `original_bytes`、`projected_bytes`、使用 +`utf8-bytes-ceil-div-4/v1` 得到的原始/投影 Token 估算、源与投影摘要、字节/Token 差值、 +`repeat_key`、`content_ref` 和 `transform_algorithm`。`loss_mode` 为 `none`、 +`bounded_preview`、`head_tail`、`deterministic_transform` 或 `composite`。无损结果使用 +内联 SHA-256 引用;有损结果会把完整原文保存在不可变的 `a3s://tool-output/...` 制品 URI +下。这些数值是 harness 观察结果,不是 Provider 计费记录。 + +### 二进制安全的本地下载 + +`download` 把 HTTP(S) 资源写入可写的本地工作区。S3、浏览器和其他非本地后端不会 +注册它。模型选择的调用属于工作区修改操作,会进入正常权限策略与 HITL 路径;直接调用 +`session.tool(...)` 则是宿主明确做出的特权决定。 + +```ts +const result = await session.tool('download', { + url: 'https://downloads.example.com/model.bin?signature=...', + file_path: 'artifacts/model.bin', + overwrite: false, + connections: 4, + max_bytes: 536870912, + timeout: 300, + expected_sha256: + '0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef', +}); +``` + +| 参数 | 必填 | 契约 | +| ----------------- | ---- | ---------------------------------------------------------------------------------------------------------------------------- | +| `url` | 是 | 公网 `http://` 或 `https://` URL。拒绝用户信息并移除 fragment;实际请求会保留签名查询参数。 | +| `file_path` | 否 | 工作区相对目标路径。省略时依次从 `Content-Disposition`、URL 的 `file` / `filename`、路径或 `download.bin` 推断并净化文件名。 | +| `overwrite` | 否 | 仅在替换文件完整并通过校验后覆盖已有普通文件;默认 `false`。 | +| `connections` | 否 | 1–4 个 Range 并发连接。省略时按大小自适应选择;小文件或没有稳定校验器的资源只使用一个一致响应。 | +| `max_bytes` | 否 | 声明大小和实际流式字节的上限;默认 512 MiB(`536870912`),硬上限 8 GiB(`8589934592`)。 | +| `timeout` | 否 | 总截止时间(秒),覆盖重试、哈希和原子提升;默认 300,硬上限 3600。 | +| `expected_sha256` | 否 | 必须正好是 64 位十六进制字符;不匹配时不改变目标文件。 | + +共享安全 HTTP 传输会在每次重定向前检查目标,阻止 SSRF 敏感地址。直接连接会拒绝混合 +公网/私网 DNS 结果,并把该跳已验证地址固定给请求。重定向次数有上限且逐跳重新校验; +跨源重定向不会继承凭据或 `If-Range`。显式代理负责解析域名时,URL 字面地址检查仍然执行。 + +下载器会探测 Range 支持,严格校验 `Content-Range` 与响应体边界,并且只有稳定校验器 +存在时才并发获取独立分段。传输、限流和服务端失败只会有限重试;并发协议或网络尝试 +不安全时回退到单个顺序响应。数据先写入目标旁的临时文件。取消、超时、超限或摘要校验 +失败都会清理残留;只有完整同步并完成可选校验后才原子提升为目标文件。 + +结果元数据包含工作区路径、字节数、内容类型、策略、连接数、Range 支持、覆盖状态、 +安全来源锚点,以及请求校验时的摘要。签名查询参数只用于请求,在来源锚点和诊断中会被 +移除,因此不会通过工具元数据泄露。 + +### 仓库上下文模式 + +已知相关路径时,使用 `read.files` 在一次有边界的响应中批量读取,减少工具回合: + +```json +{ + "files": [ + { "path": "src/lib.rs" }, + { "path": "src/config.rs", "offset": 40, "limit": 80 } + ], + "max_output_bytes": 65536 +} +``` + +共享字节预算包含文件头和续传信息。结果保持请求顺序,单个文件不可读不会丢弃其他 +成功结果。当 `metadata.batch.truncated` 为 `true` 时,把 +`metadata.batch.continuation` 原样放入下一次调用的 `files`;其中的偏移量和剩余额度 +会从未完成位置继续,不重复已经返回的行。 + +`search` 是模型侧唯一的工作区搜索工具,必须传入 `mode` 和 `query`:正则内容搜索使用 +`grep`,路径发现使用 `glob`,原生词汇相关性排序使用 `bm25`。这些模式是分离的平面: +`grep` 不会打开持久 zvec FTS,`bm25` 也不会成为精确匹配权威。宿主显式启用 Workspace +Retrieval 后,同一个 Schema 还会提供 `semantic`,在 Session 独占的内存索引上执行 +Exact Cosine Ranking,以及 `hybrid`,对 Exact、Lexical、Symbol 与 Semantic Evidence +执行 Reciprocal-rank Fusion。关闭状态不会暴露这两个 Mode。所有模式共用 `path`;Grep、 +BM25、Semantic 与 Hybrid 使用 `include` 过滤候选文件。 + +在 `mode: "grep"` 下,`output_mode` 用于选择满足任务所需的最小结果形状: + +| 模式 | 结果 | +| -------------------- | ------------------------------------------ | +| `content` | 匹配行与可选上下文(默认) | +| `files_with_matches` | 按词法顺序、使用游标分页的匹配路径 | +| `count` | 按词法顺序、使用游标分页的每文件匹配行计数 | +| `summary` | 完整扫描后的行数与文件数,不渲染匹配内容 | + +内置工作区后端在非内容模式下不会构造随后被丢弃的匹配文本。在默认 `local-code` +下,清单支持的工作区还会在 `.a3s-code/grep-trigram` 下惰性构建进程内 trigram +候选缓存,使字面量模式在精确正则扫描前打开更少文件。非字面量模式与索引失败会失败 +开放回今日的全量扫描。S3 结果会设置 +`metadata.search.truncated`;当对象扫描上限导致总数或路径不完整时还会给出警告。 + +在 `mode: "glob"` 下,`query` 就是 glob 表达式。默认保留后端的相关性或最近使用顺序; +需要稳定的词法分页时,在游标分页前设置 `sort: "path"`。 + +在 `mode: "bm25"` 下,`query` 是普通文本。纯 Rust 的有界评分器会拆分代码标识符和 +CJK 文本,对 80 行分块排序,只返回 top-k 片段与来源锚点。它先用工作区搜索缩小候选, +最多处理 256 个文件、每文件 512 KiB、总计 16 MiB;无需数据库、嵌入模型或外部 +reranker。 + +```ts +const ranked = await session.tool('search', { + mode: 'bm25', + query: 'workspace permission policy', + path: 'core/src', + include: '*.rs', + limit: 8, + context: 2, +}); +``` + +Semantic 与 Hybrid 仍使用同一个工具,不会额外注册 Vector Database 工具: + +```json +{ + "mode": "hybrid", + "query": "where session shutdown releases temporary indexes", + "path": "core/src", + "include": "*.rs", + "limit": 8 +} +``` + +索引构建异步运行,并按 File 原子发布。返回前会重新读取当前源文件并校验 Digest;关闭 +Session 会释放全部向量。启用方式、Partial Readiness、Embedding Route、资源上限与 +Lifecycle 见[工作区检索](/guide/context#workspace-retrieval)。 + +对固定字符串修改,先以 `dry_run: true` 调用 `edit`,获取与真实写入相同的前后差异 +元数据,但不写文件。应用修改时,把预览得到的数量放入 `expected_replacements`,并可用 +`max_replacements` 设置独立上限。Dry run 会声明为只读,可安全参与 `batch` 并行执行。 + +`batch` 最多接受 32 个调用,并行度最高为 16。只有全部子调用都声明为安全、只读且幂等 +时才会扇出;修改型或能力未知的工具会串行执行。部分失败会标出失败索引,并把编排本身 +视为已完成,调用方只需重试失败项。多任务 `task` 扇出同样最多接受 32 个任务,并会先 +结算已经取消的子任务,再发布终止状态。 + +模型侧 `task` schema 始终使用包含 1–32 项的 `tasks`。一项代表聚焦子运行且可设置 +`background`;多个相互独立的项会并发执行,不能设置 `background: true`。每项接受 +`agent`、`description`、`prompt`,以及可选的 `max_steps` 和 `output_schema`。 +`min_success_count` 仅能与 `allow_partial_failure: true` 一起使用,并且必须介于 1 和 +提交任务数之间。Provider 和子运行时拥有类型化重试策略,扇出层不会根据错误文本重放 +分支。 + +### 结构门控的网页搜索 + +`web_search` 会在元数据中报告 `complete`、`partial` 或 `failed`。默认路径先执行 +无头浏览器引擎;只有合并结果未达到结构化检索要求时,才运行普通 HTTP/RSS 引擎; +前两层仍不足时,才运行原生 API。因此浏览器发现和池创建都是惰性的。A3S Code v8.5.1 +锁定 `a3s-search` v3.1.0(a3s-search v3.1.0),并默认使用打包或共享缓存中的 Moli Runtime。精简 Rust +嵌入可以关闭 default features;Chrome 和 Lightpanda 仍需显式配置。 + +Moli 会依次检查显式可执行文件、包内 Sidecar、已校验的按用户缓存和可发现的系统安装, +然后才通过 HTTPS 下载固定版本。缓存由跨进程安装锁和原子收据保护,因此多个 +`a3s-code` 进程会复用同一个安装。严格离线部署可设置 `auto_download_moli = false`。 +Linux musl 会生成 `MOLI_UNAVAILABLE`,因为上游 Moli 没有 musl 资产;请提供系统/显式 +可执行文件或选择其他后端。 + +```acl +search { + headless { + backend = "moli" + auto_download_moli = true + max_tabs = 4 + } +} +``` + +如果完整级联仍未达到结构化检索要求,最终结果会失败关闭。成功的 JSON 输出保持结果 +数组契约;要求未满足的 JSON 输出是错误,并携带类型化 +`retrieval_requirements_not_met` envelope、候选行、观测到的检索健康度和结构阈值。 +Search 不依赖外部语义验证器或重排序 API。 + +会话级 closed/open/half-open 熔断状态会跳过已知的配额、权限、限流、传输、重复空结果 +和超时故障,避免每个请求都重试;服务端的 `Retry-After` 会被保留。委派研究上下文共享 +搜索 bulkhead、有限浏览器重试预算和相同请求合并。请求级代理会到达惰性浏览器层, +`search_coalescing` 元数据会报告 leader、共享、绕过和放弃请求。显式传入 `engines` +时只执行请求的层级。含引擎错误的空结果属于失败,而不是成功的空搜索。超时、取消、 +无效参数、部分失败和限流会携带结构化错误类型;层级决策、检索健康度、引擎结果、尝试 +耗时和重试上下文保留在元数据中。 + +`web_fetch` 同样保留失败语义,不从渲染后的文本推断是否重试。请求或响应体 I/O 失败、 +HTTP 408 和 HTTP 5xx 使用类型化 `transport`;HTTP 429 使用 `rate_limited`,并在 +服务端提供时携带解析后的 `Retry-After` 延迟;工具外层截止时间使用 `timeout`。其他 +HTTP 状态错误保持普通状态失败,除非运行时掌握可以安全重试的类型化证据。 + +## 直接工具调用 + +```ts +const files = await session.glob('src/**/*.rs'); +const hits = await session.grep('PermissionPolicy'); +const status = await session.git('status'); +const output = await session.bash('cargo test -p a3s-code-core'); +const raw = await session.tool('read', { file_path: 'README.md' }); +const schemas = session.toolDefinitions(); +const hitLines = hits.split('\n').filter(Boolean).length; + +console.log(files.length, hitLines, output.length, schemas.length); +console.log(status.output); +console.log(raw.output); +``` + +直接工具调用在会话工作区下执行,应视为宿主侧特权操作。它们不更新对话记录, +因此不占用会话的单任务对话租约。`session.tool(...)`、 +`session.program(...)`、`session.git(...)`、`session.writeFile(...)`、 +`session.ls(...)`、`session.editFile(...)` 和 `session.patchFile(...)` 返回 +`ToolResult`,读取 `output`、`exitCode` 和可选的 `metadataJson`。类型化 +读取、搜索和命令行辅助方法返回更简单的值:`readFile`、`grep`、`bash` 返回字符串, +`glob` 返回字符串数组。长输出应在进入提示词前先摘要。 + +## 结构化输出:`generate_object` + +`generate_object` 工具会让配置的大语言模型生成 JSON 值,对响应执行 JSON Schema +校验,并且只在零退出码结果中返回校验后的值。它支持根对象、数组、枚举、常量、组合 +关键字和本地 `$ref` 定义。有效超时从获得模型生成准入后开始;支持活动传输预算的 +客户端会收到这份超时,并且有边界的 Schema 修复过程始终受同一个截止时间约束。 +它有两种使用方式: + +1. **智能体自主调用**:大语言模型在工具列表中看到 `generate_object`,在需要结构化输出时自行决定调用。 +2. **直接调用**:应用通过 `session.tool('generate_object', ...)` 绕过模型驱动的工具选择步骤;工具内部仍会调用配置的 LLM。 + +```ts +const result = await session.tool('generate_object', { + schema: { + type: 'object', + required: ['sentiment', 'confidence'], + properties: { + sentiment: { type: 'string', enum: ['positive', 'negative', 'neutral'] }, + confidence: { type: 'number', minimum: 0, maximum: 1 }, + }, + }, + prompt: '分类: "这个产品太棒了!"', + schema_name: 'sentiment', + mode: 'tool', + max_repair_attempts: 2, +}); + +if (result.exitCode !== 0) { + throw new Error(result.output); +} + +const { object } = JSON.parse(result.output); +// object = { sentiment: "positive", confidence: 0.95 } +``` + +### 参数 + +| 参数 | 类型 | 必填 | 说明 | +| --------------------- | ------ | ---- | ----------------------------------------------------------------------- | +| `schema` | 对象 | 是 | 用于校验输出值的 JSON Schema | +| `prompt` | 字符串 | 是 | 非空白的生成或提取指令 | +| `schema_name` | 字符串 | 否 | 1–59 个 ASCII 字母、数字、`_` 或 `-`(默认 `result`) | +| `schema_description` | 字符串 | 否 | 合成工具描述,最多 4,096 字节 | +| `system` | 字符串 | 否 | 可选系统提示,最多 32,768 字节 | +| `mode` | 字符串 | 否 | `"auto"` / `"strict"` / `"json"` / `"tool"` / `"prompt"`(默认 `auto`) | +| `max_repair_attempts` | 整数 | 否 | 0–5(默认 2) | +| `include_raw_text` | 布尔值 | 否 | 返回用于提取的 Provider 文本或工具参数(默认 `false`) | +| `timeout_ms` | 整数 | 否 | 活动生成截止时间,1,000–600,000 毫秒(默认 120,000) | + +### 模式 + +- **`tool`**:Provider 支持强制工具调用时,要求调用一个参数符合 Schema 的合成工具。 +- **`prompt`**:将 Schema 指令追加到提示词中。它适合作为纯提示词回退路径,但更依赖模型遵循指令。 +- **`auto`**:优先选择强制工具模式,不支持时退回提示词模式。 +- **`strict`**:Provider 支持时使用原生严格 JSON Schema,否则安全退回强制工具或提示词模式。 +- **`json`**:Provider 支持时使用原生 JSON 对象模式,否则安全退回强制工具或提示词模式。 + +每一种解析后的模式都会把面向 Provider 的响应 Schema 保留为仅宿主可见的校验元数据; +它不会作为额外字段序列化进 Provider 请求。Rust 组合客户端可以使用 +`structured::is_complete_streamed_value(...)`,只接受完整且通过该 Schema 校验的 +JSON 值,包括缺少终止流事件的端点。客户端还可以先检查 +`LlmClient::has_distinct_non_streaming_transport()`,再决定阻塞调用能否作为独立回退, +避免把同一种流式故障换一个方法名后再次执行。 + +### 流式 + +通过 `session.stream()` 调用时,部分对象以 `tool_output_delta` 事件发出。快照最多每 +100 毫秒发送一次;对象超过事件预算时,事件只发送字节计数,不重复携带完整值: + +```ts +const stream = await session.stream('提取所有发票...'); + +while (true) { + const { value: ev, done } = await stream.next(); + if (done) break; + if (!ev) continue; + + if (ev.type === 'tool_output_delta' && ev.toolName === 'generate_object') { + const { object_partial } = JSON.parse(ev.text); + renderProgress(object_partial); + } +} +``` + +### 修复重试 + +如果大语言模型输出未通过模式校验,工具会自动将校验错误反馈给模型并重试。这能处理缺少必填字段、枚举值错误等边界情况,无需应用层重试逻辑。 + +委派子任务可以直接使用 SDK 辅助方法,它们底层仍是同一组核心工具: + +```ts +await session.task({ + agent: 'explore', + description: '查找认证文件', + prompt: '检查认证相关文件,并返回紧凑证据列表。', +}); + +await session.tasks([ + { agent: 'explore', description: '查找测试', prompt: '定位认证测试。' }, + { + agent: 'verification', + description: '检查风险', + prompt: '审查认证边界情况。', + }, +]); +``` + +自动子智能体委派也使用同一组核心工具。`autoParallel: false` 只关闭自动并行扇出, +不会移除手动 `task` 扇出或 `session.tasks(...)`。 + +## 程序化工具调用 + +程序化工具调用不是单次直接工具调用。`program` 工具会在内嵌 QuickJS 虚拟机中运行 +受限 JavaScript 脚本;脚本定义 `async function run(ctx, inputs)`,用一个有边界的 +程序替代多轮模型工具调用。 + +不要让模型反复消耗工具回合: + +```text +grep -> read -> grep -> read -> summarize +``` + +可以让模型请求 `program` 运行一个脚本: + +```js +// search-auth.js +export default async function run(ctx, inputs) { + const hits = await ctx.grep(inputs.query, { glob: '*.rs' }); + const files = await ctx.glob('crates/**/*.rs'); + const snippets = []; + + for (const file of files.slice(0, 20)) { + const content = await ctx.readFile(file); + if (content.includes(inputs.query)) { + snippets.push({ file, preview: content.slice(0, 1200) }); + } + } + + return { + summary: `找到 ${snippets.length} 个与 ${inputs.query} 相关的候选文件`, + evidence: snippets, + rawSearch: hits, + }; +} +``` + +SDK 的 `session.program(...)` 支持内联 `source`,也支持 workspace 相对路径 `.js` 或 `.mjs` 文件: + +```js +await session.program({ + path: 'scripts/ptc/search-auth.js', + inputs: { query: 'PermissionPolicy' }, + allowedTools: ['grep', 'glob', 'read'], + limits: { + timeoutMs: 30000, + maxToolCalls: 30, + maxOutputBytes: 65536, + }, +}); +``` + +`session.program(...)` 等价于 `session.tool('program', { type: 'script', language: 'javascript', ... })`,只是使用 SDK 原生命名。如果省略 `allowedTools` / `allowed_tools`,脚本可以调用除 `program` 之外的所有已注册工具。需要更小能力面时,再显式传入 allow-list。 + +`ctx.readFile(path, options)` 返回选定文本,不包含 `read` 工具的行号锚点和续传页脚, +适合直接进行字符串处理。脚本需要完整工具结果、带行号的 `output`、退出码、元数据和 +续传证据时,应使用 `ctx.read(path, options)`。两种形式都会调用同一条受治理的 `read` +工具,并消耗相同的脚本调用预算。 + +QuickJS VM 不获得文件系统、网络、子进程或环境变量权限。脚本唯一有用的能力来自 `ctx`,这些方法会回到 A3S Code 的受控工具执行路径。PTC 会返回可读的 `ToolResult.output`,结构化数据位于 `ToolResult.metadataJson`。原始大输出不应反复塞进 prompt;应先总结发现、证据引用、风险和建议下一步。 + +在 `a3s code` 中,`DynamicWorkflowRuntime` 也会使用 PTC。默认 TUI PTC allow-list +不会允许递归调用 `program`、`dynamic_workflow` 和已移除的 `parallel_task` 别名。 +QuickJS 可以用单个 `tasks` 项调用 `task`,但会阻止直接多任务扇出。动态工作流需要本地 +并行子智能体时,会调度名为 `task` 的 Flow 步骤,由 TUI 宿主在 QuickJS 外执行。完成 +`/login` 且会话中已经注册宿主运行时工具后,动态工作流的 +PTC 步骤还可以调用 `ctx.tool("runtime", ...)`。 + +模型选择的 `dynamic_workflow` 获得授权后,其私有 QuickJS 执行引擎不需要额外的 +`program(*)` 规则。`allowed_tools` 中列出的每个实际工具仍会重新经过正常的权限、 +确认、Hook、预算、沙箱与取消路径。 + +动态工作流的结构化生成默认单路执行。只有相互独立的 `generate_object` 步骤需要扇出 +时,才把 `limits.maxConcurrentGenerations` 设置为 2–4;每个获得准入的步骤都会得到 +绑定到精确运行和步骤身份的客户端 fork。无法 fork 会话的 Provider 仍保持单路执行。 +Rust 辅助函数 `dynamic_workflow::recover_dynamic_workflow_step_output(...)` 只有在运行 +标识、原始查询和步骤标识全部匹配时,才会恢复一个已经完成的持久化步骤。它不是跨运行 +查询缓存,也不会把未完成步骤提升为成功结果。 + +## 验证 + +使用验证命令把“已经完成”转换成可检查的证据: + +```ts +const report = await session.verifyCommands('发布就绪检查', [ + { + id: 'unit', + kind: 'test', + description: '运行核心测试', + command: 'cargo test -p a3s-code-core', + required: true, + timeoutMs: 120000, + }, +]); +``` diff --git a/website/docs/v8.5.1/zh/guide/tui.mdx b/website/docs/v8.5.1/zh/guide/tui.mdx new file mode 100644 index 00000000..368bc40c --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/tui.mdx @@ -0,0 +1,377 @@ +--- +title: 'A3S Code TUI' +description: '安装、配置并使用 a3s code 终端应用' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# A3S Code TUI + +`a3s code` 是 A3S Code 的交互式终端应用。它由 +[`a3s` CLI](https://github.com/A3S-Lab/a3s) 发布,嵌入 +`a3s-code-core` 运行时,并使用 [`a3s-tui`](https://github.com/A3S-Lab/TUI) +渲染运行时事件流。 + +需要现成的终端编程智能体时使用 TUI;要构建自己的运行框架、IDE 扩展、 +服务器工作进程、工作流运行器或产品界面时使用 SDK。 + +## 界面结构 + +| 组成部分 | 仓库 | 职责 | +| ------------ | ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | +| A3S Code SDK | [A3S-Lab/Code](https://github.com/A3S-Lab/Code) | Rust 运行时 crate,以及 Node.js、Python 和 Go SDK。 | +| `a3s code` | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | 驱动 A3S Code 会话的终端编程智能体应用。 | +| A3S Use | [A3S-Lab/Use](https://github.com/A3S-Lab/Use) | 独立发布的 Browser、原生 Office、内置 OCR、可选 Office 兼容层,以及通过标准 MCP 与 Skill 投影到 Code 的签名外部应用能力。 | +| `a3s-tui` | [A3S-Lab/TUI](https://github.com/A3S-Lab/TUI) | CLI 使用的终端界面框架,不是智能体运行时。 | +| A3S 单体仓库 | [A3S-Lab/a3s](https://github.com/A3S-Lab/a3s) | 产品文档、发布编排、子模块版本固定与相关 crate。 | + +## 安装 + +先运行对应平台的一键安装脚本: + + + + +```bash +curl --proto '=https' --tlsv1.2 -LsSf \ + https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.sh | sh +``` + + + + +```powershell +[Net.ServicePointManager]::SecurityProtocol = [Net.ServicePointManager]::SecurityProtocol -bor [Net.SecurityProtocolType]::Tls12 +irm https://raw.githubusercontent.com/A3S-Lab/a3s/main/install.ps1 | iex +``` + + + + +脚本会选择当前系统与架构对应的发布包并校验 SHA-256。也可以使用 +`brew install a3s-lab/tap/a3s` 或 `cargo install a3s`。 + +安装后,在希望 Agent 检查的 Workspace 中运行: + +```bash +a3s code +a3s code resume +a3s code resume +a3s code update +``` + +顶层 `a3s update` 命令也会被 CLI 接受,并路由到同一个更新程序。 + +## 配置 + +TUI 按以下顺序发现配置: + +1. `A3S_CONFIG_FILE` +2. 从当前目录向上查找的 `.a3s/config.acl` +3. `~/.a3s/config.acl` + +TUI 使用 `.a3s/config.acl` 作为项目本地配置。SDK 示例经常使用 +`agent.acl`,因为嵌入方会通过 `Agent.create("agent.acl")` 显式传入 +配置路径。二者都是 ACL 文件;关键区别在于由谁发现文件。 +可选的 A3S OS 端点使用 `os = "https://..."` 配置。 + +`/ide` 与 `/config` 共用同一套全屏编辑器。文件树使用终端安全的文件类型与目录 +符号、语义化图标颜色和对齐的展开标记;编辑器标题、文件元数据与带分隔线的 +行号 gutter 复用同一图标来源,不依赖 emoji 或 Nerd Font。 + +不要把真实密钥、私有服务提供商 URL、租户标识或用户本地路径提交到仓库。 +公开模板应通过环境变量解析凭据: + +```acl +default_model = "provider/model-id" +os = env("A3S_OS_URL") + +providers "provider" { + apiKey = env("PROVIDER_API_KEY") + baseUrl = env("PROVIDER_BASE_URL") + + models "model-id" { + tool_call = true + limit = { + context = 128000 + output = 4096 + } + } +} +``` + +## 会话恢复 + +核心会话快照会自动保存到 +`/.a3s/tui/sessions/v1/sessions`,TUI 自己维护的逐会话状态保存在 +`/.a3s/tui/session-state/v1`。退出后,CLI 会打印完整的 +`a3s code resume ` 命令;启用颜色输出时,该命令会高亮显示。不带 id 的 +`a3s code resume` 会恢复当前工作区中最近保存的会话。 + +恢复时会保留该会话的模型及凭据来源、推理强度、执行模式 +(`default`、`plan` 或 `auto`)和语法高亮主题。如果退出中断了 durable `/goal`,启动时 +会显示“继续目标”与“保持暂停”:前者从下一次目标迭代继续,且不改变已恢复的执行模式; +后者进入会话并保持目标暂停,之后可用 `/goal resume` 继续。 + +## 文件系统优先工作流 + +TUI 面向工作区。它在仓库中启动时,可以加载 SDK 会话也会使用的文件系统优先约定: + +| 路径 | TUI 中的作用 | +| --------------------------- | ------------------------------------------------ | +| `AGENTS.md` | 加载到上下文中的项目说明。 | +| `.a3s/config.acl` | 项目本地模型、服务提供商、技能、存储和委派策略。 | +| `.a3s/agents/` | 可供 `task` 和自动委派使用的工作角色。 | +| `.a3s/skills/` 与 `skills/` | 暴露给会话的可复用项目技能。 | +| `.a3s/kb/` | TUI 知识工作流使用的项目知识库。 | + +这些文件让行为可审查,但不会绕过运行时边界。权限策略、确认提示、 +工作区检查、工具可见性、响应契约与验证仍然通过 A3S Code 执行路径。 + +## 通过 A3S Use 使用浏览器、办公文档与文字识别 + +A3S Use 是独立发布的首次使用组件。终端接管前,`a3s code` 会复用健康安装;网络和 +自动准备策略允许时,会安装经过校验的发行版。`--offline`、`A3S_OFFLINE=1` 和 +`A3S_NO_AUTO_INSTALL=1` 始终禁止该修改。准备失败不会阻止 Code 启动,并会通过 `/use` +保持可诊断。运行时与模型资源也可以显式提前准备: + +```bash +a3s install use --source release +a3s install use/browser + +# 可选的 OfficeCLI 兼容 provider;原生 Office 已内置在 Use 中。 +a3s install use/office + +# 安装或修复固定版本的本地 PP-OCRv6 模型。 +a3s install use/ocr +a3s use ocr doctor --json +``` + +当 Use 主程序就绪后,Code 会读取它带版本号的能力注册表,并同时监听代次与内容修订。 +之后服务提供商变化,以及扩展的安装、升级、启用或禁用,都会把对应 MCP 与已校验的 +`SKILL.md` 热插拔到当前会话。如果启动时缺少组件且首次准备被禁用或失败,则需要在 +显式安装后重启 Code 一次;之后的能力变化不需要重启。 + +`/use` 与 `/use status` 会显示 Use 可执行文件路径和版本、注册表收敛状态、 +服务提供商就绪状态、MCP 连接与工具数量,以及技能的校验和加载状态。`/use repair` +只打印非破坏性的修复指引,绝不会自动执行安装或改变扩展状态。 + +主编码模型有意看不到原始 `mcp__use_*` 定义;它通过 `task` 看到专用的 `use` +工作智能体,并把应用操作委派给它。Use 工作智能体只能看到 `mcp__use_*`,不能使用 +命令行、工作区、无关 MCP 或递归委派工具。受管技能只提供领域指引,不能扩大权限。 +用户可以直接说“用浏览器检查这个页面”“用 Office 更新这个工作簿”或“对这张扫描图做 +文字识别”,无需手写工具名。能力不可用时,工作智能体会返回带类型的失败结果,不会 +换用其他工具兜底。默认浏览器接口包含诊断和有边界的安装工具;缺少受管浏览器时, +只能在父 TUI 确认后由 Use 工作智能体请求安装。 + +| 能力 | Code 接口 | 服务提供商与确认行为 | +| ----------- | --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | +| 浏览器 | `mcp__use_browser__*` | 使用已发现的浏览器,或通过需要父级 HITL 的修改请求安装受管服务提供商。封闭会话内的只读检查可直接执行;导航、联网读取、输入、点击与提交需要 HITL。 | +| 原生 Office | `mcp__use_office__*` | 使用带修订与冲突检查的有界原生 Office 包内核。封闭环境内的读取可直接执行;修改与破坏性保存需要 HITL。 | +| 内置 OCR | `mcp__use_ocr__*` | 通过 ONNX Runtime 在本地运行固定版本的 `PP-OCRv6_small` 检测与识别模型。诊断与提取均为封闭环境只读工具,源文件字节不会离开设备。 | + +MCP 行为元数据只能提升风险,不能降低父级策略。只有同时声明 +`readOnlyHint=true`、`openWorldHint=false` 且不是破坏性操作时,才不需要额外确认。 +缺少注解、开放世界、修改、破坏性或提交风险都会向父 TUI 请求确认;父级拒绝始终有效。 +Office 修改绝不会被自动重试,尤其 `use.office.outcome_unknown` 表示变更可能已经 +落地,工作智能体必须保留证据并停止。 + +Code 也不会把 MCP 结果压扁成纯文本:`outputSchema`、`structuredContent`、图片、 +文本或二进制资源、协议元数据和带 SHA-256 的有界制品都会进入正常的对话记录和 +工具结果路径。 + +## 运行时事件流 + +TUI 将 `AgentSession::stream()` 视为事实来源。运行时事件驱动对话文本、工具调用 +进度、规划状态、批准提示、记忆视图、Git 和文件面板、验证证据与可回放的运行状态。 + +`Ctrl+T` 打开完整、实时的语义会话记录,其中包括用户与 +助手消息、计划、每个工具的完整生命周期与完整输出、子智能体状态,以及当前仍在 +流式生成的 Markdown 尾部。 + +这个分层对产品构建者很重要:TUI 只是运行时上方的一个有主张的客户端。 +SDK 嵌入方可以基于同样的会话、工具、事件、持久化与验证 API 构建不同控制面。 + +## HITL 权限确认 + +HITL 是执行边界,不是按工具名称粗略分类。Default 模式会让受约束的 Workspace +文件修改和沙箱内普通命令保持流畅;显式主机 Shell、缺少沙箱时的 Shell、受保护控制 +元数据、修改型 Git 操作和带外部副作用的调用会在执行前暂停。Plan 模式拒绝修改, +Auto 模式永远不会打开确认界面,无法证明安全的操作会直接失败。 + +确认界面显示规范化后的真实参数,并提供四个明确决定: + +1. 仅允许这一次。 +2. 在当前会话中允许这一精确能力。 +3. 把这一精确能力规则写入项目的 `.a3s/permissions.acl`。 +4. 拒绝,并把原因返回给 Agent。 + +`/permissions` 可以检查并撤销会话或项目授权。项目授权使用原子 ACL 重写;撤销只影响 +之后的检查,不会取消已经开始执行的工具。 + +## 本地下载 + +在可写的本地工作区中,模型可以调用 `download`,把公网 HTTP(S) 资源流式写入仓库。 +它会显示为普通修改型工具事件,并遵循当前模式:Default 可以暂停等待 HITL,Plan 拒绝 +修改,Auto 只允许有边界且位于工作区内的形式。 + +工具卡会展示最终工作区路径、字节数、顺序或并发 Range 策略,以及可选的已验证 +SHA-256,但不会显示签名 URL 查询参数。下载先使用相邻临时文件,取消或失败会清理部分 +数据;只有满足字节数和超时上限并通过可选 `expected_sha256` 校验后,结果才会原子出现 +在目标路径。完整参数和默认值见[工具](/guide/tools#二进制安全的本地下载)。 + +## 代码智能 + +Agent 和 TUI `/ide` 共用一个只读代码智能运行时。目前支持 Rust 与 +TypeScript/JavaScript;启动 `a3s code` 前应安装对应的 `rust-analyzer` 或 +`typescript-language-server`。缺少某个语言服务只会让该语言显示为不可用或降级, +不会关闭编辑器、普通文件工具或其他正常语言。 + +语义结果始终基于磁盘上已保存的文件。未保存缓冲区不会发布给共享运行时,界面会明确 +显示“已保存版本”,保存后才刷新 Workspace 清单与语义结果。 + +在 `/ide` 中按 `:` 后可使用: + +| 命令 | 结果 | +| ------------------------ | --------------------------------------------- | +| `:status` | 语言状态与协商后的能力。 | +| `:symbols [query]` | 当前文件符号,或有上限的 Workspace 符号搜索。 | +| `:definition` | 光标位置的定义。 | +| `:declaration` | 光标位置的声明。 | +| `:references` | 光标位置的引用。 | +| `:implementations` | 光标位置的实现。 | +| `:diagnostics` | 当前已保存文件的诊断。 | +| `:diagnostics workspace` | 有上限的 Workspace 诊断。 | + +Agent 通过 `code_symbols`、`code_navigation` 和 `code_diagnostics` 使用同一能力; +源码读取、文本搜索和修改仍只经过 `read`、统一 `search` 的 grep 模式、`edit` 与 +`patch`,代码智能不会创建第二条修改路径。 + +## 跨会话上下文检索 + +本地 `ctx` 已安装并完成索引时,TUI 在启动时启用两层召回:长期 Memory 保存经过筛选的 +稳定信息,`/ctx` 则搜索跨工具、跨会话的原始历史。它用于找回以前的决定、命令、错误 +和测试结果,而不是重新推断已经完成的工作。 + +| 命令 | 行为 | +| --------------- | -------------------------------------------------------------------- | +| `/ctx ` | 搜索本地会话索引,并显示最多八个可选择结果。 | +| `/ctx ` | 拉取第 n 个结果附近的有界记录,并一次性附加到下一条消息。 | +| `/ctx save ` | 保存为长期情节记忆,并保留 `ctx_event_id` 与 `ctx_session_id` 回链。 | + +附加的历史记录会去除终端控制字符、限制大小,并作为不可信引用文本包裹;其中旧指令不会被 +当成当前用户指令执行。保存到 Memory 的结果带有 `source=ctx` 来源,因此可以从记忆追溯 +到原始会话。 + +## 渐进式 API 与 Runtime 工具 + +这两项能力只在配置 A3S OS 并通过 `/login` 登录后出现。渐进式 API 使用一个按当前账号 +权限过滤的入口,按需执行: + +```text +list → search → describe → execute +模块 找操作 读单个 Schema 执行 +``` + +模型先发现模块,再搜索操作,只为将要调用的那一个操作加载完整输入输出 Schema,因此 +无需把整个平台能力目录塞进会话上下文。`execute` 使用 `shaped=true` 时可以返回受信任 +视图,TUI 会把有效 `.view` 或 `viewUrl` 显示为内联“打开视图”操作。 + +登录后注册的 `runtime` 工具用于 A3S OS Function as a Service 批执行。它按 UUID 或名称 +解析 tool-kind Worker,把相互独立的输入作为一个批任务提交,流式返回每项进度,最后 +聚合所有结果。未登录时该工具不会出现在模型工具列表中;本地文件、Shell、MCP、 +`task`、`dynamic_workflow`、Memory 与 `/ctx` 仍可正常使用。 + +## 斜杠命令 + +内置斜杠命令包括: + +| 命令 | 作用 | +| -------------------------------------- | -------------------------------------------------------------------- | +| `/model` | 在已配置模型和已登录账号模型之间切换。 | +| `/init` | 分析工作区并生成 `AGENTS.md`。 | +| `/config` | 在编辑器中打开当前生效的 ACL 配置。 | +| `/use` / `/use status` / `/use repair` | 查看实时 A3S Use 能力状态,或打印不会自动执行的显式修复指引。 | +| `/theme` | 切换代码高亮主题。 | +| `/flow` | 选择或草拟工作流资源,用于本地开发与可选宿主集成。 | +| `/agent` | 选择智能体定义,用于本地开发与可选宿主集成。 | +| `/mcp` | 选择 MCP 服务资源,用于本地开发与可选宿主集成。 | +| `/skill` | 选择技能资源,用于本地开发与可选宿主集成。 | +| `/okf` | 选择 OKF 包,用于本地开发与知识包管理。 | +| `/login` 与 `/logout` | 登录或退出 `os` 端点对应的 A3S OS 账号。 | +| “打开视图” | 点击内联按钮,在原生窗口中打开最近验证过的本地报告或可信运行时视图。 | +| `/plugin` 与 `/reload` | 启用、禁用并重新扫描技能与插件。 | +| `/ide` | 打开工作区文件树和代码查看器。 | +| `/memory` | 浏览长期记忆。 | +| `/kb` | 把文本、文件或目录加入项目知识库。 | +| `/ctx` | 搜索历史会话、附加结果,或保存到记忆。 | +| `/effort` | 调整对所有服务提供商生效的宿主预算,并在其支持时传递原生推理强度。 | +| `/compact` | 总结并压缩对话上下文。 | +| `/goal ` | 为会话设置持续目标。 | +| `/goal resume` | 继续会话恢复时保持暂停的持久目标。 | +| `/loop` | 自动继续执行任务,直到完成或停止。 | +| `/sleep` | 把今天的工作整理进长期记忆。 | +| `/help` | 显示命令与快捷键。 | +| `/fork` | 从当前点分支出新的已保存会话。 | +| `/clear` | 重置对话。 | +| `/auto` | 切换到自动批准模式。 | +| `/update` | 将 CLI 升级到最新发行版。 | +| `/exit` | 退出 `a3s code`。 | + +部分命令只能在会话空闲时执行,因为它们会改变对话、模型、上下文或进程状态。 + +## 推理强度与深度研究 + +| 级别 | 推理预算 | 工具轮次 | 续写次数 | 并行任务数 | +| ----------- | -------: | -------: | -------: | ---------: | +| `low` | 2,048 | 240 | 4 | 4 | +| `medium` | 8,192 | 800 | 8 | 8 | +| `high` | 16,384 | 1,200 | 12 | 12 | +| `xhigh` | 32,768 | 1,800 | 16 | 16 | +| `max` | 65,536 | 2,400 | 24 | 24 | +| `ultracode` | 65,536 | 3,200 | 32 | 32 | + +`low` 到 `max` 保持 Codex 原生 `reasoning.effort` 映射,并关闭运行时自动委派; +`task` 仍可被明确调用,多个独立 `tasks[]` 项会并发扇出。只有 `ultracode` 会启用自动委派、 +自动规划与动态工作流指引。仅用于生成最终回答的合成续写不会再次启动子智能体。 + +深度研究当前先收集有数量上限的网页种子,再根据复杂度选择直接快速路径,或本地 +`task` 多项扇出轨道与有上限的后续回合。在函数即服务支持就绪前,A3S OS 运行时的 +工具调用扇出保持禁用。 + +在 TUI 输入 `? ` 可启动深度研究。默认情况下,它允许直接网页访问和委派网页 +工具;查询中包含 `no web`、`do not use web` 或 `不要联网` 时,这些网页路径会被 +禁用,同时保留本地证据收集与后续深度。 + +每次运行的制品位于: + +```text +.a3s/research//report.md +.a3s/research//index.html +``` + +已完成报告必须能够追溯到本次运行收集的来源。若证据收集或合成无法满足这个条件, +A3S 会写入明确标注的低置信度深度研究恢复报告,而不是把无证据结论伪装成已完成报告。 +本地 HTML 由仅环回地址、无需认证的查看器提供,因此退出登录时也可以通过 +“打开视图”打开。 + +无界面命令使用相同工作流: + +```bash +a3s code deepresearch [--local|--os] [--local-only|--web] +``` + +`--local` 只表示本地编排,仍允许网页证据;`--local-only`(也接受 `--offline`) +会强制禁用网络,`--web` 会显式启用网页与工作区证据。自然语言中的 `no web` +仅作为兼容性兜底。`--os` 当前禁用。无界面命令的所有阶段共同消耗 +一个 60 秒绝对时限,其中宿主直接执行最多使用 20 秒;失败后仅用剩余预算执行 +最多一轮兜底,并为取消与制品收尾保留最后 3 秒。若最终只能生成恢复报告, +制品仍会落盘,但命令会以错误状态结束。 + +## 相关页面 + +1. [文件系统优先](/guide/filesystem-first) +2. [会话](/guide/sessions) +3. [命令](/guide/commands) +4. [工具](/guide/tools) +5. [安全](/guide/security) diff --git a/website/docs/v8.5.1/zh/guide/verification.mdx b/website/docs/v8.5.1/zh/guide/verification.mdx new file mode 100644 index 00000000..e133cbd4 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/verification.mdx @@ -0,0 +1,356 @@ +--- +title: '验证' +description: '用验证命令和报告证明一个回合已完成,而不是轻信模型的声明' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 验证 + +运行时将"完成"视为必须被**证明**的事实,而不仅仅是被声明的结果。当模型说某个任务已完成时,这句话本身毫无价值。验证把声明转化为证据:你声明一组*必须*成功的命令,运行时执行它们,结果会携带一份你可以检视、据以拦截或呈现给用户的报告。 + +验证是会话级的。Rust 核心运行每条命令,记录其退出状态和输出,并将每份报告汇总为一份随回合结果一同返回的摘要。 + +产品界面可以先展示交付摘要,再列出支持结论的命令输出和文件级改动。 + +## 运行验证命令 + +一条验证命令就是一个小而具名的检查:一个 `id`、一个 `kind`、一段可读的 `description`,以及要运行的 `command`。当某个失败应被视为硬失败而非警告时,将该检查标记为 `required`。 + + + + +```rust +use a3s_code_core::verification::VerificationCommand; + +let commands = vec![ + VerificationCommand::required( + "build", + "build", + "项目可以编译", + "cargo build --all-features", + ) + .with_timeout_ms(120_000), + VerificationCommand::required( + "tests", + "test", + "单元测试通过", + "cargo test", + ), +]; +let report = session + .verify_commands("release-readiness", &commands) + .await?; +println!("{report:#?}"); +``` + + + + +```ts +const report = await session.verifyCommands('release-readiness', [ + { + id: 'build', + kind: 'build', + description: 'Project compiles', + command: 'cargo build --all-features', + required: true, + timeoutMs: 120000, + }, + { + id: 'tests', + kind: 'test', + description: 'Unit tests pass', + command: 'cargo test', + required: true, + }, +]); + +console.log(report); +``` + + + + +```python +report = session.verify_commands('release-readiness', [ + { + "id": "build", + "kind": "build", + "description": "Project compiles", + "command": "cargo build --all-features", + "required": True, + "timeout_ms": 120000, + }, + { + "id": "tests", + "kind": "test", + "description": "Unit tests pass", + "command": "cargo test", + "required": True, + }, +]) + +print(report) +``` + + + + +```go +report, err := session.VerifyCommands(ctx, "release-readiness", []code.VerificationCommand{ + { + ID: "build", + Kind: "build", + Description: "项目可以编译", + Command: "cargo build --all-features", + Required: true, + TimeoutMS: code.Ptr(uint64(120000)), + }, + { + ID: "tests", + Kind: "test", + Description: "单元测试通过", + Command: "cargo test", + Required: true, + }, +}) +if err != nil { + return err +} +fmt.Println(report) +``` + + + + +`subject`(此处为 `release-readiness`)为这一批检查命名,使同一会话内的多次验证在报告中保持彼此独立。 + +## 读取回合结束后的摘要 + +每个回合的 `send()` 结果同样携带只读的验证字段,因此你无需单独发起一次验证调用即可据结果拦截。用这些字段判断该回合是否真正完成了它所声称的工作。 + + + + +```rust +let result = session.send("应用修复并运行检查", None).await?; +let summary = result.verification_summary(); + +println!("{:?}", summary.status); +println!("{}", summary.pending_required_check_count); +println!("{}", summary.failed_check_count); +println!("{}", summary.report_count); +println!("{}", result.verification_summary_text()); + +if summary.failed_check_count > 0 { + return Err(a3s_code_core::CodeError::Session( + "当前回合声称已完成,但验证失败".to_string(), + )); +} +``` + + + + +```ts +const result = await session.send('Apply the fix and run the checks'); + +console.log(result.verificationStatus); +console.log(result.pendingVerificationCount); +console.log(result.failedVerificationCount); +console.log(result.verificationReportCount); +console.log(result.verificationSummaryText); + +if (result.failedVerificationCount > 0) { + throw new Error('Turn reported done but verification failed'); +} +``` + + + + +```python +result = session.send('Apply the fix and run the checks') + +print(result.verification_status) +print(result.pending_verification_count) +print(result.failed_verification_count) +print(result.verification_report_count) +print(result.verification_summary_text) + +if result.failed_verification_count > 0: + raise RuntimeError('Turn reported done but verification failed') +``` + + + + +```go +result, err := session.Run(ctx, "应用修复并运行检查") +if err != nil { + return err +} + +summary := result.VerificationSummary +fmt.Println(summary.Status) +fmt.Println(summary.PendingRequiredCheckCount) +fmt.Println(summary.FailedCheckCount) +fmt.Println(summary.ReportCount) +fmt.Println(result.VerificationSummaryText) + +if summary.FailedCheckCount > 0 { + return errors.New("回合声称完成,但验证失败") +} +``` + + + + +## 检视报告与摘要 + +除了逐回合的字段之外,会话还暴露完整的报告集合、一份结构化摘要、可用的预设,以及一份可读的概要。该概要是向人展示某回合*为何*通过或失败的最快方式。 + + + + +```rust +let reports = session.verification_reports(); +let summary = session.verification_summary(); +let presets = session.verification_presets(); +let text = session.verification_summary_text(); + +println!( + "{} 份报告,状态 {:?},{} 个预设", + reports.len(), + summary.status, + presets.len() +); +println!("{text}"); +``` + + + + +```ts +import { formatVerificationSummary } from '@a3s-lab/code'; + +const reports = session.verificationReports(); +const summary = session.verificationSummary(); +const presets = session.verificationPresets(); + +// 会话辅助方法和独立格式化函数都能生成可读文本。 +console.log(session.verificationSummaryText()); +console.log(formatVerificationSummary(summary)); +``` + + + + +```python +reports = session.verification_reports() +summary = session.verification_summary() +presets = session.verification_presets() + +# 会话辅助方法返回可直接打印的可读摘要。 +print(session.verification_summary_text()) +``` + + + + +```go +reports, err := session.VerificationReports(ctx) +if err != nil { + return err +} +summary, err := session.VerificationSummary(ctx) +if err != nil { + return err +} +presets, err := session.VerificationPresets(ctx) +if err != nil { + return err +} +text, err := session.VerificationSummaryText(ctx) +if err != nil { + return err +} + +fmt.Println(len(reports), summary.Status, len(presets)) +fmt.Println(text) +``` + + + + +`verificationPresets()` 返回根据 workspace 文件推断出的检查模板,例如 +`Cargo.toml`、`package.json`、`pyproject.toml` 和 `go.mod`。请把它们当作起点: +在用来拦截发布或用户可见自动化之前,应审查命令、超时和 required 标记是否适合该项目。 + +## A3S Code 自身如何完成资格认证 + +回合验证回答的是一次智能体任务是否产生了它所声称的结果。仓库资格认证回答的是另一 +个问题,即每项公开的 A3S Code 能力是否仍然在 Core、各语言 SDK、资源上限和受支持的 +部署面上满足合同。仅有一次绿色编译无法回答这个问题。 + +仓库把证据分成四类: + +| 证据类别 | 能证明什么 | 不能证明什么 | +| -------------------- | ---------------------------------------------------------------------------------- | --------------------------------------- | +| 确定性正确性 | 用固定 Oracle 检查启用条件、成功行为、非法输入、权限、取消、生命周期、顺序和清理 | 真实 Provider、浏览器或对象存储是否可用 | +| 确定性资源门禁 | 限定调用、重试、记录、字节、队列、候选项、工具轮次与留存状态 | 每台机器上的绝对耗时 | +| Release 性能资格认证 | 对稳定本地工作记录 Release Build 的 p50、p95、最大值、资源计量、负载参数和机器信息 | 远程模型或公共搜索服务的延迟 | +| 外部资格认证 | 在已记录条件下验证指定真实模型、浏览器、Collector 或存储服务的兼容性 | Hermetic 可复现性或普遍性能结论 | + +能力台账把 README 宣称的 27 个产品领域逐项连接到可执行证据,并明确显示任何尚未解决 +的缺口。能力地图发生变化而台账没有同步时,CI 会直接失败。Node.js 和 Python 门禁会先 +构建并加载 Native Module,再通过公开 Wrapper 执行测试;仅通过 Rust `cargo check` +不算 SDK Runtime 证据。Go 会通过带版本的 Bridge 并使用 Race Detector 运行。 + +性能检查会区分工作放大和耗时。普通 CI 拦截 Provider Request、Vector Byte、Scratch +Space、Retry 与关闭后留存等确定性上限。专用 Release Profile 工作流会预热并重复采样, +输出 Machine-readable JSON,并将其作为 CI Artifact 保留。依赖网络的 DeepSeek 与浏览器 +耗时会单独报告,因为把这些时间混入本地执行后,就无法区分 Code Regression 与 Provider +或网络波动。 + +v8.5.1 的真实模型矩阵会运行资格 ACL 声明的每个模型,验证模型选择的工具与 Hook +参数改写、有证据门禁的多文件编码任务、自动与显式 SubAgents、Skill 发现和执行、 +并发 PTC 读取、持久 A3S Flow 回放,以及公开 steer/interrupt。确定性 Fixture 还会 +独立覆盖相同控制路径,包括过期与幂等回执、拒绝、预算、取消和清理。 + +### 最新受控资格认证 + +2026-08-18 的 Release Profile 在四个逻辑 CPU 的 x86-64 Linux Runner 上运行,六份报告 +全部通过: + +| Profile | 固定负载 | 实测 p95 | 目标 | +| ------------------- | -------------------------------------------- | ------------------------------: | ---------------: | +| Agent 收敛 | 四个完成、Guard 与恢复 Case | 4/4 Case | 4/4 Case | +| Workspace Retrieval | 25,000 × 384 Exact / Deterministic Hybrid | 15.590 / 38.506 ms | ≤ 30 / 100 ms | +| Flow / State Graph | 1,000 Step Projection / 11,008 Record Replay | 130.067 / 125.526 ms | 均 ≤ 2,000 ms | +| Code Intelligence | 5,000 File;Cold / Warm Workspace Symbol | Cold 754.397 ms / Warm 0.519 ms | ≤ 5,000 / 250 ms | +| Context / Memory | 25,000 Context Input / 2,500 Memory | 136.740 / 0.123 ms | ≤ 500 / 250 ms | +| File Persistence | 1,272,624 Byte 同步 Save / Load | 338.887 / 1.028 ms | ≤ 1,000 / 500 ms | + +Provider Request Amplification、Vector 与 Rerank Byte、RSS Delta、Serialized Graph Byte、 +Process Cleanup、无累积覆盖写入与删除清理等资源门禁也全部通过。配套 Hermetic CI 完成了 +MinIO Roundtrip、受控 HTTPS 上的生产 Chrome/CDP 与 Google Parser 路径,以及本地 +OpenTelemetry Collector 对指定 Service/Span 的精确接收。 + +[性能资格认证记录](https://github.com/A3S-Lab/Code/blob/main/manual/PERFORMANCE_QUALIFICATION.md) +包含 p50/p95/Max、精确包含规则、机器信息、资源计数、Workflow 链接与 Artifact SHA-256 +Digest。这些数据是锁定 Profile 的回归上限,不是所有硬件或远程服务的 SLA。 + +当前证据台账、缺口关闭情况、外部边界、资格认证命令和完成规则见 +[能力验证与性能合同](https://github.com/A3S-Lab/Code/blob/main/manual/CAPABILITY_VERIFICATION.md)。 +只要台账中仍有未解决的 Code-owned Gap,就不能声称仓库级目标已经完成。 + +## 为何重要 + +没有验证,一次智能体运行止于模型的一面之词。有了验证,运行止于可观测的证据:编译通过的构建、跑通的测试套件、保持沉默的代码检查器。摘要文本提供审计轨迹;结果上的计数让你能在自动化中以失败为默认(fail closed)。 + +## 相关 + +- [遥测](/guide/telemetry) —— 将追踪事件和验证报告作为运行时证据进行检视。 +- [限制](/guide/limits) —— 在验证运行之前限定一个回合能完成多少工作量。 diff --git a/website/docs/v8.5.1/zh/guide/workspace-backends.mdx b/website/docs/v8.5.1/zh/guide/workspace-backends.mdx new file mode 100644 index 00000000..81e61303 --- /dev/null +++ b/website/docs/v8.5.1/zh/guide/workspace-backends.mdx @@ -0,0 +1,528 @@ +--- +title: '工作区后端' +description: '让 A3S Code 会话运行在本地文件、S3 兼容对象存储和远端 Git 服务之上。' +--- + +import { Tab, Tabs } from '@rspress/core/theme'; + +# 工作区后端 + +工作区后端决定内置工作区工具从哪里读写文件。默认后端是以会话工作区为根目录的 +本地文件系统。四种 SDK 都提供显式本地后端、S3 兼容对象存储后端,以及可选的 +HTTP/JSON 远端 Git 配置。Node.js 与 Python 使用后端对象,Go 使用值配置。 + +当宿主负责工作区放置时使用这组能力:本地开发、浏览器或容器工作区、对象存储 +工作区、托管会话等。 + +## 能力矩阵 + +| 后端 | 文件工具 | 搜索工具 | 命令行与本地 Git | +| ---------------------------------- | -------------------------------------------------- | --------------------------------------- | -------------------------------------- | +| 默认本地工作区 | `read`、`write`、`edit`、`patch`、`download`、`ls` | `search`(`grep`、`glob`、`bm25` 模式) | `bash`、`git` | +| `LocalWorkspaceBackend` | 与默认本地工作区相同 | 与默认本地工作区相同 | 与默认本地工作区相同 | +| `S3WorkspaceBackend` | `read`、`write`、`edit`、`patch`、`ls` | 设置 `searchEnabled` 后可使用 `search` | 不注册 | +| `S3WorkspaceBackend` + `remoteGit` | S3 文件工具 | 可选降级版 S3 搜索 | 通过远端 Git 提供 `git`,不提供 `bash` | + +对象存储不能执行本地进程。不要承诺 S3 工作区上可以直接运行命令,除非宿主通过 +MCP 或 A3S Box 额外提供隔离后的执行能力。 + +## 本地后端 + +显式本地后端适合宿主希望本地与远程会话都使用同一组选项接口的场景。 + + + + +```rust +use a3s_code_core::{Agent, SessionOptions, WorkspaceServices}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let agent = Agent::new("agent.acl").await?; + let backend = WorkspaceServices::local("/repo"); + let session = agent + .session_builder("/repo") + .options(SessionOptions::new().with_workspace_backend(backend)) + .build() + .await?; + + println!("{}", session.read_file("Cargo.toml").await?); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, LocalWorkspaceBackend } from '@a3s-lab/code'; + +const agent = await Agent.create('agent.acl'); +const session = agent.session('/repo', { + workspaceBackend: new LocalWorkspaceBackend('/repo'), +}); +``` + + + + +```python +from a3s_code import Agent, LocalWorkspaceBackend, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = LocalWorkspaceBackend("/repo") +session = agent.session("/repo", opts) +``` + + + + +工作区路径本身会选择默认本地后端。当本地与远端会话需要共用同一种选项结构时, +可使用 `WorkspaceBackendConfig`: + +```go +package main + +import ( + "context" + "fmt" + "log" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + session, err := agent.Session(ctx, "/repo", &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "local", + Root: "/repo", + }, + }) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + content, err := session.ReadFile(ctx, "go.mod", nil) + if err != nil { + log.Fatal(err) + } + fmt.Println(content) +} +``` + + + + +### 本地下载目标 + +可写的本地后端还会注册 `download`。工具接收公网 HTTP(S) `url`,并且只能写入工作区 +相对 `file_path`;省略路径时会推断并净化文件名。S3 和其他非本地后端不会注册它,因为 +相邻原子临时文件、取消清理、符号链接检查和最终提升都依赖本地文件系统语义。 + +`overwrite` 默认 `false`。传输默认使用 512 MiB 的 `max_bytes` 上限和 300 秒 +`timeout`,硬上限分别是 8 GiB 与 3600 秒。Range 并发会自适应选择,也可以明确限制为 +1–4 个连接;`expected_sha256` 可要求在提升目标前匹配 64 位十六进制摘要。传输与元数据 +细节见[工具](/guide/tools#二进制安全的本地下载)。 + +## S3 后端 + +`S3WorkspaceBackend` 会把内置文件工具指向任意 S3 兼容服务,包括 AWS S3、MinIO、 +RustFS、Cloudflare R2 和 Backblaze B2。 + + + + +Rust crate 需要启用 `s3` feature。 + +```rust +use std::env; + +use a3s_code_core::{Agent, S3BackendConfig, SessionOptions, WorkspaceServices}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let bucket = env::var("WORKSPACE_S3_BUCKET") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let access_key = env::var("S3_ACCESS_KEY_ID") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let secret_key = env::var("S3_SECRET_ACCESS_KEY") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + + let mut config = + S3BackendConfig::new(bucket, "sessions/example", access_key, secret_key) + .region(env::var("WORKSPACE_S3_REGION").unwrap_or_else(|_| "us-east-1".into())) + .force_path_style(true) + .enable_search(false); + if let Ok(endpoint) = env::var("WORKSPACE_S3_ENDPOINT") { + config = config.endpoint(endpoint); + } + + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("s3://workspace-bucket/sessions/example") + .options( + SessionOptions::new().with_workspace_backend(WorkspaceServices::s3(config)), + ) + .build() + .await?; + + println!("{}", session.read_file("README.md").await?); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, S3WorkspaceBackend } from '@a3s-lab/code'; + +const backend = new S3WorkspaceBackend({ + endpoint: process.env.WORKSPACE_S3_ENDPOINT, + region: process.env.WORKSPACE_S3_REGION ?? 'us-east-1', + accessKeyId: process.env.S3_ACCESS_KEY_ID!, + secretAccessKey: process.env.S3_SECRET_ACCESS_KEY!, + bucket: process.env.WORKSPACE_S3_BUCKET!, + prefix: 'sessions/example', + forcePathStyle: true, + searchEnabled: false, +}); + +const agent = await Agent.create('agent.acl'); +const session = agent.session('s3://workspace-bucket/sessions/example', { + workspaceBackend: backend, +}); +``` + + + + +```python +import os + +from a3s_code import Agent, S3WorkspaceBackend, SessionOptions + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = S3WorkspaceBackend( + bucket=os.environ["WORKSPACE_S3_BUCKET"], + prefix="sessions/example", + access_key_id=os.environ["S3_ACCESS_KEY_ID"], + secret_access_key=os.environ["S3_SECRET_ACCESS_KEY"], + endpoint=os.environ.get("WORKSPACE_S3_ENDPOINT"), + region=os.environ.get("WORKSPACE_S3_REGION", "us-east-1"), + force_path_style=True, + search_enabled=False, +) +session = agent.session("s3://workspace-bucket/sessions/example", opts) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "os" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + forcePathStyle := true + searchEnabled := false + session, err := agent.Session( + ctx, + "s3://workspace-bucket/sessions/example", + &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "s3", + S3: &code.S3BackendConfig{ + Endpoint: os.Getenv("WORKSPACE_S3_ENDPOINT"), + Region: os.Getenv("WORKSPACE_S3_REGION"), + AccessKeyID: os.Getenv("S3_ACCESS_KEY_ID"), + SecretAccessKey: os.Getenv("S3_SECRET_ACCESS_KEY"), + Bucket: os.Getenv("WORKSPACE_S3_BUCKET"), + Prefix: "sessions/example", + ForcePathStyle: &forcePathStyle, + SearchEnabled: &searchEnabled, + }, + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + content, err := session.ReadFile(ctx, "README.md", nil) + if err != nil { + log.Fatal(err) + } + fmt.Println(content) +} +``` + + + + +S3 搜索需要显式启用。启用后,`grep` / `glob` 会降级为对象列表和有上限的下载。 +当存储桶较大或端点会限流时,应配置 `maxObjectsScanned`、 +`maxGrepBytesPerObject` 和 `searchConcurrency`。 + +### S3 选项 + +| Node.js 选项 | Python 选项 | 必填 | 作用 | +| ----------------------- | --------------------------- | ---- | ------------------------------------------------ | +| `bucket` | `bucket` | 是 | 存储工作区对象的 S3 存储桶。 | +| `prefix` | `prefix` | 是 | 存储桶内的逻辑工作区根;使用 `""` 表示存储桶根。 | +| `accessKeyId` | `access_key_id` | 是 | 访问密钥标识,通常从宿主环境读取。 | +| `secretAccessKey` | `secret_access_key` | 是 | 访问密钥,通常从宿主环境或密钥管理系统读取。 | +| `endpoint` | `endpoint` | 否 | 自定义 S3 兼容端点;AWS S3 默认端点可省略。 | +| `region` | `region` | 否 | 区域;省略时默认为 `us-east-1`。 | +| `sessionToken` | `session_token` | 否 | 使用临时凭据时的 STS 会话令牌。 | +| `forcePathStyle` | `force_path_style` | 否 | MinIO、RustFS 和多数非 AWS 端点通常设为 `true`。 | +| `maxReadBytes` | `max_read_bytes` | 否 | 单次读取的大小上限;默认 10 MiB。 | +| `searchEnabled` | `search_enabled` | 否 | 启用降级版 S3 `grep` / `glob`;默认为 `false`。 | +| `maxObjectsScanned` | `max_objects_scanned` | 否 | 单次搜索扫描对象数上限;仅启用搜索时使用。 | +| `maxGrepBytesPerObject` | `max_grep_bytes_per_object` | 否 | `grep` 的单对象下载上限;仅启用搜索时使用。 | +| `searchConcurrency` | `search_concurrency` | 否 | `grep` 时并发下载对象数;仅启用搜索时使用。 | + +Go 在 `S3BackendConfig` 上提供同一组字段:`Bucket`、`Prefix`、 +`AccessKeyID`、`SecretAccessKey`、`Endpoint`、`Region`、`SessionToken`、 +`ForcePathStyle`、`MaxReadBytes`、`SearchEnabled`、`MaxObjectsScanned`、 +`MaxGrepBytesPerObject` 与 `SearchConcurrency`。 + +## 远端 Git + +`remoteGit` 会在 `workspaceBackend` 之上挂载 HTTP/JSON Git 提供程序。它用于没有 +本地 `.git` 目录的非本地工作区。 + +`remoteGit` 必须和 `workspaceBackend` 一起传;单独传入会被拒绝。 + + + + +```rust +use std::env; + +use a3s_code_core::{ + Agent, RemoteGitBackendConfig, S3BackendConfig, SessionOptions, WorkspaceServices, +}; + +#[tokio::main] +async fn main() -> a3s_code_core::Result<()> { + let bucket = env::var("WORKSPACE_S3_BUCKET") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let access_key = env::var("S3_ACCESS_KEY_ID") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let secret_key = env::var("S3_SECRET_ACCESS_KEY") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let base_url = env::var("REMOTE_GIT_BASE_URL") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + let token = env::var("REMOTE_GIT_TOKEN") + .map_err(|error| a3s_code_core::CodeError::Config(error.to_string()))?; + + let storage = WorkspaceServices::s3(S3BackendConfig::new( + bucket, + "sessions/example", + access_key, + secret_key, + )); + let backend = storage.with_remote_git( + RemoteGitBackendConfig::new(base_url, "sessions/example").bearer_token(token), + )?; + + let agent = Agent::new("agent.acl").await?; + let session = agent + .session_builder("s3://workspace-bucket/sessions/example") + .options(SessionOptions::new().with_workspace_backend(backend)) + .build() + .await?; + + let status = session + .tool("git", serde_json::json!({ "command": "status" })) + .await?; + println!("{}", status.output); + + session.close().await; + agent.close().await; + Ok(()) +} +``` + + + + +```ts +import { Agent, S3WorkspaceBackend } from '@a3s-lab/code'; + +const backend = new S3WorkspaceBackend({ + endpoint: process.env.WORKSPACE_S3_ENDPOINT, + region: process.env.WORKSPACE_S3_REGION ?? 'us-east-1', + accessKeyId: process.env.S3_ACCESS_KEY_ID!, + secretAccessKey: process.env.S3_SECRET_ACCESS_KEY!, + bucket: process.env.WORKSPACE_S3_BUCKET!, + prefix: 'sessions/example', + forcePathStyle: true, +}); + +const agent = await Agent.create('agent.acl'); +const session = agent.session('s3://workspace-bucket/sessions/example', { + workspaceBackend: backend, + remoteGit: { + baseUrl: process.env.REMOTE_GIT_BASE_URL!, + repoId: 'sessions/example', + bearerToken: process.env.REMOTE_GIT_TOKEN, + }, +}); +``` + + + + +```python +import os + +from a3s_code import ( + Agent, + RemoteGitBackendConfig, + S3WorkspaceBackend, + SessionOptions, +) + +agent = Agent.create("agent.acl") +opts = SessionOptions() +opts.workspace_backend = S3WorkspaceBackend( + bucket=os.environ["WORKSPACE_S3_BUCKET"], + prefix="sessions/example", + access_key_id=os.environ["S3_ACCESS_KEY_ID"], + secret_access_key=os.environ["S3_SECRET_ACCESS_KEY"], + endpoint=os.environ.get("WORKSPACE_S3_ENDPOINT"), + region=os.environ.get("WORKSPACE_S3_REGION", "us-east-1"), + force_path_style=True, +) +opts.remote_git = RemoteGitBackendConfig( + base_url=os.environ["REMOTE_GIT_BASE_URL"], + repo_id="sessions/example", + bearer_token=os.environ["REMOTE_GIT_TOKEN"], +) +session = agent.session("s3://workspace-bucket/sessions/example", opts) +``` + + + + +```go +package main + +import ( + "context" + "fmt" + "log" + "os" + + code "github.com/A3S-Lab/Code/sdk/go/v8" +) + +func main() { + ctx := context.Background() + agent, err := code.Create(ctx, "agent.acl") + if err != nil { + log.Fatal(err) + } + defer agent.Close(ctx) + + forcePathStyle := true + session, err := agent.Session( + ctx, + "s3://workspace-bucket/sessions/example", + &code.SessionOptions{ + WorkspaceBackend: &code.WorkspaceBackendConfig{ + Kind: "s3", + S3: &code.S3BackendConfig{ + Endpoint: os.Getenv("WORKSPACE_S3_ENDPOINT"), + Region: os.Getenv("WORKSPACE_S3_REGION"), + AccessKeyID: os.Getenv("S3_ACCESS_KEY_ID"), + SecretAccessKey: os.Getenv("S3_SECRET_ACCESS_KEY"), + Bucket: os.Getenv("WORKSPACE_S3_BUCKET"), + Prefix: "sessions/example", + ForcePathStyle: &forcePathStyle, + }, + }, + RemoteGit: &code.RemoteGitBackendConfig{ + BaseURL: os.Getenv("REMOTE_GIT_BASE_URL"), + RepoID: "sessions/example", + BearerToken: os.Getenv("REMOTE_GIT_TOKEN"), + }, + }, + ) + if err != nil { + log.Fatal(err) + } + defer session.Close(ctx) + + status, err := session.Git(ctx, code.GitOptions{Command: "status"}) + if err != nil { + log.Fatal(err) + } + fmt.Println(status.Output) +} +``` + + + + +不要把远端 Git 凭据写入 `agent.acl` 或智能体目录。应由宿主通过环境变量或密钥 +管理系统注入。 + +### 远端 Git 选项 + +| Node.js 选项 | Python 选项 | 必填 | 作用 | +| ------------------ | -------------------- | -------- | ------------------------------------------------------- | +| `baseUrl` | `base_url` | 是 | 远端 Git 服务的基础 URL,不带末尾斜杠。 | +| `repoId` | `repo_id` | 是 | 与远端 Git 服务约定的不透明仓库标识。 | +| `bearerToken` | `bearer_token` | 生产环境 | 远端 Git 服务的持有者令牌;只应在受信任开发环境中省略。 | +| `clientCertPem` | `client_cert_pem` | 否 | mTLS 客户端证书路径;必须与客户端密钥成对设置。 | +| `clientKeyPem` | `client_key_pem` | 否 | mTLS 客户端密钥路径;必须与证书成对设置。 | +| `requestTimeoutMs` | `request_timeout_ms` | 否 | 单次 HTTP 调用超时,单位毫秒;默认 30000。 | +| `maxDiffBytes` | `max_diff_bytes` | 否 | `diff` 响应字节数客户端上限;默认 1 MiB。 | +| `maxLogEntries` | `max_log_entries` | 否 | `log` 条目数客户端上限;默认 200。 | + +Go 在 `RemoteGitBackendConfig` 上提供同一组字段:`BaseURL`、`RepoID`、 +`BearerToken`、`ClientCertPEM`、`ClientKeyPEM`、`RequestTimeoutMS`、 +`MaxDiffBytes` 与 `MaxLogEntries`。 + +## 选择后端 + +- 普通开发机和 CI 检出使用默认本地工作区。 +- 宿主希望总是传入带类型的后端对象时,使用 `LocalWorkspaceBackend`。 +- 工作区状态必须落在对象存储中时,使用 `S3WorkspaceBackend`。 +- 非本地工作区仍需要内置 `git` 工具时,追加 `remoteGit`。 +- 当前后端不能直接运行命令时,通过 MCP 或 A3S Box 提供执行能力。 diff --git a/website/docs/v8.5.1/zh/index.mdx b/website/docs/v8.5.1/zh/index.mdx new file mode 100644 index 00000000..23c7a7a6 --- /dev/null +++ b/website/docs/v8.5.1/zh/index.mdx @@ -0,0 +1,7 @@ +--- +pageType: home +title: A3S Code +description: 用 Rust 构建的受治理编码 Agent 运行时,支持异步工作区检索、模型边界证据、事件流与任务恢复,并提供 Rust、Node.js、Python、Go API。 +sidebar: false +outline: false +--- diff --git a/website/rspress.config.ts b/website/rspress.config.ts index ea7e729e..4f496bbb 100644 --- a/website/rspress.config.ts +++ b/website/rspress.config.ts @@ -42,8 +42,9 @@ export default defineConfig({ ], }, multiVersion: { - default: 'v8.4.0', + default: 'v8.5.1', versions: [ + 'v8.5.1', 'v8.4.0', 'v8.3.0', 'v8.2.0', diff --git a/website/theme/components/TuiWelcomeBanner.tsx b/website/theme/components/TuiWelcomeBanner.tsx index 98e3c8a3..2219e46d 100644 --- a/website/theme/components/TuiWelcomeBanner.tsx +++ b/website/theme/components/TuiWelcomeBanner.tsx @@ -72,7 +72,7 @@ export function TuiWelcomeBanner({

- a3s-code v8.4.0 + a3s-code v8.5.1 · openai/gpt-5 · diff --git a/website/version-snapshots.json b/website/version-snapshots.json index d139b768..1794f178 100644 --- a/website/version-snapshots.json +++ b/website/version-snapshots.json @@ -1,6 +1,13 @@ { - "current": "v8.4.0", + "current": "v8.5.1", "archives": [ + { + "version": "v8.4.0", + "sourceTag": "v8.4.0", + "sourceTree": "5959d7f6637b65eee66490beb77980699b808aa8", + "files": 128, + "sha256": "876343663254b8f2b1c1908b7d530f255683b805eeda2c0194e771262d31b74e" + }, { "version": "v8.3.0", "sourceTag": "v8.3.0",