From 50349e37812c37bd5f43071a0ee64cb2f22874d8 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Tue, 22 Sep 2026 12:02:34 -0700 Subject: [PATCH 01/14] Read agent launches in CI workflows (#823) The workflow grant read triggers, token permissions, reusable-workflow secrets (#693) and step `uses:` references (#771), and nothing about how a coding agent is launched inside a job. Changing a claude-code-action's `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` run step, or checking out the pull request head in a `pull_request_target` job each gave "No static host-grant changes detected", and a move to `issue_comment` with `pull-requests: write` gave a row with no agent context. The workflow grant now lists, as text that is never executed, fetched or evaluated: - `agent_launches[]`: a step whose `uses:` is anthropics/claude-code-action, anthropics/claude-code-base-action or openai/codex-action (any ref, any case) with the documented permission inputs it sets, or a `run:` that is one literal simple command starting with `claude -p/--print` or `codex exec`, with its documented permission flags under their primary spelling. The prompt, `--model` and undocumented flags are not compared. `job_secrets` names the secrets the job references, as context only. - `checkout_refs[]`: each actions/checkout step's `with.ref`, null for the default. Each job's multiset of launches and refs is compared, never the step label, so a rename or reorder is quiet; a difference is one `changed` row on the existing workflow row naming `job/step` and both values. Direction is claimed only by documented rules a job's launches gain, read from literal values: bypassed permission checks (either spelling), bypassed approvals and sandbox, a danger-full-access sandbox, `safety-strategy: unsafe`, or a `*` user gate. Those raise `workflow_agent_widened_` and make the row widened; every other edit, including `--allowedTools "Bash(*)"` (#824's to rate), is `changed`. A workflow row whose workflow runs an agent ends its `why` with the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step; it is a note and moves no direction. A compound `run:`, an expansion or an expression is `unresolved`, publishes none of its text and records a non-blocking coverage issue naming `job/step`, as an unread secret value does (#693): coverage stays complete, adding one is a row that claims no effect, and an edit inside one is not reported. Scripts, composite (#701) and unknown actions, and agents reached through npx/timeout/sudo are listed as unread surfaces. Every value, ref and secret name goes through the #802 label redaction; a rewritten one is null with `redacted` and is neither published nor compared. Contract 40 and host-grants 0.6 shipped in 1.1.0, so host-grants inventory, baseline and drift move to 0.7 (the 0.6 schema files are untouched) and the runtime contract to 41. A 0.4-0.6 baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier 0.20 and capability diff 0.3 do not move. No check id is added or removed and `check` decides as before. Docs: the support page (tables, rules, note, limits, a new unread-surfaces bullet), a STABILITY "Migration Note: Unreleased", CHANGELOG `## Unreleased` above 1.1.0, the agent contract page, the distribution-surfaces `capability_diff` row and its parity comment, the Stop hook note, version tables and pins, and a rebuilt llms-full.txt. The pilot ledger's source-tree column was re-measured on the Route H fixture: identical cells to the 1.1.0 engine apart from contract 41 and inventory 0.7, with byte-identical diff rows. Tests: new tests/test_workflow_agent_launches.py covers the four reproduction cases, what is and is not read, every unsupported shape, direction rules and non-rules, the acceptance's negative controls, redaction with a CLI canary sweep across every published output, the 0.6 baseline migration, schema validation, and the same row on diff, verify, the PR comment, check, the control envelope and the Stop hook. Existing tests move to 0.7/41. The host-config and cold-start replays reproduce their committed outcomes and run-of-record scores, and the sample goldens are unchanged. Closes #823 --- .well-known/agents-shipgate.json | 12 +- AGENTS.md | 6 +- CHANGELOG.md | 3 +- STABILITY.md | 59 +- docs/INDEX.md | 9 +- docs/agent-contract-current.md | 26 +- docs/design-partner-pilot-results.md | 2 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 88 +- docs/host-grants-baseline-schema.v0.7.json | 1744 ++++++++++++++++ docs/host-grants-drift-schema.v0.7.json | 318 +++ docs/host-grants-inventory-schema.v0.7.json | 1802 +++++++++++++++++ docs/integrations.md | 6 +- docs/passed-verdict-contract.md | 2 +- llms-full.txt | 32 +- llms.txt | 6 +- scripts/generate_schemas.py | 24 +- .../core/capability_diff_rows.py | 247 ++- src/agents_shipgate/core/host_grants.py | 701 ++++++- src/agents_shipgate/schemas/contract.py | 37 +- src/agents_shipgate/schemas/host_grants.py | 151 +- tests/test_agent_instructions_apply.py | 6 +- tests/test_agent_instructions_renderers.py | 6 +- tests/test_distribution_surface_parity.py | 7 + tests/test_host_audit.py | 26 +- tests/test_host_input_recovery.py | 2 +- tests/test_instruction_structure_contracts.py | 2 +- tests/test_local_contract.py | 6 +- tests/test_org_governance.py | 2 +- .../test_reusable_workflow_secret_mappings.py | 9 +- tests/test_workflow_agent_launches.py | 838 ++++++++ tests/test_workflow_step_action_references.py | 21 +- 32 files changed, 6085 insertions(+), 117 deletions(-) create mode 100644 docs/host-grants-baseline-schema.v0.7.json create mode 100644 docs/host-grants-drift-schema.v0.7.json create mode 100644 docs/host-grants-inventory-schema.v0.7.json create mode 100644 tests/test_workflow_agent_launches.py diff --git a/.well-known/agents-shipgate.json b/.well-known/agents-shipgate.json index 35f50556d..ddee15b23 100644 --- a/.well-known/agents-shipgate.json +++ b/.well-known/agents-shipgate.json @@ -309,9 +309,9 @@ "attestation_schema_version": "0.5", "registry_schema_version": "0.4", "org_evidence_bundle_schema_version": "shipgate.org_evidence_bundle/v2", - "host_grants_inventory_schema_version": "0.6", - "host_grants_baseline_schema_version": "0.6", - "host_grants_drift_schema_version": "0.6", + "host_grants_inventory_schema_version": "0.7", + "host_grants_baseline_schema_version": "0.7", + "host_grants_drift_schema_version": "0.7", "trigger_catalog_schema_version": "0.4", "capability_standard_version": "0.5", "governance_benchmark_catalog_schema_version": "0.2", @@ -525,9 +525,9 @@ "org_governance": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/org-governance-schema.v0.1.json", "org_evidence_bundle": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/org-evidence-bundle-schema.v2.json", "registry": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/registry-schema.v0.4.json", - "host_grants_inventory": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.6.json", - "host_grants_baseline": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.6.json", - "host_grants_drift": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.6.json", + "host_grants_inventory": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.7.json", + "host_grants_baseline": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.7.json", + "host_grants_drift": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.7.json", "scenario": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/scenario-schema.v0.1.json", "checks_catalog": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/checks.json", "determinism_boundary": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/determinism-boundary.json", diff --git a/AGENTS.md b/AGENTS.md index b8033a33c..60568221c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -832,9 +832,9 @@ For the short, current statement of "which fields to read", see [`docs/agent-con | Verifier schema (current) | [`docs/verifier-schema.v0.21.json`](docs/verifier-schema.v0.21.json) | `0.21` | | Agent handoff schema (current) | [`docs/agent-handoff-schema.v9.json`](docs/agent-handoff-schema.v9.json) | `shipgate.agent_handoff/v9` | | Preflight schema (current) | [`docs/preflight-schema.v0.5.json`](docs/preflight-schema.v0.5.json) | `0.5` | -| Host-grants inventory schema | [`docs/host-grants-inventory-schema.v0.6.json`](docs/host-grants-inventory-schema.v0.6.json) | `0.6` | -| Host-grants baseline schema | [`docs/host-grants-baseline-schema.v0.6.json`](docs/host-grants-baseline-schema.v0.6.json) | `0.6` | -| Host-grants drift schema | [`docs/host-grants-drift-schema.v0.6.json`](docs/host-grants-drift-schema.v0.6.json) | `0.6` | +| Host-grants inventory schema | [`docs/host-grants-inventory-schema.v0.7.json`](docs/host-grants-inventory-schema.v0.7.json) | `0.7` | +| Host-grants baseline schema | [`docs/host-grants-baseline-schema.v0.7.json`](docs/host-grants-baseline-schema.v0.7.json) | `0.7` | +| Host-grants drift schema | [`docs/host-grants-drift-schema.v0.7.json`](docs/host-grants-drift-schema.v0.7.json) | `0.7` | | Capability standard | [`docs/capability-standard.md`](docs/capability-standard.md) | `0.5` | | Capability lock schema | [`docs/capability-lock-schema.v0.8.json`](docs/capability-lock-schema.v0.8.json) | `0.8` | | Capability lock diff schema | [`docs/capability-lock-diff-schema.v0.9.json`](docs/capability-lock-diff-schema.v0.9.json) | `0.9` | diff --git a/CHANGELOG.md b/CHANGELOG.md index d98d51771..5e17ea3ee 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,8 +9,9 @@ - **The problem.** A pull request that added a Cursor plugin's `mcp.json`, removed a `beforeShellExecution` guard from `.cursor/hooks.json`, gave a dotfiles package's `claude/.claude/settings.json` `Bash(*)`, or moved a marketplace plugin's pinned `sha` printed `No static host-grant changes detected`, as a docs-only change does. Re-running a 23-PR public corpus after #812 found 11 of 23 pull requests were such coverage gaps: 0 of the 9 comparable zero-row results named the changed relevant file, and 4 of them named a file the pull request did not touch while omitting the one it did. - **What is named.** `diff`, `verify` and the manifest-free PR comment list, under `What this run established`, each path in the comparison's own changed-file set that a bounded, documented candidate rule recognises and no reader of this entry read: `mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, and an external marketplace plugin source — `plugins/demo/mcp.json (cursor): added, not read by this entry: MCP configuration in a plugin directory; no row, and loading is not established`. An external source names what it now points at, redacted, and is never fetched. The block's first line says the list includes them. Ordinary documentation, an unrelated `*.json` and an unchanged candidate name nothing. - **What it is not.** Never a row, a widening, a `check` violation or a claim that a host loads the file. Nothing is fetched or run, only plugin manifests and marketplaces are read, and at most 32 candidates are examined; the rest, and any whose rule needed a file that was not read or did not parse, are counted as not examined, on a line that names both causes. The rules are listed in `docs/host-boundary-support.md` under *Changed inputs named but not read*. - - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; host-grants stays `0.6`, and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. + - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text; every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index d49c517b0..958af0f0c 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -18,6 +18,23 @@ workspace too. `minimum_control_contract_version` stays `21`. See [the migration note](#unread-changed-inputs-821). +Unreleased, runtime contract v41 reads how a coding agent is launched inside a +workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so +host-grants inventory, baseline and drift schemas move to `0.7`: a workflow +grant adds `agent_launches[]` — a documented agent action's permission inputs, +or the permission flags of a `run:` that is one literal `claude -p` or +`codex exec` command, compared as text and never executed — and +`checkout_refs[]`, each `actions/checkout` step's `with.ref`. Only a documented +rule a job's launches gain widens (`workflow_agent_widened_`); +every other edit is a `changed` row naming `job/step`, and a workflow row that +runs an agent ends with the job facts beside each agent step. A compound +command, an expansion or an expression is `unresolved`, a named non-blocking +limit that leaves coverage complete. A `0.4`–`0.6` baseline holding a workflow +grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one +without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and +`minimum_control_contract_version` `21` are unchanged. See +[the migration note](#workflow-agent-launches-contract-v41-823). + Also unreleased, and moving no version of its own: a Claude Code setting that disables prompts or approves project MCP servers carries one rating on every surface (#827). The `audit --host` grant, the `diff`, `verify` and `check` @@ -27,7 +44,7 @@ move from `SHIP-HOST-BOUNDARY-CONFIG-PARSE-FAILED` to `SHIP-HOST-BOUNDARY-PERMISSION-WILDCARD-ALLOW`, and `defaultMode: dontAsk` from the wildcard check to `SHIP-HOST-BOUNDARY-PERMISSION-ALLOW-EXPANDED` at `medium`. Setting rows name the setting and value, and `enabledMcpjsonServers` -entries become grants. No schema, member or check id moves. See +entries become grants. See [the migration note](#claude-setting-ratings-827). Also unreleased, and moving no version of its own: `check` and `verify` route @@ -58,7 +75,7 @@ now always declares its worktree snapshot, so Git configuration the worktree readers refuse (#813) no longer leaves a preview current. See [the migration note](#preview-control-currency-807). -Runtime contract v40 reads the action reference each workflow step declares +Previous runtime contract v40 reads the action reference each workflow step declares (#771). Host-grants inventory, baseline and drift schemas move to `0.6`, and a workflow grant adds `step_actions[]`: the job, the step (`id`, else `name`, else `steps[N]`), the declared `uses`, and its `form` — `remote`, `docker`, or @@ -300,6 +317,44 @@ is added. --- + + +## Migration Note: Unreleased — workflow agent launches (host-grants `0.7`, contract v41, #823) + +Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` and runtime contract `41` rather than extending them in place. The `0.6` schema files stay published and unchanged. A workflow grant adds two members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: + +```json +{ + "agent_launches": [ + { + "job": "review", "step": "steps[1]", "agent": "anthropics/claude-code-action", + "form": "read", "unresolved_reason": null, + "settings": [ + {"name": "claude_args", "value": "--permission-mode bypassPermissions --allowedTools \"Bash(*)\"", "unresolved_reason": null} + ], + "job_secrets": ["CLAUDE_CODE_OAUTH_TOKEN"] + } + ], + "checkout_refs": [ + {"job": "review", "step": "steps[0]", "ref": "${{ github.event.pull_request.head.sha }}", "unresolved_reason": null} + ] +} +``` + +- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). +- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. The prompt, `--model` and any undocumented flag of a CLI launch are not compared; an agent action's `claude_args` or `codex-args` is compared whole. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. It is read from literal values only, so a value holding `${{ }}` never meets a rule. The grant then earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. +- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting or ref the label redaction rewrites (`redacted`) or that is not a string (`not_a_string`) has `value`/`ref: null`. Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry is still a row, one that claims no effect. Only an edit inside it is not reported. +- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, a path), and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. + +**Compatibility.** +- **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. +- **A `0.4`–`0.6` baseline with no workflow grant** stays comparable for drift. `audit --host --save-baseline` refuses to overwrite any baseline older than `0.7`, with or without a workflow grant, and exits `2` with `unsupported_baseline_schema`; move it aside and re-save. +- **Git-backed `diff`, `check` and manifest-free `verify`** read both refs with the current reader and need no migration. Their rows keep their shape, and verifier `0.20` and capability diff `0.3` do not move. What changes is values: a workflow row can now be `widened` for an agent launch, its `before`/`after` cells list changed launches and checkout refs, and the `why` of every workflow row whose workflow runs an agent gains the note, so a consumer that matches `why` text exactly sees new text. No check id is added or removed, and `check` decides as before. +- **Validators pinned to the `0.6` schemas** reject a `0.7` inventory, baseline or drift payload. +- **`minimum_control_contract_version`** stays `21`. + ## Migration Note: Unreleased — the changed inputs a host comparison does not read (verifier `0.21`, capability diff `0.4`, contract v41, #821) diff --git a/docs/INDEX.md b/docs/INDEX.md index c8a5144c5..b3d941f11 100644 --- a/docs/INDEX.md +++ b/docs/INDEX.md @@ -118,15 +118,18 @@ repository [`README.md`](../README.md) is the landing page that routes to both. - [`org-evidence-bundle-schema.v2.json`](org-evidence-bundle-schema.v2.json) — JSON Schema for `agents-shipgate org bundle`; compact CI/ledger ingestion artifact over verifier/report/attestation/org/host-grant evidence, not a release verdict - [`registry-schema.v0.4.json`](registry-schema.v0.4.json) — JSON Schema for `agents-shipgate registry query --json`, `registry summary --json`, `registry verify --json`, and `registry report --bypass --json` - [`registry-schema.v0.3.json`](registry-schema.v0.3.json) — frozen v0.3 registry reference -- [`host-grants-inventory-schema.v0.6.json`](host-grants-inventory-schema.v0.6.json) — current typed, redacted, scope-aware host inventory; records the in-tree links a read followed and each workflow step's action reference +- [`host-grants-inventory-schema.v0.7.json`](host-grants-inventory-schema.v0.7.json) — current typed, redacted, scope-aware host inventory; records the in-tree links a read followed, each workflow step's action reference, and each agent launch and checkout ref a workflow declares +- [`host-grants-inventory-schema.v0.6.json`](host-grants-inventory-schema.v0.6.json) — frozen v0.6 reference - [`host-grants-inventory-schema.v0.5.json`](host-grants-inventory-schema.v0.5.json) — frozen v0.5 reference - [`host-grants-inventory-schema.v0.4.json`](host-grants-inventory-schema.v0.4.json) — frozen v0.4 reference - [`host-grants-inventory-schema.v0.2.json`](host-grants-inventory-schema.v0.2.json) — frozen prior reference; no inferred structural comparison -- [`host-grants-baseline-schema.v0.6.json`](host-grants-baseline-schema.v0.6.json) — current acknowledged host-grant baseline +- [`host-grants-baseline-schema.v0.7.json`](host-grants-baseline-schema.v0.7.json) — current acknowledged host-grant baseline +- [`host-grants-baseline-schema.v0.6.json`](host-grants-baseline-schema.v0.6.json) — frozen v0.6 reference; compared by drift only when it holds no workflow grant - [`host-grants-baseline-schema.v0.5.json`](host-grants-baseline-schema.v0.5.json) — frozen v0.5 reference; compared by drift only when it holds no workflow grant - [`host-grants-baseline-schema.v0.4.json`](host-grants-baseline-schema.v0.4.json) — frozen v0.4 reference; compared by drift only when it holds no workflow grant - [`host-grants-baseline-schema.v0.2.json`](host-grants-baseline-schema.v0.2.json) — frozen prior reference; no inferred structural comparison -- [`host-grants-drift-schema.v0.6.json`](host-grants-drift-schema.v0.6.json) — current comparable/incomparable host-grant drift result +- [`host-grants-drift-schema.v0.7.json`](host-grants-drift-schema.v0.7.json) — current comparable/incomparable host-grant drift result +- [`host-grants-drift-schema.v0.6.json`](host-grants-drift-schema.v0.6.json) — frozen v0.6 reference - [`host-grants-drift-schema.v0.5.json`](host-grants-drift-schema.v0.5.json) — frozen v0.5 reference - [`host-grants-drift-schema.v0.4.json`](host-grants-drift-schema.v0.4.json) — frozen v0.4 reference - [`host-grants-drift-schema.v0.2.json`](host-grants-drift-schema.v0.2.json) — frozen prior reference; no inferred structural comparison diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index b61277bf8..1bae576af 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -44,6 +44,30 @@ directory, still refuses its comparison. A `0.20` verifier claiming a partial comparison or a `scope` is refused. See [the migration note](../STABILITY.md#partial-host-comparison-808). +Runtime contract v41, unreleased, reads how a coding agent is launched inside a +workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so +host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow +grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. +An agent launch is a step whose `uses:` is a documented agent action +(`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, +`openai/codex-action`) with the permission inputs it declares, or a `run:` +that is one literal `claude -p` / `codex exec` command with its documented +permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`) and +`job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, +`null` for the default. Values are compared as text and never executed. Only a +documented rule a job's launches gain — bypassed permission checks, a bypassed +or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate +opened to `*` — raises `workflow_agent_widened_` and makes the +row `widened`; every other edit is `changed`, and a workflow row that runs an +agent ends its `why` with the job facts beside each agent step. A compound +`run:`, an expansion or an expression is `unresolved` and a named non-blocking +limit. A `0.4`–`0.6` baseline holding a workflow grant is incomparable +(`baseline_workflow_agent_launches_unavailable`); one without a workflow stays +comparable. Verifier `0.20`, capability diff `0.3` and +`minimum_control_contract_version` `21` are unchanged. See +[the migration note](../STABILITY.md#workflow-agent-launches-contract-v41-823). + Previous runtime contract v40 reads the action reference each workflow step declares (#771). Host-grants inventory, baseline and drift schemas move to `0.6`, and a workflow grant adds `step_actions[]`: the job, the step (`id`, else `name`, @@ -735,7 +759,7 @@ Downstream repos generated with - Current attestation schema: `0.5` — [`docs/attestation-schema.v0.5.json`](attestation-schema.v0.5.json) - Current registry schema: `0.4` — [`docs/registry-schema.v0.4.json`](registry-schema.v0.4.json) - Current org evidence bundle schema: `shipgate.org_evidence_bundle/v2` — [`docs/org-evidence-bundle-schema.v2.json`](org-evidence-bundle-schema.v2.json) -- Current host-grants inventory, baseline, and drift schemas: `0.6` — [`inventory`](host-grants-inventory-schema.v0.6.json), [`baseline`](host-grants-baseline-schema.v0.6.json), [`drift`](host-grants-drift-schema.v0.6.json) +- Current host-grants inventory, baseline, and drift schemas: `0.7` — [`inventory`](host-grants-inventory-schema.v0.7.json), [`baseline`](host-grants-baseline-schema.v0.7.json), [`drift`](host-grants-drift-schema.v0.7.json) - Current trigger catalog schema: `0.4` — [`docs/triggers.json`](triggers.json) - Current governance benchmark catalog schema: `0.2` — [`docs/governance-benchmark-catalog-schema.v0.2.json`](governance-benchmark-catalog-schema.v0.2.json) - Current governance benchmark result schema: `0.2` — [`docs/governance-benchmark-result-schema.v0.2.json`](governance-benchmark-result-schema.v0.2.json) diff --git a/docs/design-partner-pilot-results.md b/docs/design-partner-pilot-results.md index 5f77e2ab9..d91ec036e 100644 --- a/docs/design-partner-pilot-results.md +++ b/docs/design-partner-pilot-results.md @@ -119,7 +119,7 @@ expansion signals. It shipped as a qualified release. | | Released `v1.1.0` (`pip install`) | Preview `0.16.0+preview.20260903` (`gh release download`) | Source tree | | --- | --- | --- | --- | | Runtime contract | 40 | 29 | 41 | -| Host-grant inventory schema | 0.6 | 0.2 | 0.6 | +| Host-grant inventory schema | 0.6 | 0.2 | 0.7 | | `check` on the fixture | `block` / `critical`, **4 violations** | `block` / `critical`, **4 violations** | `block` / `critical`, **4 violations** | | Coverage limit visible (`host_coverage`, `excluded_scopes`) | yes | yes | yes | | `init --write --ci` Action pin | not applicable — host audit handoff, no workflow written | `@v0.16.0+preview.20260903.gb61aca7` — **no such tag** (the release tag is `preview-`-prefixed) | not applicable — host audit handoff, no workflow written | diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index 95a975151..b65ae8912 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`), with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch or unreadable value is named only by the host inventory and `audit --host`, as for an unread secret value (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 62d780c59..c094c9360 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -19,7 +19,7 @@ and `audit --host`. | Claude Code | first-class | `.claude/settings.json`, `.claude/settings.local.json`, `.mcp.json`, `CLAUDE.md`, Claude skills | permission modes/rules, sandbox/network, additional paths, MCP restrictions, plugins and their marketplaces (`extraKnownMarketplaces`), hooks | | Cursor | first-class | `.cursor/cli.json`, `.cursor/mcp.json`, `.cursor/rules/**` | Shell/Read/Write rules, MCP declarations, instruction trust roots | | VS Code MCP | first-class | `.vscode/mcp.json` | MCP servers; `sandbox` and per-server `sandboxEnabled`; `${input:…}` references by name, never value; `envFile` recorded as a limit; other top-level keys partial | -| Shared/GitHub | first-class | `AGENTS.md`, Shipgate policies/state, skills, `.github/workflows/*` | instruction/gate weakening, workflow permissions and triggers, remote step action references, named secret sources passed to reusable workflows | +| Shared/GitHub | first-class | `AGENTS.md`, Shipgate policies/state, skills, `.github/workflows/*` | instruction/gate weakening, workflow permissions and triggers, remote step action references, named secret sources passed to reusable workflows, agent launches (documented agent action inputs, literal `claude -p` / `codex exec` run steps) and `actions/checkout` refs | A registered adapter reports `complete`, `not_applicable`, `partial`, or `experimental` coverage. A relevant malformed, unreadable, binary, oversized, @@ -64,6 +64,13 @@ the changed inputs the candidate rules at the end of this section name (#821): them while that subagent runs; no adapter reads the file. A skill's `hooks` frontmatter is type-checked with the skill's instructions, never read as a hook grant, so its events get no hook row (#714). +- **An agent launched any way the workflow reader below does not recognise** + (#823): an action outside its table, even one that takes `claude_args`; a + composite action (#701); a script the step runs (`run: ./scripts/review.sh`); + an agent CLI reached through another command (`npx @anthropic-ai/claude-code`, + `timeout 600 claude`, `sudo`, a path such as `./node_modules/.bin/claude`); + and a step's `env:`, `shell:` and `if:`. Editing one gives no row and names + no limit. Review changes to those files and fields as you would a change to the workflow, hook or server entry that holds them. @@ -117,6 +124,85 @@ reusable `uses:` target, containing credential-shaped text is published redacted and refuses the same way a step reference does, so two values that redact alike never compare as unchanged. +How a coding agent is launched inside a job is read (#823). Every value is +compared as declared text; no action is fetched, no command is run and no +expression is evaluated. Three things are listed on the workflow grant, each +naming its `job/step` (the step's `id`, else its `name`, else `steps[N]`): + +- **A documented agent action** — a step whose `uses:` is one of these + `owner/repo` references, at any ref and in any letter case — with the inputs + it declares from this table. Other inputs, such as `prompt` or an API key, + are not listed. + + | Action | Inputs compared as text | Documented widening | + |---|---|---| + | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions`; `allowed_bots` or `allowed_non_write_users` gains a `*` entry | + | `anthropics/claude-code-base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` as above | + | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access`; `allow-users` gains a `*` entry | + +- **A literal agent CLI command in `run:`** — only when the whole `run:` is one + simple command, after any literal `NAME=value` assignments (which are skipped + and never published), that starts with `claude` and passes `-p`/`--print`, or + starts with `codex exec` (`codex e`). Its documented permission flags are + listed under their primary spelling, and every other word — the prompt, + `--model`, an undocumented flag — is not compared. For `claude`: + `--permission-mode`, `--dangerously-skip-permissions`, + `--allow-dangerously-skip-permissions`, `--allowedTools`/`--allowed-tools`, + `--disallowedTools`/`--disallowed-tools`, `--add-dir`, `--mcp-config`, + `--settings` and `--permission-prompt-tool`; gaining + `--dangerously-skip-permissions` or `--permission-mode bypassPermissions` + (one rule, so moving between the two spellings is not a widening) widens. + For `codex exec`: `--sandbox`/`-s`, + `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, + `--approve-for-me`/`--not-so-yolo`, `--dangerously-bypass-hook-trust`, + `--add-dir`, `--config`/`-c` and `--profile`/`-p`; gaining + `--dangerously-bypass-approvals-and-sandbox` or `--sandbox danger-full-access` + widens. A flag the CLI reads as variadic (`--allowedTools`, `--add-dir`, …) + takes every following word up to the next word starting with `-`, as the CLI + reads it, so a prompt written after it is compared as one of its values; the + row shows it. A `run:` holding more than one command (a newline, `&&`, `;`, + `|`, a redirection or a here-doc) or quoting that does not balance, a shell + expansion (`$VAR`, `$(…)`, a backtick) or a `${{ }}` expression is listed as + `unresolved` with that reason, once for each agent CLI it starts at the head + of a command (of a line, when its quoting does not balance), and none of its + text is published. A command that launches no headless agent — `claude mcp add`, + `codex login`, an `echo` that mentions either — is not listed. +- **Each `actions/checkout` step's `with.ref`**, or the default when it + declares none or an empty one. Adding `ref: + ${{ github.event.pull_request.head.sha }}` is a `changed` row naming the + step on both sides. + +Each job's multiset of launches and of checkout refs is compared, so renaming +or reordering steps is quiet. An added, removed or changed launch or ref is a +`changed` row on the workflow naming `job/step` and the value on each side. +Direction is claimed only by the documented rules above, read from literal +values: when a job's launches gain one, the workflow earns +`workflow_agent_widened_`, the row is `widened` and its `why` +names the rule and step. Any other edit — `--allowedTools "Read"` to +`--allowedTools "Bash(*)"`, `acceptEdits`, a new plugin, a value holding +`${{ }}`, a head-ref checkout — is `changed`; rating a tool rule's reach is a +job for #824. `access` and `risk` still describe the token and triggers alone. + +A workflow row whose workflow runs an agent ends its `why` with the job facts +beside each agent step, whatever else the row is about: an untrusted-input +trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the +job's write scopes, the secrets the job references (`${{ secrets.NAME }}` in the +job, or in the workflow's `env`), and a checkout in the job of pull request +code (`github.event.pull_request.head.sha`, `.head.ref` or `.merge_commit_sha`, +`github.head_ref`, `github.event.workflow_run.head_sha` or `.head_branch`, or +`refs/pull//head` and `/merge`). It is a note, not a verdict: it moves no +direction, and `if:` conditions and the default checkout of a `pull_request` +event are not read into it. A removed workflow gets no note. + +An unresolved launch, a setting whose value the label redaction rewrites or +that is not a string, and a checkout ref of either kind publish nothing of the +value and record a **non-blocking** `unsupported` coverage issue naming the +`job/step`, printed under `audit --host` → Coverage issues. GitHub coverage +stays complete, so `check`, baselines and every other row are unaffected, and +adding, removing or re-forming such an entry is still a row that claims no +effect; only an edit inside it is not reported. `diff`, `verify` and `check` +carry no limit for it, as for an unread secret value (#693). + A workflow's labels are published redacted (#802). A job id, a step's `id` or `name`, a trigger and a permission scope name go through the same redaction as step text, and any userinfo after `scheme://` inside one is replaced, so a diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json new file mode 100644 index 000000000..21e8c7208 --- /dev/null +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -0,0 +1,1744 @@ +{ + "$defs": { + "HostAdditionalPathGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "additional_path", + "default": "additional_path", + "title": "Kind", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "path" + ], + "title": "HostAdditionalPathGrantV2", + "type": "object" + }, + "HostArtifactV4": { + "additionalProperties": false, + "properties": { + "artifact_id": { + "title": "Artifact Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "instruction_structure": { + "anyOf": [ + { + "$ref": "#/$defs/InstructionStructureEvidence" + }, + { + "type": "null" + } + ], + "default": null + }, + "kind": { + "enum": [ + "config", + "mcp", + "hooks", + "workflow", + "instructions", + "requirements" + ], + "title": "Kind", + "type": "string" + }, + "parse_status": { + "enum": [ + "parsed", + "failed", + "unsupported" + ], + "title": "Parse Status", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "redacted_sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Redacted Sha256" + }, + "resolved_through": { + "items": { + "type": "string" + }, + "title": "Resolved Through", + "type": "array" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + } + }, + "required": [ + "artifact_id", + "host", + "scope", + "path", + "kind", + "parse_status" + ], + "title": "HostArtifactV4", + "type": "object" + }, + "HostCoverageV2": { + "additionalProperties": false, + "properties": { + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "issue_ids": { + "items": { + "type": "string" + }, + "title": "Issue Ids", + "type": "array" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "sources_expected": { + "items": { + "type": "string" + }, + "title": "Sources Expected", + "type": "array" + }, + "sources_observed": { + "items": { + "type": "string" + }, + "title": "Sources Observed", + "type": "array" + }, + "status": { + "enum": [ + "complete", + "partial", + "experimental" + ], + "title": "Status", + "type": "string" + } + }, + "required": [ + "host", + "scope", + "status" + ], + "title": "HostCoverageV2", + "type": "object" + }, + "HostGrantsBaselineV7": { + "additionalProperties": false, + "properties": { + "host_grants_schema_version": { + "const": "0.7", + "default": "0.7", + "title": "Host Grants Schema Version", + "type": "string" + }, + "inventory": { + "$ref": "#/$defs/HostGrantsNormalizedSnapshotV7" + }, + "inventory_sha256": { + "title": "Inventory Sha256", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + } + }, + "required": [ + "scope", + "inventory_sha256", + "inventory" + ], + "title": "HostGrantsBaselineV7", + "type": "object" + }, + "HostGrantsNormalizedSnapshotV7": { + "additionalProperties": false, + "properties": { + "artifacts": { + "items": { + "$ref": "#/$defs/HostArtifactV4" + }, + "title": "Artifacts", + "type": "array" + }, + "grants": { + "items": { + "discriminator": { + "mapping": { + "additional_path": "#/$defs/HostAdditionalPathGrantV2", + "hook": "#/$defs/HostHookGrantV2", + "instruction_trust_root": "#/$defs/HostInstructionGrantV2", + "mcp_server": "#/$defs/HostMcpServerGrantV2", + "permission_mode": "#/$defs/HostPermissionModeGrantV2", + "permission_rule": "#/$defs/HostPermissionRuleGrantV2", + "plugin_or_app": "#/$defs/HostPluginGrantV2", + "profile": "#/$defs/HostProfileGrantV2", + "requirement": "#/$defs/HostRequirementGrantV2", + "sandbox": "#/$defs/HostSandboxGrantV2", + "workflow": "#/$defs/HostWorkflowGrantV7" + }, + "propertyName": "kind" + }, + "oneOf": [ + { + "$ref": "#/$defs/HostMcpServerGrantV2" + }, + { + "$ref": "#/$defs/HostPermissionRuleGrantV2" + }, + { + "$ref": "#/$defs/HostPermissionModeGrantV2" + }, + { + "$ref": "#/$defs/HostHookGrantV2" + }, + { + "$ref": "#/$defs/HostSandboxGrantV2" + }, + { + "$ref": "#/$defs/HostAdditionalPathGrantV2" + }, + { + "$ref": "#/$defs/HostPluginGrantV2" + }, + { + "$ref": "#/$defs/HostProfileGrantV2" + }, + { + "$ref": "#/$defs/HostRequirementGrantV2" + }, + { + "$ref": "#/$defs/HostWorkflowGrantV7" + }, + { + "$ref": "#/$defs/HostInstructionGrantV2" + } + ] + }, + "title": "Grants", + "type": "array" + }, + "host_coverage": { + "items": { + "$ref": "#/$defs/HostCoverageV2" + }, + "title": "Host Coverage", + "type": "array" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + } + }, + "required": [ + "scope" + ], + "title": "HostGrantsNormalizedSnapshotV7", + "type": "object" + }, + "HostHookGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "event": { + "title": "Event", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "hook", + "default": "hook", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "event" + ], + "title": "HostHookGrantV2", + "type": "object" + }, + "HostInstructionGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "instruction_trust_root", + "default": "instruction_trust_root", + "title": "Kind", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "path" + ], + "title": "HostInstructionGrantV2", + "type": "object" + }, + "HostMcpServerGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "endpoint": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Endpoint" + }, + "env_keys": { + "items": { + "type": "string" + }, + "title": "Env Keys", + "type": "array" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "header_keys": { + "items": { + "type": "string" + }, + "title": "Header Keys", + "type": "array" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "mcp_server", + "default": "mcp_server", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "server": { + "title": "Server", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "transport": { + "title": "Transport", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "server", + "transport" + ], + "title": "HostMcpServerGrantV2", + "type": "object" + }, + "HostPermissionModeGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "permission_mode", + "default": "permission_mode", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "setting", + "value" + ], + "title": "HostPermissionModeGrantV2", + "type": "object" + }, + "HostPermissionRuleGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "disposition": { + "enum": [ + "allow", + "ask", + "deny" + ], + "title": "Disposition", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "permission_rule", + "default": "permission_rule", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "rule": { + "title": "Rule", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "wildcard": { + "default": false, + "title": "Wildcard", + "type": "boolean" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "disposition", + "rule" + ], + "title": "HostPermissionRuleGrantV2", + "type": "object" + }, + "HostPluginGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "enabled": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Enabled" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "plugin_or_app", + "default": "plugin_or_app", + "title": "Kind", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "name" + ], + "title": "HostPluginGrantV2", + "type": "object" + }, + "HostProfileGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "profile", + "default": "profile", + "title": "Kind", + "type": "string" + }, + "profile": { + "title": "Profile", + "type": "string" + }, + "resolved": { + "title": "Resolved", + "type": "boolean" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "profile", + "resolved" + ], + "title": "HostProfileGrantV2", + "type": "object" + }, + "HostRequirementGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "requirement", + "default": "requirement", + "title": "Kind", + "type": "string" + }, + "requirement": { + "title": "Requirement", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "requirement", + "value" + ], + "title": "HostRequirementGrantV2", + "type": "object" + }, + "HostReusableWorkflowCallV6": { + "additionalProperties": false, + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "secret_mappings": { + "items": { + "$ref": "#/$defs/HostReusableWorkflowSecretV6" + }, + "title": "Secret Mappings", + "type": "array" + }, + "secrets_inherit": { + "title": "Secrets Inherit", + "type": "boolean" + }, + "uses": { + "title": "Uses", + "type": "string" + }, + "uses_redacted": { + "default": false, + "title": "Uses Redacted", + "type": "boolean" + } + }, + "required": [ + "job", + "uses", + "secrets_inherit" + ], + "title": "HostReusableWorkflowCallV6", + "type": "object" + }, + "HostReusableWorkflowSecretV6": { + "additionalProperties": false, + "description": "One named secret a job passes to the reusable workflow it calls (#693).\n\n``destination`` is the callee's secret input name as the caller writes it.\n``source`` is ``NAME`` from a whole-value ``${{ secrets.NAME }}``, and\n``form`` is then ``secret``. The name is a reference, never a value: it\ndoes not establish the secret's privilege, whether the caller has it, or\nwhat the called workflow does with it. Anything else is ``unresolved``,\nand none of its value is published or digested: a literal value, any\nother expression, a non-string, or a ``secrets`` that is neither\n``inherit`` nor a mapping (``destination`` is then ``null``). A name the\ncredential redactors rewrite is ``redacted`` and records a blocking\ncoverage issue, because two values that redact alike must never compare as\nunchanged. Every other unresolved mapping records a non-blocking one naming\nits ``job/destination``: only that value is uncompared, so the rest of the\nfile still compares.", + "properties": { + "destination": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Destination" + }, + "form": { + "enum": [ + "secret", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "source": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Source" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "literal_value", + "expression", + "not_a_string", + "redacted", + "secrets_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "destination", + "source", + "form" + ], + "title": "HostReusableWorkflowSecretV6", + "type": "object" + }, + "HostSandboxGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "sandbox", + "default": "sandbox", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "setting", + "value" + ], + "title": "HostSandboxGrantV2", + "type": "object" + }, + "HostWorkflowAgentLaunchV7": { + "additionalProperties": false, + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with:\n``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read``\nlists the documented permission inputs or flags the step declares in\n``settings``. ``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "properties": { + "agent": { + "enum": [ + "anthropics/claude-code-action", + "anthropics/claude-code-base-action", + "openai/codex-action", + "claude", + "codex" + ], + "title": "Agent", + "type": "string" + }, + "form": { + "enum": [ + "read", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "job_secrets": { + "items": { + "type": "string" + }, + "title": "Job Secrets", + "type": "array" + }, + "settings": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentSettingV7" + }, + "title": "Settings", + "type": "array" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "compound_command", + "shell_expansion", + "expression", + "inputs_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "job", + "step", + "agent", + "form" + ], + "title": "HostWorkflowAgentLaunchV7", + "type": "object" + }, + "HostWorkflowAgentSettingV7": { + "additionalProperties": false, + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, published through the workflow\nlabel redaction (#802); a flag that takes no value has ``null``. A value\nthe redaction rewrites, or one that is not a string, is ``null`` with\n``unresolved_reason``, and records a non-blocking coverage issue naming\nits ``job/step``: it is neither published nor compared.", + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "not_a_string", + "redacted" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + }, + "value": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Value" + } + }, + "required": [ + "name", + "value" + ], + "title": "HostWorkflowAgentSettingV7", + "type": "object" + }, + "HostWorkflowCheckoutRefV7": { + "additionalProperties": false, + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites, a value that is not a string, or ``with:`` that is not a mapping\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "ref": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Ref" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "not_a_string", + "redacted", + "inputs_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "job", + "step", + "ref" + ], + "title": "HostWorkflowCheckoutRefV7", + "type": "object" + }, + "HostWorkflowGrantV7": { + "additionalProperties": false, + "description": "A v0.6 workflow grant plus the agent launches and checkout refs its steps declare.\n\nBoth lists are present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``access`` and ``risk`` still describe the workflow's token and\ntriggers alone.", + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "agent_launches": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentLaunchV7" + }, + "title": "Agent Launches", + "type": "array" + }, + "checkout_refs": { + "items": { + "$ref": "#/$defs/HostWorkflowCheckoutRefV7" + }, + "title": "Checkout Refs", + "type": "array" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "effective_write_scopes": { + "items": { + "type": "string" + }, + "title": "Effective Write Scopes", + "type": "array" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "workflow", + "default": "workflow", + "title": "Kind", + "type": "string" + }, + "permission_contexts": { + "items": { + "$ref": "#/$defs/HostWorkflowPermissionsV4" + }, + "title": "Permission Contexts", + "type": "array" + }, + "pull_request_target": { + "default": false, + "title": "Pull Request Target", + "type": "boolean" + }, + "reusable_calls": { + "items": { + "$ref": "#/$defs/HostReusableWorkflowCallV6" + }, + "title": "Reusable Calls", + "type": "array" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "step_actions": { + "items": { + "$ref": "#/$defs/HostWorkflowStepActionV6" + }, + "title": "Step Actions", + "type": "array" + }, + "triggers": { + "items": { + "type": "string" + }, + "title": "Triggers", + "type": "array" + }, + "write_all": { + "default": false, + "title": "Write All", + "type": "boolean" + }, + "write_scopes": { + "items": { + "type": "string" + }, + "title": "Write Scopes", + "type": "array" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "permission_contexts", + "effective_write_scopes", + "reusable_calls" + ], + "title": "HostWorkflowGrantV7", + "type": "object" + }, + "HostWorkflowPermissionsV4": { + "additionalProperties": false, + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "permissions": { + "additionalProperties": { + "enum": [ + "read", + "write" + ], + "type": "string" + }, + "title": "Permissions", + "type": "object" + }, + "state": { + "enum": [ + "explicit", + "repository_default", + "unresolved" + ], + "title": "State", + "type": "string" + } + }, + "required": [ + "job", + "state", + "permissions" + ], + "title": "HostWorkflowPermissionsV4", + "type": "object" + }, + "HostWorkflowStepActionV6": { + "additionalProperties": false, + "description": "One step's declared action reference, read as text and never fetched.\n\n``form`` is ``remote`` for ``owner/repo[/path]@ref``, ``docker`` for\n``docker://\u2026``, and ``unresolved`` for a value Shipgate does not resolve\nto an action identity; ``unresolved_reason`` then says which. A job whose\n``steps`` is not a list, or a step that is not a mapping, is listed as\nunresolved too, with no ``uses``, so an absent list still means the steps\nwere read and declare nothing. A local\n``./\u2026`` reference is not listed: composite actions remain unread (#701).\n``step`` is the step's ``id``, else its ``name``, else ``steps[N]`` \u2014 the\nevidence a reviewer uses to find it, not part of the comparison. ``job``\nand ``step`` are published labels: credential-shaped text in either, and\nthe userinfo of any ``scheme://\u2026@`` inside it, is redacted (#802).", + "properties": { + "form": { + "enum": [ + "remote", + "docker", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "expression", + "unsupported_reference", + "not_a_string", + "redacted", + "steps_not_a_list", + "step_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + }, + "uses": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Uses" + } + }, + "required": [ + "job", + "step", + "uses", + "form" + ], + "title": "HostWorkflowStepActionV6", + "type": "object" + }, + "InstructionStructureEvidence": { + "additionalProperties": false, + "properties": { + "profile": { + "title": "Profile", + "type": "string" + }, + "reason": { + "title": "Reason", + "type": "string" + }, + "sha256": { + "anyOf": [ + { + "pattern": "^sha256:[0-9a-f]{64}$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Sha256" + }, + "status": { + "enum": [ + "guidance", + "structured", + "unresolved" + ], + "title": "Status", + "type": "string" + } + }, + "required": [ + "profile", + "status", + "reason" + ], + "title": "InstructionStructureEvidence", + "type": "object" + } + }, + "$id": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.7.json", + "$ref": "#/$defs/HostGrantsBaselineV7", + "$schema": "https://json-schema.org/draft/2020-12/schema", + "description": "JSON Schema for a human-acknowledged, scope-bound host-grants baseline.", + "title": "Agents Shipgate Host Grants Baseline v0.7" +} diff --git a/docs/host-grants-drift-schema.v0.7.json b/docs/host-grants-drift-schema.v0.7.json new file mode 100644 index 000000000..34a92a642 --- /dev/null +++ b/docs/host-grants-drift-schema.v0.7.json @@ -0,0 +1,318 @@ +{ + "$defs": { + "HostArtifactChangeV2": { + "additionalProperties": false, + "properties": { + "artifact_id": { + "title": "Artifact Id", + "type": "string" + }, + "baseline": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Baseline" + }, + "current": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Current" + } + }, + "required": [ + "artifact_id" + ], + "title": "HostArtifactChangeV2", + "type": "object" + }, + "HostCoverageChangeV2": { + "additionalProperties": false, + "properties": { + "baseline": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Baseline" + }, + "current": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Current" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + } + }, + "required": [ + "host" + ], + "title": "HostCoverageChangeV2", + "type": "object" + }, + "HostGrantChangeV2": { + "additionalProperties": false, + "properties": { + "baseline": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Baseline" + }, + "current": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Current" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + } + }, + "required": [ + "grant_id" + ], + "title": "HostGrantChangeV2", + "type": "object" + }, + "HostGrantsDriftV7": { + "additionalProperties": false, + "properties": { + "artifact_changes": { + "items": { + "$ref": "#/$defs/HostArtifactChangeV2" + }, + "title": "Artifact Changes", + "type": "array" + }, + "baseline_file": { + "title": "Baseline File", + "type": "string" + }, + "baseline_sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Baseline Sha256" + }, + "changes": { + "items": { + "$ref": "#/$defs/HostGrantChangeV2" + }, + "title": "Changes", + "type": "array" + }, + "comparison_status": { + "enum": [ + "comparable", + "incomparable" + ], + "title": "Comparison Status", + "type": "string" + }, + "coverage_changes": { + "items": { + "$ref": "#/$defs/HostCoverageChangeV2" + }, + "title": "Coverage Changes", + "type": "array" + }, + "current_sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Current Sha256" + }, + "expansion_signals": { + "items": { + "type": "string" + }, + "title": "Expansion Signals", + "type": "array" + }, + "has_drift": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Has Drift" + }, + "host_grants_schema_version": { + "const": "0.7", + "default": "0.7", + "title": "Host Grants Schema Version", + "type": "string" + }, + "incomparable_reasons": { + "items": { + "type": "string" + }, + "title": "Incomparable Reasons", + "type": "array" + }, + "issues": { + "items": { + "$ref": "#/$defs/HostInventoryIssueV2" + }, + "title": "Issues", + "type": "array" + }, + "next_action": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Next Action" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + } + }, + "required": [ + "baseline_file", + "scope", + "comparison_status", + "has_drift" + ], + "title": "HostGrantsDriftV7", + "type": "object" + }, + "HostInventoryIssueV2": { + "additionalProperties": false, + "properties": { + "blocking": { + "title": "Blocking", + "type": "boolean" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "issue_id": { + "title": "Issue Id", + "type": "string" + }, + "kind": { + "enum": [ + "parse_failed", + "unreadable", + "unsupported", + "unresolved_precedence", + "dynamic_source_excluded", + "remote_source_excluded" + ], + "title": "Kind", + "type": "string" + }, + "message": { + "title": "Message", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "issue_id", + "kind", + "host", + "source", + "message", + "blocking" + ], + "title": "HostInventoryIssueV2", + "type": "object" + } + }, + "$id": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.7.json", + "$ref": "#/$defs/HostGrantsDriftV7", + "$schema": "https://json-schema.org/draft/2020-12/schema", + "description": "JSON Schema for scope-aware host-grant drift and incomparability.", + "title": "Agents Shipgate Host Grants Drift v0.7" +} diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json new file mode 100644 index 000000000..2f9b6c6f2 --- /dev/null +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -0,0 +1,1802 @@ +{ + "$defs": { + "HostAdditionalPathGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "additional_path", + "default": "additional_path", + "title": "Kind", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "path" + ], + "title": "HostAdditionalPathGrantV2", + "type": "object" + }, + "HostArtifactV4": { + "additionalProperties": false, + "properties": { + "artifact_id": { + "title": "Artifact Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "instruction_structure": { + "anyOf": [ + { + "$ref": "#/$defs/InstructionStructureEvidence" + }, + { + "type": "null" + } + ], + "default": null + }, + "kind": { + "enum": [ + "config", + "mcp", + "hooks", + "workflow", + "instructions", + "requirements" + ], + "title": "Kind", + "type": "string" + }, + "parse_status": { + "enum": [ + "parsed", + "failed", + "unsupported" + ], + "title": "Parse Status", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "redacted_sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Redacted Sha256" + }, + "resolved_through": { + "items": { + "type": "string" + }, + "title": "Resolved Through", + "type": "array" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + } + }, + "required": [ + "artifact_id", + "host", + "scope", + "path", + "kind", + "parse_status" + ], + "title": "HostArtifactV4", + "type": "object" + }, + "HostCoverageV2": { + "additionalProperties": false, + "properties": { + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "issue_ids": { + "items": { + "type": "string" + }, + "title": "Issue Ids", + "type": "array" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "sources_expected": { + "items": { + "type": "string" + }, + "title": "Sources Expected", + "type": "array" + }, + "sources_observed": { + "items": { + "type": "string" + }, + "title": "Sources Observed", + "type": "array" + }, + "status": { + "enum": [ + "complete", + "partial", + "experimental" + ], + "title": "Status", + "type": "string" + } + }, + "required": [ + "host", + "scope", + "status" + ], + "title": "HostCoverageV2", + "type": "object" + }, + "HostGrantsInventoryV7": { + "additionalProperties": false, + "properties": { + "artifacts": { + "items": { + "$ref": "#/$defs/HostArtifactV4" + }, + "title": "Artifacts", + "type": "array" + }, + "excluded_scopes": { + "items": { + "type": "string" + }, + "title": "Excluded Scopes", + "type": "array" + }, + "grants": { + "items": { + "discriminator": { + "mapping": { + "additional_path": "#/$defs/HostAdditionalPathGrantV2", + "hook": "#/$defs/HostHookGrantV2", + "instruction_trust_root": "#/$defs/HostInstructionGrantV2", + "mcp_server": "#/$defs/HostMcpServerGrantV2", + "permission_mode": "#/$defs/HostPermissionModeGrantV2", + "permission_rule": "#/$defs/HostPermissionRuleGrantV2", + "plugin_or_app": "#/$defs/HostPluginGrantV2", + "profile": "#/$defs/HostProfileGrantV2", + "requirement": "#/$defs/HostRequirementGrantV2", + "sandbox": "#/$defs/HostSandboxGrantV2", + "workflow": "#/$defs/HostWorkflowGrantV7" + }, + "propertyName": "kind" + }, + "oneOf": [ + { + "$ref": "#/$defs/HostMcpServerGrantV2" + }, + { + "$ref": "#/$defs/HostPermissionRuleGrantV2" + }, + { + "$ref": "#/$defs/HostPermissionModeGrantV2" + }, + { + "$ref": "#/$defs/HostHookGrantV2" + }, + { + "$ref": "#/$defs/HostSandboxGrantV2" + }, + { + "$ref": "#/$defs/HostAdditionalPathGrantV2" + }, + { + "$ref": "#/$defs/HostPluginGrantV2" + }, + { + "$ref": "#/$defs/HostProfileGrantV2" + }, + { + "$ref": "#/$defs/HostRequirementGrantV2" + }, + { + "$ref": "#/$defs/HostWorkflowGrantV7" + }, + { + "$ref": "#/$defs/HostInstructionGrantV2" + } + ] + }, + "title": "Grants", + "type": "array" + }, + "host_coverage": { + "items": { + "$ref": "#/$defs/HostCoverageV2" + }, + "title": "Host Coverage", + "type": "array" + }, + "host_grants_inventory_schema_version": { + "const": "0.7", + "default": "0.7", + "title": "Host Grants Inventory Schema Version", + "type": "string" + }, + "issues": { + "items": { + "$ref": "#/$defs/HostInventoryIssueV2" + }, + "title": "Issues", + "type": "array" + }, + "runtime_session_verified": { + "const": false, + "default": false, + "title": "Runtime Session Verified", + "type": "boolean" + }, + "scope": { + "default": "repository", + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "static_analysis_only": { + "const": true, + "default": true, + "title": "Static Analysis Only", + "type": "boolean" + }, + "workspace": { + "title": "Workspace", + "type": "string" + } + }, + "required": [ + "workspace" + ], + "title": "HostGrantsInventoryV7", + "type": "object" + }, + "HostHookGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "event": { + "title": "Event", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "hook", + "default": "hook", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "event" + ], + "title": "HostHookGrantV2", + "type": "object" + }, + "HostInstructionGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "instruction_trust_root", + "default": "instruction_trust_root", + "title": "Kind", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "path" + ], + "title": "HostInstructionGrantV2", + "type": "object" + }, + "HostInventoryIssueV2": { + "additionalProperties": false, + "properties": { + "blocking": { + "title": "Blocking", + "type": "boolean" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "issue_id": { + "title": "Issue Id", + "type": "string" + }, + "kind": { + "enum": [ + "parse_failed", + "unreadable", + "unsupported", + "unresolved_precedence", + "dynamic_source_excluded", + "remote_source_excluded" + ], + "title": "Kind", + "type": "string" + }, + "message": { + "title": "Message", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "issue_id", + "kind", + "host", + "source", + "message", + "blocking" + ], + "title": "HostInventoryIssueV2", + "type": "object" + }, + "HostMcpServerGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "endpoint": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Endpoint" + }, + "env_keys": { + "items": { + "type": "string" + }, + "title": "Env Keys", + "type": "array" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "header_keys": { + "items": { + "type": "string" + }, + "title": "Header Keys", + "type": "array" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "mcp_server", + "default": "mcp_server", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "server": { + "title": "Server", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "transport": { + "title": "Transport", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "server", + "transport" + ], + "title": "HostMcpServerGrantV2", + "type": "object" + }, + "HostPermissionModeGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "permission_mode", + "default": "permission_mode", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "setting", + "value" + ], + "title": "HostPermissionModeGrantV2", + "type": "object" + }, + "HostPermissionRuleGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "disposition": { + "enum": [ + "allow", + "ask", + "deny" + ], + "title": "Disposition", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "permission_rule", + "default": "permission_rule", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "rule": { + "title": "Rule", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "wildcard": { + "default": false, + "title": "Wildcard", + "type": "boolean" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "disposition", + "rule" + ], + "title": "HostPermissionRuleGrantV2", + "type": "object" + }, + "HostPluginGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "enabled": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Enabled" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "plugin_or_app", + "default": "plugin_or_app", + "title": "Kind", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "name" + ], + "title": "HostPluginGrantV2", + "type": "object" + }, + "HostProfileGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "profile", + "default": "profile", + "title": "Kind", + "type": "string" + }, + "profile": { + "title": "Profile", + "type": "string" + }, + "resolved": { + "title": "Resolved", + "type": "boolean" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "profile", + "resolved" + ], + "title": "HostProfileGrantV2", + "type": "object" + }, + "HostRequirementGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "requirement", + "default": "requirement", + "title": "Kind", + "type": "string" + }, + "requirement": { + "title": "Requirement", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "requirement", + "value" + ], + "title": "HostRequirementGrantV2", + "type": "object" + }, + "HostReusableWorkflowCallV6": { + "additionalProperties": false, + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "secret_mappings": { + "items": { + "$ref": "#/$defs/HostReusableWorkflowSecretV6" + }, + "title": "Secret Mappings", + "type": "array" + }, + "secrets_inherit": { + "title": "Secrets Inherit", + "type": "boolean" + }, + "uses": { + "title": "Uses", + "type": "string" + }, + "uses_redacted": { + "default": false, + "title": "Uses Redacted", + "type": "boolean" + } + }, + "required": [ + "job", + "uses", + "secrets_inherit" + ], + "title": "HostReusableWorkflowCallV6", + "type": "object" + }, + "HostReusableWorkflowSecretV6": { + "additionalProperties": false, + "description": "One named secret a job passes to the reusable workflow it calls (#693).\n\n``destination`` is the callee's secret input name as the caller writes it.\n``source`` is ``NAME`` from a whole-value ``${{ secrets.NAME }}``, and\n``form`` is then ``secret``. The name is a reference, never a value: it\ndoes not establish the secret's privilege, whether the caller has it, or\nwhat the called workflow does with it. Anything else is ``unresolved``,\nand none of its value is published or digested: a literal value, any\nother expression, a non-string, or a ``secrets`` that is neither\n``inherit`` nor a mapping (``destination`` is then ``null``). A name the\ncredential redactors rewrite is ``redacted`` and records a blocking\ncoverage issue, because two values that redact alike must never compare as\nunchanged. Every other unresolved mapping records a non-blocking one naming\nits ``job/destination``: only that value is uncompared, so the rest of the\nfile still compares.", + "properties": { + "destination": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Destination" + }, + "form": { + "enum": [ + "secret", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "source": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Source" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "literal_value", + "expression", + "not_a_string", + "redacted", + "secrets_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "destination", + "source", + "form" + ], + "title": "HostReusableWorkflowSecretV6", + "type": "object" + }, + "HostSandboxGrantV2": { + "additionalProperties": false, + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "sandbox", + "default": "sandbox", + "title": "Kind", + "type": "string" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "setting", + "value" + ], + "title": "HostSandboxGrantV2", + "type": "object" + }, + "HostWorkflowAgentLaunchV7": { + "additionalProperties": false, + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with:\n``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read``\nlists the documented permission inputs or flags the step declares in\n``settings``. ``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "properties": { + "agent": { + "enum": [ + "anthropics/claude-code-action", + "anthropics/claude-code-base-action", + "openai/codex-action", + "claude", + "codex" + ], + "title": "Agent", + "type": "string" + }, + "form": { + "enum": [ + "read", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "job_secrets": { + "items": { + "type": "string" + }, + "title": "Job Secrets", + "type": "array" + }, + "settings": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentSettingV7" + }, + "title": "Settings", + "type": "array" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "compound_command", + "shell_expansion", + "expression", + "inputs_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "job", + "step", + "agent", + "form" + ], + "title": "HostWorkflowAgentLaunchV7", + "type": "object" + }, + "HostWorkflowAgentSettingV7": { + "additionalProperties": false, + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, published through the workflow\nlabel redaction (#802); a flag that takes no value has ``null``. A value\nthe redaction rewrites, or one that is not a string, is ``null`` with\n``unresolved_reason``, and records a non-blocking coverage issue naming\nits ``job/step``: it is neither published nor compared.", + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "not_a_string", + "redacted" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + }, + "value": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Value" + } + }, + "required": [ + "name", + "value" + ], + "title": "HostWorkflowAgentSettingV7", + "type": "object" + }, + "HostWorkflowCheckoutRefV7": { + "additionalProperties": false, + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites, a value that is not a string, or ``with:`` that is not a mapping\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "ref": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Ref" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "not_a_string", + "redacted", + "inputs_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + } + }, + "required": [ + "job", + "step", + "ref" + ], + "title": "HostWorkflowCheckoutRefV7", + "type": "object" + }, + "HostWorkflowGrantV7": { + "additionalProperties": false, + "description": "A v0.6 workflow grant plus the agent launches and checkout refs its steps declare.\n\nBoth lists are present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``access`` and ``risk`` still describe the workflow's token and\ntriggers alone.", + "properties": { + "access": { + "enum": [ + "none", + "read", + "write", + "execute", + "external", + "admin", + "unknown" + ], + "title": "Access", + "type": "string" + }, + "agent_launches": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentLaunchV7" + }, + "title": "Agent Launches", + "type": "array" + }, + "checkout_refs": { + "items": { + "$ref": "#/$defs/HostWorkflowCheckoutRefV7" + }, + "title": "Checkout Refs", + "type": "array" + }, + "config_sha256": { + "title": "Config Sha256", + "type": "string" + }, + "effective_write_scopes": { + "items": { + "type": "string" + }, + "title": "Effective Write Scopes", + "type": "array" + }, + "grant_id": { + "title": "Grant Id", + "type": "string" + }, + "host": { + "enum": [ + "codex", + "claude-code", + "cursor", + "vscode", + "github" + ], + "title": "Host", + "type": "string" + }, + "kind": { + "const": "workflow", + "default": "workflow", + "title": "Kind", + "type": "string" + }, + "permission_contexts": { + "items": { + "$ref": "#/$defs/HostWorkflowPermissionsV4" + }, + "title": "Permission Contexts", + "type": "array" + }, + "pull_request_target": { + "default": false, + "title": "Pull Request Target", + "type": "boolean" + }, + "reusable_calls": { + "items": { + "$ref": "#/$defs/HostReusableWorkflowCallV6" + }, + "title": "Reusable Calls", + "type": "array" + }, + "risk": { + "enum": [ + "none", + "low", + "medium", + "high", + "critical", + "unknown" + ], + "title": "Risk", + "type": "string" + }, + "scope": { + "enum": [ + "repository", + "local_static" + ], + "title": "Scope", + "type": "string" + }, + "source": { + "title": "Source", + "type": "string" + }, + "step_actions": { + "items": { + "$ref": "#/$defs/HostWorkflowStepActionV6" + }, + "title": "Step Actions", + "type": "array" + }, + "triggers": { + "items": { + "type": "string" + }, + "title": "Triggers", + "type": "array" + }, + "write_all": { + "default": false, + "title": "Write All", + "type": "boolean" + }, + "write_scopes": { + "items": { + "type": "string" + }, + "title": "Write Scopes", + "type": "array" + } + }, + "required": [ + "grant_id", + "host", + "scope", + "source", + "config_sha256", + "access", + "risk", + "permission_contexts", + "effective_write_scopes", + "reusable_calls" + ], + "title": "HostWorkflowGrantV7", + "type": "object" + }, + "HostWorkflowPermissionsV4": { + "additionalProperties": false, + "properties": { + "job": { + "title": "Job", + "type": "string" + }, + "permissions": { + "additionalProperties": { + "enum": [ + "read", + "write" + ], + "type": "string" + }, + "title": "Permissions", + "type": "object" + }, + "state": { + "enum": [ + "explicit", + "repository_default", + "unresolved" + ], + "title": "State", + "type": "string" + } + }, + "required": [ + "job", + "state", + "permissions" + ], + "title": "HostWorkflowPermissionsV4", + "type": "object" + }, + "HostWorkflowStepActionV6": { + "additionalProperties": false, + "description": "One step's declared action reference, read as text and never fetched.\n\n``form`` is ``remote`` for ``owner/repo[/path]@ref``, ``docker`` for\n``docker://\u2026``, and ``unresolved`` for a value Shipgate does not resolve\nto an action identity; ``unresolved_reason`` then says which. A job whose\n``steps`` is not a list, or a step that is not a mapping, is listed as\nunresolved too, with no ``uses``, so an absent list still means the steps\nwere read and declare nothing. A local\n``./\u2026`` reference is not listed: composite actions remain unread (#701).\n``step`` is the step's ``id``, else its ``name``, else ``steps[N]`` \u2014 the\nevidence a reviewer uses to find it, not part of the comparison. ``job``\nand ``step`` are published labels: credential-shaped text in either, and\nthe userinfo of any ``scheme://\u2026@`` inside it, is redacted (#802).", + "properties": { + "form": { + "enum": [ + "remote", + "docker", + "unresolved" + ], + "title": "Form", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "step": { + "title": "Step", + "type": "string" + }, + "unresolved_reason": { + "anyOf": [ + { + "enum": [ + "expression", + "unsupported_reference", + "not_a_string", + "redacted", + "steps_not_a_list", + "step_not_a_mapping" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Unresolved Reason" + }, + "uses": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Uses" + } + }, + "required": [ + "job", + "step", + "uses", + "form" + ], + "title": "HostWorkflowStepActionV6", + "type": "object" + }, + "InstructionStructureEvidence": { + "additionalProperties": false, + "properties": { + "profile": { + "title": "Profile", + "type": "string" + }, + "reason": { + "title": "Reason", + "type": "string" + }, + "sha256": { + "anyOf": [ + { + "pattern": "^sha256:[0-9a-f]{64}$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Sha256" + }, + "status": { + "enum": [ + "guidance", + "structured", + "unresolved" + ], + "title": "Status", + "type": "string" + } + }, + "required": [ + "profile", + "status", + "reason" + ], + "title": "InstructionStructureEvidence", + "type": "object" + } + }, + "$id": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.7.json", + "$ref": "#/$defs/HostGrantsInventoryV7", + "$schema": "https://json-schema.org/draft/2020-12/schema", + "description": "JSON Schema for shipgate audit --host --json. The inventory summarizes local coding-agent host grants and does not gate releases.", + "title": "Agents Shipgate Host Grants Inventory v0.7" +} diff --git a/docs/integrations.md b/docs/integrations.md index edc5f1ab3..8e55adec6 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -224,7 +224,11 @@ Without a configured manifest, when every changed file is host configuration — stays quiet when no row widens what the agent can do. A workflow step moved to a different action reference, such as a pinned SHA to `@main`, is a non-widening row, so the hook stays quiet about it; `diff` and the PR comment -still show it. It names each widening +still show it. The same holds for an agent launch in a workflow whose settings +change without gaining a documented widening rule, and for a checkout's ref. +One that gains a rule, such as `claude_args` gaining +`--dangerously-skip-permissions`, widens, and the hook announces it (#823). +It names each widening row once, and repeats the announcement only when the change or its rows change. A missing base ref, an incomparable inventory or unparsed output is never quiet. When host configuration changes beside other files, the Stop hook diff --git a/docs/passed-verdict-contract.md b/docs/passed-verdict-contract.md index ab8e336fe..99221e3c5 100644 --- a/docs/passed-verdict-contract.md +++ b/docs/passed-verdict-contract.md @@ -1,6 +1,6 @@ # Evidence-backed `passed` verdict -In the Agents Shipgate `1.0.0` runtime (contract v40, report schema v1.0), +In the Agents Shipgate `1.0.0` runtime (contract v41, report schema v1.0), `release_decision.decision: passed` means the configured root agent and its complete reachable tool/handoff graph were statically proven, and every reachable capability has complete, conflict-free static identity, diff --git a/llms-full.txt b/llms-full.txt index 6d3f4110f..3b780438f 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -857,9 +857,9 @@ For the short, current statement of "which fields to read", see [`docs/agent-con | Verifier schema (current) | [`docs/verifier-schema.v0.21.json`](docs/verifier-schema.v0.21.json) | `0.21` | | Agent handoff schema (current) | [`docs/agent-handoff-schema.v9.json`](docs/agent-handoff-schema.v9.json) | `shipgate.agent_handoff/v9` | | Preflight schema (current) | [`docs/preflight-schema.v0.5.json`](docs/preflight-schema.v0.5.json) | `0.5` | -| Host-grants inventory schema | [`docs/host-grants-inventory-schema.v0.6.json`](docs/host-grants-inventory-schema.v0.6.json) | `0.6` | -| Host-grants baseline schema | [`docs/host-grants-baseline-schema.v0.6.json`](docs/host-grants-baseline-schema.v0.6.json) | `0.6` | -| Host-grants drift schema | [`docs/host-grants-drift-schema.v0.6.json`](docs/host-grants-drift-schema.v0.6.json) | `0.6` | +| Host-grants inventory schema | [`docs/host-grants-inventory-schema.v0.7.json`](docs/host-grants-inventory-schema.v0.7.json) | `0.7` | +| Host-grants baseline schema | [`docs/host-grants-baseline-schema.v0.7.json`](docs/host-grants-baseline-schema.v0.7.json) | `0.7` | +| Host-grants drift schema | [`docs/host-grants-drift-schema.v0.7.json`](docs/host-grants-drift-schema.v0.7.json) | `0.7` | | Capability standard | [`docs/capability-standard.md`](docs/capability-standard.md) | `0.5` | | Capability lock schema | [`docs/capability-lock-schema.v0.8.json`](docs/capability-lock-schema.v0.8.json) | `0.8` | | Capability lock diff schema | [`docs/capability-lock-diff-schema.v0.9.json`](docs/capability-lock-diff-schema.v0.9.json) | `0.9` | @@ -1618,6 +1618,30 @@ directory, still refuses its comparison. A `0.20` verifier claiming a partial comparison or a `scope` is refused. See [the migration note](../STABILITY.md#partial-host-comparison-808). +Runtime contract v41, unreleased, reads how a coding agent is launched inside a +workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so +host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow +grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. +An agent launch is a step whose `uses:` is a documented agent action +(`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, +`openai/codex-action`) with the permission inputs it declares, or a `run:` +that is one literal `claude -p` / `codex exec` command with its documented +permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`) and +`job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, +`null` for the default. Values are compared as text and never executed. Only a +documented rule a job's launches gain — bypassed permission checks, a bypassed +or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate +opened to `*` — raises `workflow_agent_widened_` and makes the +row `widened`; every other edit is `changed`, and a workflow row that runs an +agent ends its `why` with the job facts beside each agent step. A compound +`run:`, an expansion or an expression is `unresolved` and a named non-blocking +limit. A `0.4`–`0.6` baseline holding a workflow grant is incomparable +(`baseline_workflow_agent_launches_unavailable`); one without a workflow stays +comparable. Verifier `0.20`, capability diff `0.3` and +`minimum_control_contract_version` `21` are unchanged. See +[the migration note](../STABILITY.md#workflow-agent-launches-contract-v41-823). + Previous runtime contract v40 reads the action reference each workflow step declares (#771). Host-grants inventory, baseline and drift schemas move to `0.6`, and a workflow grant adds `step_actions[]`: the job, the step (`id`, else `name`, @@ -2309,7 +2333,7 @@ Downstream repos generated with - Current attestation schema: `0.5` — [`docs/attestation-schema.v0.5.json`](attestation-schema.v0.5.json) - Current registry schema: `0.4` — [`docs/registry-schema.v0.4.json`](registry-schema.v0.4.json) - Current org evidence bundle schema: `shipgate.org_evidence_bundle/v2` — [`docs/org-evidence-bundle-schema.v2.json`](org-evidence-bundle-schema.v2.json) -- Current host-grants inventory, baseline, and drift schemas: `0.6` — [`inventory`](host-grants-inventory-schema.v0.6.json), [`baseline`](host-grants-baseline-schema.v0.6.json), [`drift`](host-grants-drift-schema.v0.6.json) +- Current host-grants inventory, baseline, and drift schemas: `0.7` — [`inventory`](host-grants-inventory-schema.v0.7.json), [`baseline`](host-grants-baseline-schema.v0.7.json), [`drift`](host-grants-drift-schema.v0.7.json) - Current trigger catalog schema: `0.4` — [`docs/triggers.json`](triggers.json) - Current governance benchmark catalog schema: `0.2` — [`docs/governance-benchmark-catalog-schema.v0.2.json`](governance-benchmark-catalog-schema.v0.2.json) - Current governance benchmark result schema: `0.2` — [`docs/governance-benchmark-result-schema.v0.2.json`](governance-benchmark-result-schema.v0.2.json) diff --git a/llms.txt b/llms.txt index d1abc6abe..1f088484d 100644 --- a/llms.txt +++ b/llms.txt @@ -87,9 +87,9 @@ - Attestation schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/attestation-schema.v0.5.json - Registry schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/registry-schema.v0.4.json - Org evidence bundle schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/org-evidence-bundle-schema.v2.json -- Host-grants inventory schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.6.json -- Host-grants baseline schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.6.json -- Host-grants drift schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.6.json +- Host-grants inventory schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.7.json +- Host-grants baseline schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.7.json +- Host-grants drift schema (current): https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.7.json - Capability standard: https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/capability-standard.md - Governance benchmark catalog/result schemas: https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/governance-benchmark-catalog-schema.v0.2.json and https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/governance-benchmark-result-schema.v0.2.json - Check catalog: https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/checks.json diff --git a/scripts/generate_schemas.py b/scripts/generate_schemas.py index f55c7b422..b3f6740ac 100644 --- a/scripts/generate_schemas.py +++ b/scripts/generate_schemas.py @@ -51,13 +51,13 @@ - docs/registry-schema.v0.4.json (from agents_shipgate.schemas.registry. RegistryQueryResultV1) -- docs/host-grants-inventory-schema.v0.6.json +- docs/host-grants-inventory-schema.v0.7.json (from agents_shipgate.schemas.host_grants. - HostGrantsInventoryArtifactV6) -- docs/host-grants-baseline-schema.v0.6.json - (from HostGrantsBaselineArtifactV6) -- docs/host-grants-drift-schema.v0.6.json - (from HostGrantsDriftArtifactV6) + HostGrantsInventoryArtifactV7) +- docs/host-grants-baseline-schema.v0.7.json + (from HostGrantsBaselineArtifactV7) +- docs/host-grants-drift-schema.v0.7.json + (from HostGrantsDriftArtifactV7) - docs/capability-lock-schema.v0.8.json (from agents_shipgate.schemas.capabilities. CapabilityLockFileArtifactV1) @@ -2494,10 +2494,10 @@ def build_host_grants_inventory_schema() -> tuple[Path, str]: from agents_shipgate.schemas.host_grants import ( HOST_GRANTS_INVENTORY_SCHEMA_VERSION, - HostGrantsInventoryArtifactV6, + HostGrantsInventoryArtifactV7, ) - schema = HostGrantsInventoryArtifactV6.model_json_schema() + schema = HostGrantsInventoryArtifactV7.model_json_schema() minor = HOST_GRANTS_INVENTORY_SCHEMA_VERSION schema["$id"] = ( "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/" @@ -2518,10 +2518,10 @@ def build_host_grants_baseline_schema() -> tuple[Path, str]: from agents_shipgate.schemas.host_grants import ( HOST_GRANTS_BASELINE_SCHEMA_VERSION, - HostGrantsBaselineArtifactV6, + HostGrantsBaselineArtifactV7, ) - schema = HostGrantsBaselineArtifactV6.model_json_schema() + schema = HostGrantsBaselineArtifactV7.model_json_schema() minor = HOST_GRANTS_BASELINE_SCHEMA_VERSION schema["$id"] = ( "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/" @@ -2541,10 +2541,10 @@ def build_host_grants_drift_schema() -> tuple[Path, str]: from agents_shipgate.schemas.host_grants import ( HOST_GRANTS_DRIFT_SCHEMA_VERSION, - HostGrantsDriftArtifactV6, + HostGrantsDriftArtifactV7, ) - schema = HostGrantsDriftArtifactV6.model_json_schema() + schema = HostGrantsDriftArtifactV7.model_json_schema() minor = HOST_GRANTS_DRIFT_SCHEMA_VERSION schema["$id"] = ( "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/" diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index e6159a263..f5f016324 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -27,11 +27,17 @@ from typing import Any from agents_shipgate.core.host_grants import ( + AGENT_WIDENING_RULES, + UNTRUSTED_INPUT_TRIGGERS, + agent_launch_key, + checkout_ref_key, + gained_agent_widenings, hook_loading_basis, host_grant_expansion_signals, permission_rule_replacements, published_setting_value, published_workflow_label, + pull_request_code_ref, secret_mapping_key, step_action_key, ) @@ -237,6 +243,210 @@ def _moved_between_jobs( moved.append((item, match)) return moved, [*changed, *arriving] + +def _job_entry_changes( + before: dict[str, Any] | None, + after: dict[str, Any] | None, + field: str, + key: Any, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """The entries of ``field`` only one side of a changed workflow declares (#823). + + Compared by ``key``, the way the comparator decides whether the grant + changed, so a renamed or reordered step appears on neither side. + """ + + if not _is_workflow_pair(before, after): + return [], [] + + def only_in(side: list[dict[str, Any]], other: list[dict[str, Any]]) -> list[dict[str, Any]]: + surplus = Counter(key(item) for item in side) - Counter(key(item) for item in other) + picked: list[dict[str, Any]] = [] + for item in side: + if surplus[key(item)] > 0: + surplus[key(item)] -= 1 + picked.append(item) + return picked + + old, new = (before or {}).get(field, []), (after or {}).get(field, []) + return only_in(old, new), only_in(new, old) + + +def _agent_label(agent: str) -> str: + """How a reviewer recognises the agent a step launches.""" + + return {"claude": "claude -p", "codex": "codex exec"}.get(agent, agent) + + +def _agent_launch_value(item: dict[str, Any]) -> str: + """One agent launch as a cell shows it: where, which agent, and its declared settings.""" + + where = f"{item['job']}/{item['step']}" + label = _agent_label(str(item["agent"])) + reason = item.get("unresolved_reason") + if reason: + return f"{where}: runs {label} (unresolved: {str(reason).replace('_', ' ')})" + cli = item["agent"] in {"claude", "codex"} + parts = [] + for setting in item.get("settings") or []: + unread = setting.get("unresolved_reason") + name = str(setting["name"]) + if unread: + parts.append(f"{name} (unresolved: {str(unread).replace('_', ' ')})") + elif setting.get("value") is None: + parts.append(name) + else: + parts.append(f"{name} {setting['value']}" if cli else f"{name}: {setting['value']}") + if not parts: + return f"{where}: runs {label} with no permission {'flags' if cli else 'inputs'}" + return f"{where}: runs {label} with " + "; ".join(parts) + + +def _checkout_ref_value(item: dict[str, Any]) -> str: + where = f"{item['job']}/{item['step']}" + reason = item.get("unresolved_reason") + if reason: + return f"{where}: checkout ref (unresolved: {str(reason).replace('_', ' ')})" + if item.get("ref") is None: + return f"{where}: checkout of the default ref" + return f"{where}: checkout of ref {item['ref']}" + + +def _agent_launch_reasons( + before: dict[str, Any] | None, + after: dict[str, Any] | None, + gone: list[dict[str, Any]], + new: list[dict[str, Any]], + gone_checkouts: list[dict[str, Any]], + new_checkouts: list[dict[str, Any]], +) -> list[str]: + """What changed in how an agent is launched, and which of it widens (#823). + + Only a documented rule gained by a job's agent launches is called a + widening, and the sentence says which rule and where. Every other + agent-launch or checkout edit is a change: its settings are compared as + declared text, and nothing here ranks one value against another. + """ + + def where(item: dict[str, Any]) -> str: + return f"{item['job']}/{item['step']}" + + reasons: list[str] = [] + widened_at: set[str] = set() + for _job, rule, detail, entry in gained_agent_widenings(before, after): + widened_at.add(where(entry)) + what = AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") + reasons.append(f"an agent launch now {what} ({where(entry)})") + # The step label only words the sentence; what changed was decided by the + # comparator's key, which never reads it. + old = {where(item) for item in gone} + now = {where(item) for item in new} + groups: dict[str, list[str]] = {} + for item in (*new, *gone): + label = where(item) + if label in widened_at: + continue + verb = "changed" if label in old and label in now else ("added" if label in now else "removed") + groups.setdefault(verb, []) + if label not in groups[verb]: + groups[verb].append(label) + phrases = [ + f"{wording} ({', '.join(groups[verb])})" + for verb, wording in ( + ("changed", "an agent launch's declared settings changed"), + ("added", "a step now launches an agent"), + ("removed", "a step no longer launches an agent"), + ) + if verb in groups + ] + if phrases: + reasons.append( + f"{_joined_words(phrases)}; agent launch settings are compared as declared text, " + "and a change that gains no documented widening rule is not counted as a widening" + ) + unread = list(dict.fromkeys(where(item) for item in new if item.get("form") != "read")) + if unread: + reasons.append( + f"a step launches an agent in a form this audit does not read ({', '.join(unread)}); " + "its settings are not compared, so this row does not say what that agent may do" + ) + checkouts = list(dict.fromkeys( + f"{item['job']}/{item['step']}" for item in (*gone_checkouts, *new_checkouts) + )) + if checkouts: + reasons.append( + f"a checkout's declared ref changed ({', '.join(checkouts)}); a ref names which " + "commit's code the job runs and adds no scope" + ) + return reasons + + +#: How many agent steps one note names before counting the rest. +_NOTE_STEP_LIMIT = 5 + + +def _agent_composition_note(grant: dict[str, Any]) -> str | None: + """The job facts beside each agent step, as a note on the workflow's row (#823). + + Named, never scored: an untrusted-input trigger, the job's write scopes, + the secrets the job references and a checkout of pull request code in the + job. It is not a verdict and moves no direction; it says where on the + workflow an agent already runs with those facts. + """ + + launches = grant.get("agent_launches") or [] + if not launches: + return None + triggers = [name for name in grant.get("triggers", []) if name in UNTRUSTED_INPUT_TRIGGERS] + contexts = {context["job"]: context for context in grant.get("permission_contexts", [])} + by_job: dict[str, list[dict[str, Any]]] = {} + for item in launches: + by_job.setdefault(str(item["job"]), []).append(item) + notes: list[str] = [] + named = 0 + for job, items in by_job.items(): + shown = items[: max(0, _NOTE_STEP_LIMIT - named)] + if not shown: + break + named += len(shown) + steps = ", ".join( + f"{item['job']}/{item['step']} ({_agent_label(str(item['agent']))})" for item in shown + ) + facts: list[str] = [] + if triggers: + noun = "trigger" if len(triggers) == 1 else "triggers" + facts.append(f"the untrusted-input {noun} {_joined_words(triggers)}") + context = contexts.get(job) or {} + writes = [scope for scope, level in (context.get("permissions") or {}).items() if level == "write"] + if "*" in writes: + facts.append("write-all token permissions") + elif writes: + noun = "scope" if len(writes) == 1 else "scopes" + facts.append(f"the write {noun} {_joined_words(writes)}") + secrets = sorted({name for item in items for name in item.get("job_secrets", [])}) + if secrets: + noun = "secret" if len(secrets) == 1 else "secrets" + facts.append(f"the {noun} {_joined_words(secrets)}") + pull_request_code = [ + f"{checkout['job']}/{checkout['step']}" + for checkout in grant.get("checkout_refs", []) + if checkout["job"] == job and pull_request_code_ref(checkout.get("ref")) + ] + if pull_request_code: + facts.append(f"a checkout of pull request code ({', '.join(pull_request_code)})") + note = f"an agent runs at {steps}" + if facts: + note += " beside " + _joined_words(facts) + notes.append(note) + rest = len(launches) - named + if rest: + notes.append(f"{rest} more agent step(s) run in this workflow") + return "; ".join(notes) + + +def _joined_words(items: list[str]) -> str: + return items[0] if len(items) == 1 else f"{', '.join(items[:-1])} and {items[-1]}" + #: Direction is deliberately coarse here. Presence is certain: a grant is #: in one side and not the other. *Width* is not — deciding that #: `Bash(npm *)` -> `Bash(npm test:*)` narrows needs the pattern lattice in @@ -254,14 +464,16 @@ def _grant_value( redact_permission_arguments: bool = False, step_actions: list[dict[str, Any]] | None = None, secret_mappings: list[SecretMapping] | None = None, + agent_launches: list[dict[str, Any]] | None = None, + checkout_refs: list[dict[str, Any]] | None = None, ) -> str: """What a reader recognises this grant by. A workflow has no single name — its authority *is* the combination of access and triggers, so both sides render that combination or the row - reads "workflow -> workflow" and says nothing. ``step_actions`` and - ``secret_mappings`` are the step references and named secrets this side - alone declares; unchanged ones are not repeated. + reads "workflow -> workflow" and says nothing. ``step_actions``, + ``secret_mappings``, ``agent_launches`` and ``checkout_refs`` are the + entries this side alone declares; unchanged ones are not repeated. """ if not grant: @@ -298,6 +510,8 @@ def _grant_value( parts.append(f"{call['job']}: {forwarding}{call['uses']}") parts.extend(_secret_mapping_value(item) for item in secret_mappings or []) parts.extend(_step_action_value(item) for item in step_actions or []) + parts.extend(_checkout_ref_value(item) for item in checkout_refs or []) + parts.extend(_agent_launch_value(item) for item in agent_launches or []) return ", ".join(part for part in parts if part) or kind if kind in _SETTING_KINDS and grant.get("setting"): # A setting row read `True` or `dontAsk` alone, which names no setting @@ -330,11 +544,14 @@ def _why( new_steps: list[dict[str, Any]] | None = None, gone_secrets: list[SecretMapping] | None = None, new_secrets: list[SecretMapping] | None = None, + agent_reasons: list[str] | None = None, ) -> str: """Why a reviewer should care, in the reviewer's terms. Stated as what the grant *permits*, never as a prediction about what the agent will do with it — the engine reads configuration, not behaviour. + A workflow that launches an agent ends with the job facts beside each + agent step (#823), read off ``grant``, the side the row describes. """ kind = str(grant.get("kind") or "") @@ -390,7 +607,11 @@ def _why( + "); it names different code to run with that job's existing " "token permissions and adds no scope" ) - return "; ".join(reasons) or "changes the workflow's own authority" + reasons.extend(agent_reasons or []) + why = "; ".join(reasons) or "changes the workflow's own authority" + # A removed workflow runs nothing any more, so it gets no note. + note = None if direction == REMOVED else _agent_composition_note(grant) + return f"{why}; {note}" if note else why if kind == "hook": # The basis, stated in the row, because the row is what a reviewer # reads: a parsed hook file is not proof a host loads it (#714). @@ -823,10 +1044,24 @@ def capability_diff_rows( direction = WIDENED gone_steps, new_steps = _step_action_changes(before_grant, after_grant) gone_secrets, new_secrets = _secret_mapping_changes(before_grant, after_grant) + gone_agents, new_agents = _job_entry_changes( + before_grant, after_grant, "agent_launches", agent_launch_key + ) + gone_checkouts, new_checkouts = _job_entry_changes( + before_grant, after_grant, "checkout_refs", checkout_ref_key + ) + agent_reasons = ( + _agent_launch_reasons( + before_grant, after_grant, gone_agents, new_agents, gone_checkouts, new_checkouts + ) + if grant.get("kind") == "workflow" + else [] + ) why = _why( grant, direction, gone_steps=gone_steps, new_steps=new_steps, gone_secrets=gone_secrets, new_secrets=new_secrets, + agent_reasons=agent_reasons, ) row = CapabilityDiffRow( subject=_subject(grant), @@ -835,12 +1070,16 @@ def capability_diff_rows( redact_permission_arguments=redact_permission_arguments, step_actions=gone_steps, secret_mappings=gone_secrets, + agent_launches=gone_agents, + checkout_refs=gone_checkouts, ), after=_grant_value( after_grant, redact_permission_arguments=redact_permission_arguments, step_actions=new_steps, secret_mappings=new_secrets, + agent_launches=new_agents, + checkout_refs=new_checkouts, ), direction=direction, why=why, diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 4587315da..26a6ebfb6 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -14,6 +14,7 @@ import os import posixpath import re +import shlex import stat import sys import tomllib @@ -83,8 +84,9 @@ HostGrantsBaselineV4, HostGrantsBaselineV5, HostGrantsBaselineV6, - HostGrantsDriftV6, - HostGrantsInventoryV6, + HostGrantsBaselineV7, + HostGrantsDriftV7, + HostGrantsInventoryV7, ) HOST_GRANTS_SCHEMA_VERSION = HOST_GRANTS_BASELINE_SCHEMA_VERSION @@ -1601,6 +1603,611 @@ def step_action_key(entry: dict[str, Any]) -> tuple[str, str, str, str]: ) +# --- agent launches (#823) ---------------------------------------------------------- +# +# How a coding agent is launched inside a job: a known agent action's permission +# inputs, a literal agent CLI command in a `run:` step, and the ref each +# `actions/checkout` step declares. Every value is compared as declared text; no +# action is fetched, no command is run and no expression is evaluated. A shape +# this reader does not read — a compound shell command, an expansion, a script, +# a composite or unknown action — is never guessed at: where it can be told +# apart it is listed as unresolved and named as a non-blocking limit, and +# otherwise it is one of the unread surfaces the support page names. + + +@dataclass(frozen=True) +class _AgentAction: + """What one documented agent action's inputs mean to this reader (#823). + + ``inputs`` are compared as text. ``gates`` are comma-separated user lists + where a ``*`` entry opens the gate to every user. ``args`` names the input + that carries agent CLI arguments, read with the family's flag table for + the documented widening rules. ``modes`` are inputs whose value is itself + a documented widening. + """ + + family: Literal["claude", "codex"] + inputs: tuple[str, ...] + gates: tuple[str, ...] = () + args: str | None = None + modes: tuple[tuple[str, str], ...] = () + + +#: The documented agent actions, by ``owner/repo`` (matched case-insensitively, +#: at any ref). The input names are those the actions' own `action.yml` +#: declare; `allowed_tools`, `disallowed_tools` and `mcp_config` are the +#: Claude actions' earlier inputs, still read when a workflow sets them. +_AGENT_ACTIONS: dict[str, _AgentAction] = { + "anthropics/claude-code-action": _AgentAction( + family="claude", + inputs=( + "additional_permissions", "allowed_bots", "allowed_non_write_users", + "allowed_tools", "claude_args", "disallowed_tools", "mcp_config", + "plugin_marketplaces", "plugins", "settings", + ), + gates=("allowed_bots", "allowed_non_write_users"), + args="claude_args", + ), + "anthropics/claude-code-base-action": _AgentAction( + family="claude", + inputs=( + "allowed_tools", "claude_args", "disallowed_tools", "mcp_config", + "plugin_marketplaces", "plugins", "settings", + ), + args="claude_args", + ), + "openai/codex-action": _AgentAction( + family="codex", + inputs=( + "allow-bot-users", "allow-bots", "allow-users", "codex-args", + "permission-profile", "safety-strategy", "sandbox", + ), + gates=("allow-users",), + args="codex-args", + modes=(("sandbox", "danger-full-access"), ("safety-strategy", "unsafe")), + ), +} + +#: Flag spelling -> (primary spelling, arity): ``0`` takes no value, ``1`` one, +#: ``None`` every following word up to the next one starting with ``-``, as the +#: CLI's variadic options read them. From Claude Code's CLI reference. +_CLAUDE_FLAGS: dict[str, tuple[str, int | None]] = { + "--permission-mode": ("--permission-mode", 1), + "--dangerously-skip-permissions": ("--dangerously-skip-permissions", 0), + "--allow-dangerously-skip-permissions": ("--allow-dangerously-skip-permissions", 0), + "--allowedTools": ("--allowedTools", None), + "--allowed-tools": ("--allowedTools", None), + "--disallowedTools": ("--disallowedTools", None), + "--disallowed-tools": ("--disallowedTools", None), + "--add-dir": ("--add-dir", None), + "--mcp-config": ("--mcp-config", None), + "--settings": ("--settings", 1), + "--permission-prompt-tool": ("--permission-prompt-tool", 1), +} + +#: The same for ``codex exec``, from the Codex CLI's shared option definitions. +_CODEX_FLAGS: dict[str, tuple[str, int | None]] = { + "--sandbox": ("--sandbox", 1), + "-s": ("--sandbox", 1), + "--dangerously-bypass-approvals-and-sandbox": ("--dangerously-bypass-approvals-and-sandbox", 0), + "--yolo": ("--dangerously-bypass-approvals-and-sandbox", 0), + "--approve-for-me": ("--approve-for-me", 0), + "--not-so-yolo": ("--approve-for-me", 0), + "--dangerously-bypass-hook-trust": ("--dangerously-bypass-hook-trust", 0), + "--add-dir": ("--add-dir", 1), + "--config": ("--config", 1), + "-c": ("--config", 1), + "--profile": ("--profile", 1), + "-p": ("--profile", 1), +} + +_AGENT_FLAG_TABLES = {"claude": _CLAUDE_FLAGS, "codex": _CODEX_FLAGS} + +#: The widening each documented rule names, as a row's ``why`` says it. +AGENT_WIDENING_RULES: dict[str, str] = { + "bypass_permissions": "skips permission checks (bypassPermissions)", + "bypass_approvals_and_sandbox": "bypasses approvals and the sandbox", + "danger_full_access": "runs without a sandbox (danger-full-access)", + "unsafe_safety_strategy": "runs without privilege restrictions (safety-strategy: unsafe)", + "open_gate": "accepts runs triggered by any user", +} + +#: A literal checkout ref that names pull request code, not the base branch's: +#: the documented pull request and workflow-run head expressions, and +#: ``refs/pull//head`` or ``/merge``. +_PULL_REQUEST_CODE_REF_RE = re.compile( + r"\$\{\{\s*(?:github\.event\.pull_request\.(?:head\.(?:sha|ref)|merge_commit_sha)" + r"|github\.head_ref|github\.event\.workflow_run\.head_(?:sha|branch))\s*\}\}" + r"|refs/pull/(?:\d+|\$\{\{[^}]*\}\})/(?:head|merge)" +) + +#: Triggers that run with the base repository's context on input from people +#: who need not hold write access, as the support page lists them. +UNTRUSTED_INPUT_TRIGGERS = ("issue_comment", "issues", "pull_request_target", "workflow_run") + +_ASSIGNMENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*=") +_SHELL_OPERATOR_CHARS = ";&|()<>\n" +_SECRET_REFERENCE_RE = re.compile(r"\bsecrets\.([A-Za-z_][A-Za-z0-9_]*)") +_EXPRESSION_RE = re.compile(r"\$\{\{(.*?)\}\}", re.S) + + +def pull_request_code_ref(ref: str | None) -> bool: + """Whether a published checkout ref names pull request code (#823).""" + + return ref is not None and _PULL_REQUEST_CODE_REF_RE.fullmatch(ref.strip()) is not None + + +def _shell_words(text: str) -> list[str] | None: + """``text`` split into shell words and operator tokens, or ``None`` when unbalanced. + + Newlines, ``;``, ``&``, ``|``, parentheses and redirections outside quotes + are operator tokens of their own; a backslash-newline joins two lines, as + the shell reads it. + """ + + lexer = shlex.shlex( + re.sub(r"\\\r?\n", "", text), posix=True, punctuation_chars=_SHELL_OPERATOR_CHARS + ) + lexer.whitespace = " \t\r" + lexer.whitespace_split = True + lexer.commenters = "" + try: + return list(lexer) + except ValueError: + return None + + +def _is_operator(token: str) -> bool: + return bool(token) and all(char in _SHELL_OPERATOR_CHARS for char in token) + + +def _has_shell_expansion(text: str) -> bool: + """A ``$`` or backtick the shell would expand: anywhere outside single quotes.""" + + quote: str | None = None + escaped = False + for char in text: + if escaped: + escaped = False + elif char == "\\" and quote != "'": + escaped = True + elif quote == "'": + quote = None if char == "'" else quote + elif char in {"$", "`"}: + return True + elif char == quote: + quote = None + elif quote is None and char in {"'", '"'}: + quote = char + return False + + +def _agent_command(words: list[str]) -> tuple[str, list[str]] | None: + """The agent CLI one simple command launches headless, and its arguments. + + ``claude`` with ``-p``/``--print`` anywhere in its arguments, or ``codex`` + with ``exec`` (or its alias ``e``) as its first argument, after any leading + ``NAME=value`` assignments. Any other command — ``claude mcp add``, + ``codex login``, an ``echo`` that mentions either — launches none. + """ + + index = 0 + while index < len(words) and _ASSIGNMENT_RE.match(words[index]): + index += 1 + if index >= len(words): + return None + command, arguments = words[index], words[index + 1:] + if command == "claude" and any(word in {"-p", "--print"} for word in arguments): + return "claude", arguments + if command == "codex" and arguments[:1] in (["exec"], ["e"]): + return "codex", arguments[1:] + return None + + +def _line_agent(line: str) -> str | None: + """The agent CLI one line starts headless, read as plain words: a fallback only. + + Used when a ``run:``'s quoting cannot be split into commands, to name the + launch as unresolved rather than miss it; nothing it reads is published. + """ + + return (_agent_command(line.split()) or (None, None))[0] + + +def _read_flags(words: list[str], table: dict[str, tuple[str, int | None]]) -> list[tuple[str, str | None]]: + """The documented permission flags among ``words``, each with its declared value. + + A flag is read by name wherever it is a word of its own; every other word + — the prompt, ``--model`` and any undocumented flag — is not compared. + """ + + flags: list[tuple[str, str | None]] = [] + index = 0 + while index < len(words): + word = words[index] + name, equals, attached = word.partition("=") if word.startswith("--") else (word, "", "") + spec = table.get(name) + index += 1 + if spec is None: + continue + primary, arity = spec + if arity == 0: + flags.append((primary, None)) + continue + values = [attached] if equals else [] + if arity == 1 and not equals and index < len(words): + values.append(words[index]) + index += 1 + elif arity is None: + while index < len(words) and not words[index].startswith("-"): + values.append(words[index]) + index += 1 + if not values: + flags.append((primary, None)) + else: + flags.append((primary, values[0] if arity == 1 else shlex.join(values))) + return flags + + +def _published_setting(name: str, value: Any) -> dict[str, Any]: + """One setting as it may be published: declared text, or why it is not.""" + + if isinstance(value, bool): + text = "true" if value else "false" + elif value is None: + text = "" + elif isinstance(value, (int, float)): + text = str(value) + elif isinstance(value, str): + text = value.strip() + else: + return {"name": name, "value": None, "unresolved_reason": "not_a_string"} + shown = published_workflow_label(text) + if shown != text: + return {"name": name, "value": None, "unresolved_reason": "redacted"} + return {"name": name, "value": text, "unresolved_reason": None} + + +def _published_flag(name: str, value: str | None) -> dict[str, Any]: + if value is None: + return {"name": name, "value": None, "unresolved_reason": None} + return _published_setting(name, value) + + +def _setting_key(setting: dict[str, Any]) -> tuple[str, str, str]: + return ( + str(setting["name"]), + str(setting.get("unresolved_reason") or ""), + "" if setting.get("value") is None else str(setting["value"]), + ) + + +def _job_secrets(job: dict[Any, Any], workflow_env: Any) -> list[str]: + """The secret names a job references in ``${{ }}``, and those the workflow ``env`` passes. + + Context for the row that names an agent step in the job, never compared. + Each name is a published label (#802). + """ + + names: set[str] = set() + + def walk(value: Any) -> None: + if isinstance(value, str): + for expression in _EXPRESSION_RE.findall(value): + names.update(_SECRET_REFERENCE_RE.findall(expression)) + elif isinstance(value, dict): + for item in value.values(): + walk(item) + elif isinstance(value, list): + for item in value: + walk(item) + + walk(job) + walk(workflow_env) + return sorted({published_workflow_label(name) for name in names}) + + +def _action_identity(uses: Any) -> str | None: + """A step's ``owner/repo`` in lower case when ``uses:`` is ``owner/repo@ref``.""" + + if not isinstance(uses, str): + return None + text = uses.strip() + if not _REMOTE_STEP_ACTION_RE.fullmatch(text): + return None + return text.rpartition("@")[0].casefold() + + +def _action_launch(job: str, step_label: str, agent: str, step: dict[Any, Any]) -> dict[str, Any]: + spec = _AGENT_ACTIONS[agent] + entry: dict[str, Any] = { + "job": job, "step": step_label, "agent": agent, "form": "read", + "unresolved_reason": None, "settings": [], + } + inputs = step.get("with") + if inputs is None: + return entry + if not isinstance(inputs, dict): + return {**entry, "form": "unresolved", "unresolved_reason": "inputs_not_a_mapping"} + wanted = {name.casefold(): name for name in spec.inputs} + settings = [ + _published_setting(wanted[str(key).casefold()], value) + for key, value in inputs.items() + if str(key).casefold() in wanted + ] + return {**entry, "settings": sorted(settings, key=_setting_key)} + + +def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: + """The agent CLIs a ``run:`` step launches: read when literal, else unresolved. + + Read only when the whole ``run:`` is one simple command that starts with + the agent CLI (after literal ``NAME=value`` assignments), holds no shell + expansion and no ``${{ }}`` expression: then its documented permission + flags are its settings. Otherwise an agent CLI found at the start of any + simple command in it is one ``unresolved`` entry with the reason, and none + of its text is published. + """ + + run = run.strip() + words = _shell_words(run) + base = {"job": job, "step": step_label} + if words is None: + # Quoting that does not balance cannot be split into commands, so no + # word of it is read; a line that starts an agent CLI is still named. + agents = sorted({ + agent for line in run.splitlines() + if (agent := _line_agent(line)) is not None + }) + return [ + {**base, "agent": agent, "form": "unresolved", "unresolved_reason": "compound_command", "settings": []} + for agent in agents + ] + commands: list[list[str]] = [[]] + for word in words: + if _is_operator(word): + commands.append([]) + else: + commands[-1].append(word) + launches = [launch for command in commands if (launch := _agent_command(command))] + if not launches: + return [] + single = len([command for command in commands if command]) == 1 and not any( + _is_operator(word) or word.startswith("#") for word in words + ) + if "${{" in run: + reason: str | None = "expression" + elif not single: + reason = "compound_command" + elif _has_shell_expansion(run): + reason = "shell_expansion" + else: + reason = None + if reason is not None: + agents = sorted({agent for agent, _arguments in launches}) + return [ + {**base, "agent": agent, "form": "unresolved", "unresolved_reason": reason, "settings": []} + for agent in agents + ] + (agent, arguments), = launches + settings = [_published_flag(name, value) for name, value in _read_flags(arguments, _AGENT_FLAG_TABLES[agent])] + return [{ + **base, "agent": agent, "form": "read", "unresolved_reason": None, + "settings": sorted(settings, key=_setting_key), + }] + + +def _step_agent_launches(job: str, step: dict[Any, Any], index: int) -> list[dict[str, Any]]: + """The agent launches one step declares: a known action, or a ``run:`` agent CLI.""" + + if "uses" in step: + agent = _action_identity(step["uses"]) + if agent is not None and agent in _AGENT_ACTIONS: + return [_action_launch(job, _step_label(step, index), agent, step)] + return [] + run = step.get("run") + if isinstance(run, str) and ("claude" in run or "codex" in run): + return _run_launches(job, _step_label(step, index), run) + return [] + + +def _checkout_ref(job: str, step: dict[Any, Any], index: int) -> dict[str, Any] | None: + """An ``actions/checkout`` step and the ``with.ref`` it declares, else ``None``.""" + + if _action_identity(step.get("uses")) != "actions/checkout": + return None + entry: dict[str, Any] = { + "job": job, "step": _step_label(step, index), "ref": None, "unresolved_reason": None, + } + inputs = step.get("with") + if inputs is None: + return entry + if not isinstance(inputs, dict): + return {**entry, "unresolved_reason": "inputs_not_a_mapping"} + if "ref" not in inputs: + return entry + setting = _published_setting("ref", inputs["ref"]) + if setting["unresolved_reason"] is not None: + return {**entry, "unresolved_reason": setting["unresolved_reason"]} + return {**entry, "ref": setting["value"] or None} + + +def agent_launch_key(entry: dict[str, Any]) -> tuple[Any, ...]: + """What an agent launch is compared by: its job, agent, form and settings, not its step.""" + + return ( + str(entry["job"]), + str(entry["agent"]), + str(entry["form"]), + str(entry.get("unresolved_reason") or ""), + tuple(sorted(_setting_key(setting) for setting in entry.get("settings", []))), + ) + + +def checkout_ref_key(entry: dict[str, Any]) -> tuple[str, str, str]: + """What a checkout ref is compared by: its job and the declared ref, not its step.""" + + return ( + str(entry["job"]), + str(entry.get("unresolved_reason") or ""), + "" if entry.get("ref") is None else str(entry["ref"]), + ) + + +def _argument_words(value: str, *, json_list: bool) -> list[str] | None: + """An arguments input's words, or ``None`` when they cannot be read literally.""" + + if "${{" in value: + return None + if json_list and value.startswith("["): + try: + loaded = json.loads(value) + except json.JSONDecodeError: + return None + if isinstance(loaded, list) and all(isinstance(item, str) for item in loaded): + return loaded + return None + words = _shell_words(value) + if words is None or any(_is_operator(word) for word in words): + return None + return words + + +def _flag_rules(family: str, flags: list[tuple[str, str | None]]) -> set[str]: + rules: set[str] = set() + for name, value in flags: + if family == "claude" and ( + name == "--dangerously-skip-permissions" + or (name == "--permission-mode" and value == "bypassPermissions") + ): + rules.add("bypass_permissions") + if family == "codex" and name == "--dangerously-bypass-approvals-and-sandbox": + rules.add("bypass_approvals_and_sandbox") + if family == "codex" and name == "--sandbox" and value == "danger-full-access": + rules.add("danger_full_access") + return rules + + +def agent_widening_rules(entry: dict[str, Any]) -> set[tuple[str, str]]: + """The documented widening rules one published agent launch meets (#823). + + Each is ``(rule, detail)``: ``detail`` names the input for an opened gate + (``allowed_non_write_users``) and is empty otherwise. Read from published + settings only, so a redacted value, a value holding an expression, an + unresolved launch and an unknown flag meet no rule. + """ + + if entry.get("form") != "read": + return set() + agent = str(entry["agent"]) + readable = { + str(setting["name"]): str(setting["value"]) + for setting in entry.get("settings", []) + if setting.get("unresolved_reason") is None and setting.get("value") is not None + } + rules: set[tuple[str, str]] = set() + if agent in {"claude", "codex"}: + flags = [(str(setting["name"]), setting.get("value")) for setting in entry.get("settings", []) + if setting.get("unresolved_reason") is None] + rules.update((rule, "") for rule in _flag_rules(agent, flags)) + return rules + spec = _AGENT_ACTIONS[agent] + for name in spec.gates: + if name in readable and "*" in {part.strip() for part in readable[name].split(",")}: + rules.add(("open_gate", name)) + for name, value in spec.modes: + if readable.get(name) == value: + rules.add(("danger_full_access" if name == "sandbox" else "unsafe_safety_strategy", "")) + if spec.args and spec.args in readable: + words = _argument_words(readable[spec.args], json_list=spec.family == "codex") + if words is not None: + flags = _read_flags(words, _AGENT_FLAG_TABLES[spec.family]) + rules.update((rule, "") for rule in _flag_rules(spec.family, flags)) + return rules + + +def agent_family(agent: str) -> str: + """``claude`` or ``codex``: which agent's rules a launch is read by.""" + + return _AGENT_ACTIONS[agent].family if agent in _AGENT_ACTIONS else agent + + +def gained_agent_widenings( + before: dict[str, Any] | None, after: dict[str, Any] | None +) -> list[tuple[str, str, str, dict[str, Any]]]: + """Documented widening rules a job's agent launches meet at ``after`` and not at ``before``. + + Keyed by job, agent family and rule, so moving a launch between steps or + spellings (``--dangerously-skip-permissions`` and + ``--permission-mode bypassPermissions`` are one rule) gains nothing. Each + result is ``(job, rule, detail, entry)`` for the first launch at ``after`` + that meets it. + """ + + def met(grant: dict[str, Any] | None) -> dict[tuple[str, str, str, str], dict[str, Any]]: + found: dict[tuple[str, str, str, str], dict[str, Any]] = {} + for entry in (grant or {}).get("agent_launches", []): + for rule, detail in sorted(agent_widening_rules(entry)): + key = (str(entry["job"]), agent_family(str(entry["agent"])), rule, detail) + found.setdefault(key, entry) + return found + + old = met(before) + return [ + (key[0], key[2], key[3], entry) for key, entry in met(after).items() if key not in old + ] + + +#: How an unresolved agent launch or checkout reads in the limit that names it. +_UNRESOLVED_AGENT_PHRASES = { + "compound_command": "part of a `run:` that holds more than one command, or quoting this audit cannot split", + "shell_expansion": "a command with a shell expansion this static audit does not evaluate", + "expression": "a `run:` holding a `${{ }}` expression, which GitHub substitutes before the shell reads it", + "inputs_not_a_mapping": "a step whose `with:` is not a mapping", +} + + +def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: + """One message per agent launch setting or checkout ref a workflow does not compare (#823). + + Not blocking, like an unread secret value (#693): the launch's job, agent, + form and reason are still compared, so adding, removing or re-forming one + is a row. Only an edit inside what is named here is not reported. + """ + + texts: list[str] = [] + for entry in grant.get("agent_launches", []): + where = f"{entry['job']}/{entry['step']} ({entry['agent']})" + reason = entry.get("unresolved_reason") + if reason: + texts.append( + f"the agent launch at {where} is " + f"{_UNRESOLVED_AGENT_PHRASES.get(str(reason), 'in an unsupported form')}; its " + "settings are neither published nor compared, so an edit to them is not reported" + ) + for setting in entry.get("settings", []): + unread = setting.get("unresolved_reason") + if unread: + what = "contains credential-shaped text" if unread == "redacted" else "is not a string" + texts.append( + f"the {setting['name']} value of the agent launch at {where} {what}; it is " + "neither published nor compared, so an edit to it is not reported" + ) + for entry in grant.get("checkout_refs", []): + unread = entry.get("unresolved_reason") + if unread: + what = { + "redacted": "a ref that contains credential-shaped text", + "not_a_string": "a ref that is not a string", + }.get(str(unread), "a `with:` that is not a mapping") + texts.append( + f"the checkout at {entry['job']}/{entry['step']} declares {what}; it is " + "neither published nor compared, so an edit to it is not reported" + ) + # Two steps whose labels publish alike name one limit once. + return list(dict.fromkeys(texts)) + + #: A whole ``secrets:`` value of this form names its source (#693). Only the #: property form is read; ``secrets['NAME']`` and every other expression stay #: unresolved rather than guessed. The ``secrets`` context name is matched as @@ -1818,10 +2425,11 @@ def _workflow_grant( Each label is published by :func:`published_workflow_label` before it is used anywhere, so the job in ``permission_contexts``, ``reusable_calls``, - ``step_actions``, the ``write_scopes`` and ``effective_write_scopes`` - prefixes, every row built from them, and ``config_sha256`` all hold the - same label, and none holds the raw text. ``collided`` receives the kinds - of label of which two distinct raw values publish alike. + ``step_actions``, ``agent_launches``, ``checkout_refs``, the + ``write_scopes`` and ``effective_write_scopes`` prefixes, every row built + from them, and ``config_sha256`` all hold the same label, and none holds + the raw text. ``collided`` receives the kinds of label of which two + distinct raw values publish alike. """ if not isinstance(data, dict): @@ -1838,6 +2446,8 @@ def _workflow_grant( permission_contexts: list[dict[str, Any]] = [] reusable_calls: list[dict[str, Any]] = [] step_actions: list[dict[str, Any]] = [] + agent_launches: list[dict[str, Any]] = [] + checkout_refs: list[dict[str, Any]] = [] def collect(perms: Any, where: str) -> None: if perms == "write-all": @@ -1876,6 +2486,7 @@ def collect(perms: Any, where: str) -> None: if not isinstance(steps, list): step_actions.append(_unreadable_step(label, "steps", "steps_not_a_list")) steps = [] + job_launches: list[dict[str, Any]] = [] for index, step in enumerate(steps): if not isinstance(step, dict): step_actions.append( @@ -1885,6 +2496,17 @@ def collect(perms: Any, where: str) -> None: action = _step_action(label, step, index) if action is not None: step_actions.append(action) + job_launches.extend(_step_agent_launches(label, step, index)) + checkout = _checkout_ref(label, step, index) + if checkout is not None: + checkout_refs.append(checkout) + if job_launches: + # Context for the note on the row, never compared (#823). + secrets = _job_secrets(job, data.get("env")) + agent_launches.extend( + {**launch, "job_secrets": secrets} if secrets else launch + for launch in job_launches + ) pull_target = "pull_request_target" in triggers write_all = any(entry.endswith(": write-all") for entry in effective_write_scopes) unknown = not permission_contexts or any( @@ -1905,6 +2527,11 @@ def collect(perms: Any, where: str) -> None: # Omitted when empty, as the schema omits it, so a workflow whose steps # declare no listed reference keeps its earlier fingerprint. projection["step_actions"] = step_actions + # Omitted when empty for the same reason (#823). + if agent_launches: + projection["agent_launches"] = agent_launches + if checkout_refs: + projection["checkout_refs"] = checkout_refs return { **_grant_base( host="github", scope="repository", source=source, kind="workflow", @@ -2063,7 +2690,12 @@ def _collect_file( _inventory_issue( kind="unsupported", host=host, source=source, message=text, blocking=False, ) - for text in uncompared_secret_mapping_texts(grant) + for text in ( + *uncompared_secret_mapping_texts(grant), + # An agent launch or checkout ref this reader does not + # compare narrows that entry, not the workflow (#823). + *uncompared_agent_launch_texts(grant), + ) ) elif host == "codex" and kind == "requirements": grants.extend(_codex_requirement_grants(data, scope=scope, source=source)) @@ -3398,7 +4030,7 @@ def note_unusable_selected_hooks(data: Any, *, source: str) -> None: "static_analysis_only": True, "runtime_session_verified": False, } - inventory = HostGrantsInventoryV6.model_validate(payload).model_dump(mode="json") + inventory = HostGrantsInventoryV7.model_validate(payload).model_dump(mode="json") return HostBoundarySnapshot( inventory=inventory, cache=cache, input_failures=dict(cache.input_failures), plugin_reference_issue_ids=frozenset(plugin_reference_issue_ids), @@ -3437,7 +4069,7 @@ def host_audit_inventory( if snapshot is None: snapshot = build_host_boundary_snapshot(workspace, scope=scope, cache=cache) - inventory = HostGrantsInventoryV6.model_validate(snapshot.inventory) + inventory = HostGrantsInventoryV7.model_validate(snapshot.inventory) if inventory.scope != scope: raise ValueError( f"Host boundary snapshot scope {inventory.scope!r} does not match {scope!r}" @@ -3553,7 +4185,7 @@ def build_host_grants_baseline(inventory: dict[str, Any]) -> dict[str, Any]: "inventory_sha256": host_grants_sha256(normalized), "inventory": normalized, } - return HostGrantsBaselineV6.model_validate(payload).model_dump(mode="json") + return HostGrantsBaselineV7.model_validate(payload).model_dump(mode="json") def load_host_grants_baseline(path: Path) -> dict[str, Any]: @@ -3597,7 +4229,7 @@ def load_host_grants_baseline_with_text( "and repair or replace it deliberately." ) return data, text - if version not in {"0.2", "0.3", "0.4", "0.5", HOST_GRANTS_BASELINE_SCHEMA_VERSION}: + if version not in {"0.2", "0.3", "0.4", "0.5", "0.6", HOST_GRANTS_BASELINE_SCHEMA_VERSION}: raise ValueError( f"Host-grants baseline {path} has unsupported schema version " f"{version!r}. A human must review migration or replacement." @@ -3605,7 +4237,7 @@ def load_host_grants_baseline_with_text( try: model = {"0.2": HostGrantsBaselineV2, "0.3": HostGrantsBaselineV3, "0.4": HostGrantsBaselineV4, "0.5": HostGrantsBaselineV5, - "0.6": HostGrantsBaselineV6}[version] + "0.6": HostGrantsBaselineV6, "0.7": HostGrantsBaselineV7}[version] parsed = model.model_validate(data).model_dump(mode="json") except ValidationError: return ( @@ -3787,15 +4419,24 @@ def _same_workflow_grant(before: dict | None, after: dict | None) -> bool: # Step references compare as each job's multiset of declared references: # a reordered, renamed or re-id'd step that declares the same reference # changes no modeled fact, and no execution-order dependency is evaluated. - ignored = {"write_scopes", "config_sha256", "step_actions"} + # Agent launches and checkout refs compare the same way (#823): as each + # job's multiset of declared facts, never by step label, and a launch's + # `job_secrets` is context for the row, never compared. + ignored = {"write_scopes", "config_sha256", "step_actions", "agent_launches", "checkout_refs"} def comparable(grant: dict[str, Any]) -> dict[str, Any]: # Both sides are v0.4+ shapes here, and a drift payload refuses a - # v0.4/v0.5 baseline holding a workflow, so a missing key is "none". + # v0.4-v0.6 baseline holding a workflow, so a missing key is "none". projection = {key: value for key, value in grant.items() if key not in ignored} projection["step_actions"] = sorted( step_action_key(item) for item in grant.get("step_actions", []) ) + projection["agent_launches"] = sorted( + agent_launch_key(item) for item in grant.get("agent_launches", []) + ) + projection["checkout_refs"] = sorted( + checkout_ref_key(item) for item in grant.get("checkout_refs", []) + ) # Named secrets compare as each call's set of facts, whatever order a # saved snapshot lists them in (#693). projection["reusable_calls"] = [ @@ -3960,6 +4601,10 @@ def inherited_calls(grant): } if inherited_calls(after) - inherited_calls(previous): signals.append(f"workflow_secrets_inherited_{prefix}: {after['source']}") + # Only a documented rule gained by a job's agent launches widens + # (#823); every other agent-launch or checkout edit is a change. + if gained_agent_widenings(before, after): + signals.append(f"workflow_agent_widened_{prefix}: {after['source']}") return sorted(set(signals)) @@ -4109,7 +4754,7 @@ def _incomparable_payload( # and also route to a human before any first acknowledgement. "next_action": None, } - return HostGrantsDriftV6.model_validate(payload).model_dump(mode="json") + return HostGrantsDriftV7.model_validate(payload).model_dump(mode="json") #: Baseline versions a drift comparison reads as current. v0.5 only adds @@ -4117,9 +4762,12 @@ def _incomparable_payload( #: v0.4 inventory refused such a link as unreadable, and an incomplete #: inventory can never be saved, so a v0.4 baseline holds no artifact that v0.5 #: would describe differently. Accepting it keeps every saved baseline usable. -#: v0.6 adds workflow step references (#771); the rule below narrows which -#: v0.4/v0.5 baselines that acceptance still covers. -_COMPARABLE_BASELINE_SCHEMA_VERSIONS = frozenset({"0.4", "0.5", HOST_GRANTS_BASELINE_SCHEMA_VERSION}) +#: v0.6 adds workflow step references (#771) and v0.7 agent launches and +#: checkout refs (#823); the rules below narrow which older baselines that +#: acceptance still covers. +_COMPARABLE_BASELINE_SCHEMA_VERSIONS = frozenset( + {"0.4", "0.5", "0.6", HOST_GRANTS_BASELINE_SCHEMA_VERSION} +) #: Baseline versions whose workflow grants never read step action references #: (#771). Such a grant's missing ``step_actions`` is not evidence that no @@ -4129,6 +4777,10 @@ def _incomparable_payload( #: claims were compared. _STEP_ACTIONS_UNREAD_BASELINE_SCHEMA_VERSIONS = frozenset({"0.4", "0.5"}) +#: Baseline versions whose workflow grants never read agent launches or +#: checkout refs (#823), by the same rule: silence is not evidence of none. +_AGENT_LAUNCHES_UNREAD_BASELINE_SCHEMA_VERSIONS = frozenset({"0.4", "0.5", "0.6"}) + def build_host_drift_payload( *, baseline: dict[str, Any], inventory: dict[str, Any], baseline_file: str @@ -4140,14 +4792,19 @@ def build_host_drift_payload( reasons.append( str(baseline.get("_load_error") or "unsupported_baseline_schema") ) - elif baseline.get("host_grants_schema_version") in _STEP_ACTIONS_UNREAD_BASELINE_SCHEMA_VERSIONS: + elif baseline.get("host_grants_schema_version") in _AGENT_LAUNCHES_UNREAD_BASELINE_SCHEMA_VERSIONS: + version = baseline.get("host_grants_schema_version") workflows = [ grant for grant in (baseline.get("inventory") or {}).get("grants", []) if grant.get("kind") == "workflow" ] - if workflows: + if workflows and version in _STEP_ACTIONS_UNREAD_BASELINE_SCHEMA_VERSIONS: reasons.append("baseline_workflow_step_actions_unavailable") - if any(grant.get("reusable_calls") for grant in workflows): + if workflows: + reasons.append("baseline_workflow_agent_launches_unavailable") + if version in _STEP_ACTIONS_UNREAD_BASELINE_SCHEMA_VERSIONS and any( + grant.get("reusable_calls") for grant in workflows + ): # Such a call's missing ``secret_mappings`` is not evidence that it # passed no named secret: those snapshots never read them (#693). reasons.append("baseline_reusable_workflow_secret_mappings_unavailable") @@ -4196,7 +4853,7 @@ def _comparable_drift_payload( "incomparable_reasons": [], "next_action": None, } - return HostGrantsDriftV6.model_validate(payload).model_dump(mode="json") + return HostGrantsDriftV7.model_validate(payload).model_dump(mode="json") def build_host_comparison_payload( diff --git a/src/agents_shipgate/schemas/contract.py b/src/agents_shipgate/schemas/contract.py index 7c3ab79b0..081668c13 100644 --- a/src/agents_shipgate/schemas/contract.py +++ b/src/agents_shipgate/schemas/contract.py @@ -214,19 +214,30 @@ # ``host_comparison.coverage`` and ``shipgate diff --json`` (capability diff # 0.3) the same block. It is evidence beside the rows and moves no state, # permission, route or row. A 0.19 verifier reads with coverage not recorded. -# v41 names the changed inputs a host comparison does not read (#821): verifier -# 0.21 and ``shipgate diff --json`` (capability diff 0.4) add a -# ``changed_not_read`` coverage item with its ``candidate`` rule, a -# ``read_sources_only`` that is ``false`` while one is named, and whether the -# comparison's changed files were examined. A name is never a row, a widening -# or a ``check`` violation. The one route it moves, on ``verify`` and -# ``verify --preview`` alike: a manifest-free comparison whose only -# host-relevant change is such an input, or a changed candidate it counts as -# not examined, is now published, on the host route's -# ``audit --host`` next action, instead of the setup route (``verify``) or the -# ``initialize`` next action (``verify --preview``). A 0.20 verifier reads -# with the search not recorded. ``MINIMUM_CONTROL_CONTRACT_VERSION`` stays at -# 21. +# v41, unreleased, carries two changes. It names the changed inputs a host +# comparison does not read (#821): verifier 0.21 and ``shipgate diff --json`` +# (capability diff 0.4) add a ``changed_not_read`` coverage item with its +# ``candidate`` rule, a ``read_sources_only`` that is ``false`` while one is +# named, and whether the comparison's changed files were examined. A name is +# never a row, a widening or a ``check`` violation. The one route it moves, on +# ``verify`` and ``verify --preview`` alike: a manifest-free comparison whose +# only host-relevant change is such an input, or a changed candidate it counts +# as not examined, is now published, on the host route's ``audit --host`` next +# action, instead of the setup route (``verify``) or the ``initialize`` next +# action (``verify --preview``). A 0.20 verifier reads with the search not +# recorded. And it reads how a coding agent is launched inside a workflow job +# (#823). Host-grants 0.6 shipped in 1.1.0, so this mints host-grants +# inventory, baseline and drift 0.7: a workflow grant adds ``agent_launches[]`` +# (a documented agent action's permission inputs, or the permission flags of a +# literal ``claude -p`` / ``codex exec`` run step, compared as text) and +# ``checkout_refs[]`` (each ``actions/checkout`` step's ``with.ref``), both +# omitted when empty. Only a documented rule gained by a job's agent launches +# widens; every other edit is a ``changed`` row, and a workflow row that runs +# an agent names the job facts beside it. A 0.4-0.6 baseline holding a +# workflow grant is incomparable +# (``baseline_workflow_agent_launches_unavailable``); one without a workflow +# stays comparable. #823 moves neither verifier 0.21 nor capability diff 0.4: +# its rows keep their shape. ``MINIMUM_CONTROL_CONTRACT_VERSION`` stays at 21. # v41 also keeps what a comparison established outside a plugin directory it # could not compare (#808), extended in place because v41 is unreleased: where # every blocking limit is a plugin-reference limit that its plugin directory diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index ed08f8f10..f23c41481 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -6,9 +6,9 @@ from agents_shipgate.schemas.instruction_structure import InstructionStructureEvidence -HOST_GRANTS_INVENTORY_SCHEMA_VERSION = "0.6" -HOST_GRANTS_BASELINE_SCHEMA_VERSION = "0.6" -HOST_GRANTS_DRIFT_SCHEMA_VERSION = "0.6" +HOST_GRANTS_INVENTORY_SCHEMA_VERSION = "0.7" +HOST_GRANTS_BASELINE_SCHEMA_VERSION = "0.7" +HOST_GRANTS_DRIFT_SCHEMA_VERSION = "0.7" HostName = Literal["codex", "claude-code", "cursor", "vscode", "github"] HostGrantScope = Literal["repository", "local_static"] @@ -618,4 +618,147 @@ class HostGrantsDriftArtifactV6(RootModel[HostGrantsDriftV6]): root: HostGrantsDriftV6 -__all__ = [name for name in globals() if name.startswith("Host") or name.startswith("HOST_")] +# v0.7 reads how a coding agent is launched inside a job (#823). The v0.6 +# workflow grant above stays frozen: a v0.4-v0.6 snapshot never read agent +# launches or checkout refs, so its silence cannot assert that none changed. +class HostWorkflowAgentSettingV7(BaseModel): + """One permission input or flag an agent launch declares, compared as text (#823). + + ``name`` is the documented input (``claude_args``, ``sandbox``, …) or the + flag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too). + ``value`` is the declared text, stripped, published through the workflow + label redaction (#802); a flag that takes no value has ``null``. A value + the redaction rewrites, or one that is not a string, is ``null`` with + ``unresolved_reason``, and records a non-blocking coverage issue naming + its ``job/step``: it is neither published nor compared. + """ + + model_config = ConfigDict(extra="forbid") + + name: str + value: str | None + unresolved_reason: Literal["not_a_string", "redacted"] | None = None + + +class HostWorkflowAgentLaunchV7(BaseModel): + """A step that launches a known coding agent, read as text and never run (#823). + + ``agent`` is a documented action reference's ``owner/repo`` (the step's + ``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with: + ``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read`` + lists the documented permission inputs or flags the step declares in + ``settings``. ``form: unresolved`` names why the step's settings were not + read — a ``run:`` holding more than one command or quoting that does not + balance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is + not a mapping — with no + settings, and records a non-blocking coverage issue. ``job_secrets`` names + the secrets the step's job references (``${{ secrets.NAME }}``) and the + workflow-level ``env`` passes: context for the row that names this step, + never compared. ``job`` and ``step`` are published labels (#802). + """ + + model_config = ConfigDict(extra="forbid") + + job: str + step: str + agent: Literal[ + "anthropics/claude-code-action", + "anthropics/claude-code-base-action", + "openai/codex-action", + "claude", + "codex", + ] + form: Literal["read", "unresolved"] + unresolved_reason: Literal[ + "compound_command", + "shell_expansion", + "expression", + "inputs_not_a_mapping", + ] | None = None + settings: list[HostWorkflowAgentSettingV7] = Field(default_factory=list) + job_secrets: list[str] = Field(default_factory=list, exclude_if=lambda value: not value) + + +class HostWorkflowCheckoutRefV7(BaseModel): + """One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823). + + ``ref`` is ``null`` when the step declares none, or an empty one: the + checkout's default for the triggering event. A ref the label redaction + rewrites, a value that is not a string, or ``with:`` that is not a mapping + is ``null`` with ``unresolved_reason`` and records a non-blocking coverage + issue. The ref is never resolved or fetched. + """ + + model_config = ConfigDict(extra="forbid") + + job: str + step: str + ref: str | None + unresolved_reason: Literal["not_a_string", "redacted", "inputs_not_a_mapping"] | None = None + + +class HostWorkflowGrantV7(HostWorkflowGrantV6): + """A v0.6 workflow grant plus the agent launches and checkout refs its steps declare. + + Both lists are present only when a step declares one. In a v0.7 grant an + absent list means the steps were read and declare none; the schema + version, not the key, separates that from a legacy grant that never read + them. ``access`` and ``risk`` still describe the workflow's token and + triggers alone. + """ + + agent_launches: list[HostWorkflowAgentLaunchV7] = Field( + default_factory=list, exclude_if=lambda value: not value, + ) + checkout_refs: list[HostWorkflowCheckoutRefV7] = Field( + default_factory=list, exclude_if=lambda value: not value, + ) + + +HostGrantV7 = Annotated[ + HostMcpServerGrantV2 + | HostPermissionRuleGrantV2 + | HostPermissionModeGrantV2 + | HostHookGrantV2 + | HostSandboxGrantV2 + | HostAdditionalPathGrantV2 + | HostPluginGrantV2 + | HostProfileGrantV2 + | HostRequirementGrantV2 + | HostWorkflowGrantV7 + | HostInstructionGrantV2, + Field(discriminator="kind"), +] + + +class HostGrantsInventoryV7(HostGrantsInventoryV6): + host_grants_inventory_schema_version: Literal["0.7"] = "0.7" + grants: list[HostGrantV7] = Field(default_factory=list) + + +class HostGrantsNormalizedSnapshotV7(HostGrantsNormalizedSnapshotV6): + grants: list[HostGrantV7] = Field(default_factory=list) + + +class HostGrantsBaselineV7(HostGrantsBaselineV6): + host_grants_schema_version: Literal["0.7"] = "0.7" + inventory: HostGrantsNormalizedSnapshotV7 + + +class HostGrantsDriftV7(HostGrantsDriftV6): + host_grants_schema_version: Literal["0.7"] = "0.7" + + +class HostGrantsInventoryArtifactV7(RootModel[HostGrantsInventoryV7]): + root: HostGrantsInventoryV7 + + +class HostGrantsBaselineArtifactV7(RootModel[HostGrantsBaselineV7]): + root: HostGrantsBaselineV7 + + +class HostGrantsDriftArtifactV7(RootModel[HostGrantsDriftV7]): + root: HostGrantsDriftV7 + + +__all__ =[name for name in globals() if name.startswith("Host") or name.startswith("HOST_")] diff --git a/tests/test_agent_instructions_apply.py b/tests/test_agent_instructions_apply.py index 822625edf..867aa636b 100644 --- a/tests/test_agent_instructions_apply.py +++ b/tests/test_agent_instructions_apply.py @@ -205,9 +205,9 @@ def test_local_contract_renderer_has_required_fields() -> None: assert payload["registry_schema_version"] == "0.4" assert payload["org_evidence_bundle_schema_version"] == ("shipgate.org_evidence_bundle/v2") assert payload["agent_boundary_result_schema_version"] == ("shipgate.agent_boundary_result/v3") - assert payload["host_grants_inventory_schema_version"] == "0.6" - assert payload["host_grants_baseline_schema_version"] == "0.6" - assert payload["host_grants_drift_schema_version"] == "0.6" + assert payload["host_grants_inventory_schema_version"] == "0.7" + assert payload["host_grants_baseline_schema_version"] == "0.7" + assert payload["host_grants_drift_schema_version"] == "0.7" assert payload["trigger_catalog_schema_version"] == "0.4" assert payload["gating_signal"] == "release_decision.decision" assert payload["default_paths"]["local_contract"] == ".shipgate/agent-contract.json" diff --git a/tests/test_agent_instructions_renderers.py b/tests/test_agent_instructions_renderers.py index 8ce0e4c87..58120718b 100644 --- a/tests/test_agent_instructions_renderers.py +++ b/tests/test_agent_instructions_renderers.py @@ -220,9 +220,9 @@ def test_local_contract_renderer_exposes_agent_operational_fields() -> None: assert payload["attestation_schema_version"] == "0.5" assert payload["registry_schema_version"] == "0.4" assert payload["org_evidence_bundle_schema_version"] == ("shipgate.org_evidence_bundle/v2") - assert payload["host_grants_inventory_schema_version"] == "0.6" - assert payload["host_grants_baseline_schema_version"] == "0.6" - assert payload["host_grants_drift_schema_version"] == "0.6" + assert payload["host_grants_inventory_schema_version"] == "0.7" + assert payload["host_grants_baseline_schema_version"] == "0.7" + assert payload["host_grants_drift_schema_version"] == "0.7" assert payload["trigger_catalog_schema_version"] == "0.4" assert payload["agent_result_control_fields"] == [ "decision", diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 9056ebbff..a78a02f51 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -227,6 +227,13 @@ def paths(self) -> list[Path]: # control route, so it adds no claim; every route to the same object, # and every refusal it must keep, is held by # `tests/test_partial_host_comparison.py`. + # Its agent-launch cells and note (#823) + # restate no answer: direction comes from the engine's + # `workflow_agent_widened_*` expansion signal, and the note reads the + # triggers, write scopes, secrets and checkout refs the engine already + # published on the grant, so it adds no claim; + # `tests/test_workflow_agent_launches.py` holds diff, verify, the PR + # comment, check and the control envelope to the same row. {}, ), Surface( diff --git a/tests/test_host_audit.py b/tests/test_host_audit.py index aab339289..0ffa2004c 100644 --- a/tests/test_host_audit.py +++ b/tests/test_host_audit.py @@ -30,10 +30,10 @@ load_host_grants_baseline, ) from agents_shipgate.schemas.host_grants import ( - HostGrantsBaselineV6, - HostGrantsDriftV6, + HostGrantsBaselineV7, + HostGrantsDriftV7, HostGrantsInventoryArtifactV4, - HostGrantsInventoryV6, + HostGrantsInventoryV7, ) runner = CliRunner() @@ -170,8 +170,8 @@ def _drift_json(tmp_path: Path, *extra: str) -> tuple[int, dict]: def test_inventory_v02_collects_typed_multi_host_grants(tmp_path: Path) -> None: inventory = host_audit_inventory(_seed_workspace(tmp_path)) - assert inventory["host_grants_inventory_schema_version"] == "0.6" - HostGrantsInventoryV6.model_validate(inventory) + assert inventory["host_grants_inventory_schema_version"] == "0.7" + HostGrantsInventoryV7.model_validate(inventory) assert inventory["scope"] == "repository" assert inventory["static_analysis_only"] is True assert inventory["runtime_session_verified"] is False @@ -749,8 +749,8 @@ def test_v02_baseline_is_typed_portable_redacted_and_idempotent(tmp_path: Path) _seed_workspace(tmp_path) baseline_path = _save_baseline(tmp_path) payload = json.loads(baseline_path.read_text(encoding="utf-8")) - HostGrantsBaselineV6.model_validate(payload) - assert payload["host_grants_schema_version"] == "0.6" + HostGrantsBaselineV7.model_validate(payload) + assert payload["host_grants_schema_version"] == "0.7" assert payload["scope"] == "repository" assert "workspace" not in payload["inventory"] assert payload["inventory"]["artifacts"] @@ -956,7 +956,7 @@ def test_clean_and_changed_v02_drift(tmp_path: Path) -> None: _save_baseline(tmp_path) code, clean = _drift_json(tmp_path) assert code == 0 - HostGrantsDriftV6.model_validate(clean) + HostGrantsDriftV7.model_validate(clean) assert clean["comparison_status"] == "comparable" assert clean["has_drift"] is False assert clean["baseline_sha256"] == clean["current_sha256"] @@ -1213,14 +1213,14 @@ def test_legacy_v01_baseline_is_incomparable_advisory_and_strict_20(tmp_path: Pa inventory=host_audit_inventory(tmp_path), baseline_file=".agents-shipgate/host-grants.json", ) - HostGrantsDriftV6.model_validate(shared) + HostGrantsDriftV7.model_validate(shared) assert shared["comparison_status"] == "incomparable" assert shared["next_action"] is None assert "--save-baseline" not in json.dumps(shared) code, payload = _drift_json(tmp_path) assert code == 0 - HostGrantsDriftV6.model_validate(payload) + HostGrantsDriftV7.model_validate(payload) assert payload["comparison_status"] == "incomparable" assert payload["has_drift"] is None assert "baseline_schema_v0.1" in payload["incomparable_reasons"][0] @@ -1272,7 +1272,7 @@ def test_malformed_nested_v02_baseline_is_incomparable_not_a_crash(tmp_path: Pat ) code, payload = _drift_json(tmp_path) assert code == 0 - HostGrantsDriftV6.model_validate(payload) + HostGrantsDriftV7.model_validate(payload) assert payload["comparison_status"] == "incomparable" assert payload["has_drift"] is None assert payload["incomparable_reasons"] == ["malformed_v0.2_baseline"] @@ -1629,9 +1629,9 @@ def denied_read_text( def test_generated_models_reject_unknown_fields_and_invalid_literals(tmp_path: Path) -> None: payload = host_audit_inventory(tmp_path) with pytest.raises(ValidationError): - HostGrantsInventoryV6.model_validate({**payload, "legacy_parse_warnings": []}) + HostGrantsInventoryV7.model_validate({**payload, "legacy_parse_warnings": []}) with pytest.raises(ValidationError): - HostGrantsInventoryV6.model_validate({**payload, "scope": "runtime"}) + HostGrantsInventoryV7.model_validate({**payload, "scope": "runtime"}) def test_inventory_schema_uses_discriminated_typed_grants() -> None: diff --git a/tests/test_host_input_recovery.py b/tests/test_host_input_recovery.py index 23277410a..9a287f30a 100644 --- a/tests/test_host_input_recovery.py +++ b/tests/test_host_input_recovery.py @@ -303,6 +303,6 @@ def fail(self): snapshot = build_host_boundary_snapshot(tmp_path) for name, payload in ( ("agent-boundary-result-schema.v3.json", _result(tmp_path, snapshot)), - ("host-grants-inventory-schema.v0.6.json", snapshot.inventory), + ("host-grants-inventory-schema.v0.7.json", snapshot.inventory), ): jsonschema.validate(payload, json.loads((Path("docs") / name).read_text())) diff --git a/tests/test_instruction_structure_contracts.py b/tests/test_instruction_structure_contracts.py index 6fbfd2070..89e1fb93f 100644 --- a/tests/test_instruction_structure_contracts.py +++ b/tests/test_instruction_structure_contracts.py @@ -49,7 +49,7 @@ def test_old_models_reject_structural_claims_and_old_baseline_is_not_restamped(r assert drift["has_drift"] is None assert "baseline_instruction_structure_unavailable" in drift["incomparable_reasons"] assert path.read_bytes() == captured - schema = json.loads((ROOT / "docs/host-grants-inventory-schema.v0.6.json").read_text()) + schema = json.loads((ROOT / "docs/host-grants-inventory-schema.v0.7.json").read_text()) Draft202012Validator(schema).validate(inventory) diff --git a/tests/test_local_contract.py b/tests/test_local_contract.py index 7724377f2..d2eb5b004 100644 --- a/tests/test_local_contract.py +++ b/tests/test_local_contract.py @@ -156,9 +156,9 @@ def test_local_agent_contract_is_minimal_agent_operational_payload() -> None: assert payload["attestation_schema_version"] == "0.5" assert payload["registry_schema_version"] == "0.4" assert payload["org_evidence_bundle_schema_version"] == ("shipgate.org_evidence_bundle/v2") - assert payload["host_grants_inventory_schema_version"] == "0.6" - assert payload["host_grants_baseline_schema_version"] == "0.6" - assert payload["host_grants_drift_schema_version"] == "0.6" + assert payload["host_grants_inventory_schema_version"] == "0.7" + assert payload["host_grants_baseline_schema_version"] == "0.7" + assert payload["host_grants_drift_schema_version"] == "0.7" assert payload["trigger_catalog_schema_version"] == "0.4" assert payload["agent_result_schema_version"] == "agent_result_v3" assert payload["agent_result_schema_path"] == "docs/agent-result-schema.v3.json" diff --git a/tests/test_org_governance.py b/tests/test_org_governance.py index 302a78e70..c2eb57993 100644 --- a/tests/test_org_governance.py +++ b/tests/test_org_governance.py @@ -575,7 +575,7 @@ def test_org_bundle_projects_platform_artifacts_without_second_gate( assert payload["registry_row"]["source_attestation_sha256"] == attestation_sha256 assert payload["org_status"]["summary"]["policy_pack_count"] == 1 assert payload["policy_packs"][0]["status"] == "verified" - assert payload["host_grants"]["host_grants_inventory_schema_version"] == "0.6" + assert payload["host_grants"]["host_grants_inventory_schema_version"] == "0.7" assert payload["artifacts"]["verifier"]["sha256"] diff --git a/tests/test_reusable_workflow_secret_mappings.py b/tests/test_reusable_workflow_secret_mappings.py index 14da55b4d..5ecfeef0b 100644 --- a/tests/test_reusable_workflow_secret_mappings.py +++ b/tests/test_reusable_workflow_secret_mappings.py @@ -786,6 +786,7 @@ def test_a_legacy_baseline_holding_a_reusable_call_does_not_assert_no_mappings(t assert drift["comparison_status"] == "incomparable" assert drift["incomparable_reasons"] == [ "baseline_reusable_workflow_secret_mappings_unavailable", + "baseline_workflow_agent_launches_unavailable", "baseline_workflow_step_actions_unavailable", ] assert drift["has_drift"] is None and drift["changes"] == [] @@ -806,9 +807,9 @@ def test_a_current_baseline_compares_mappings_and_validates_against_the_schemas( path.write_text(STAGING) inventory = host_audit_inventory(tmp_path) baseline = build_host_grants_baseline(inventory) - assert baseline["host_grants_schema_version"] == "0.6" - Draft202012Validator(json.loads((ROOT / "docs/host-grants-inventory-schema.v0.6.json").read_text())).validate(inventory) - Draft202012Validator(json.loads((ROOT / "docs/host-grants-baseline-schema.v0.6.json").read_text())).validate(baseline) + assert baseline["host_grants_schema_version"] == "0.7" + Draft202012Validator(json.loads((ROOT / "docs/host-grants-inventory-schema.v0.7.json").read_text())).validate(inventory) + Draft202012Validator(json.loads((ROOT / "docs/host-grants-baseline-schema.v0.7.json").read_text())).validate(baseline) unchanged = build_host_drift_payload(baseline=baseline, inventory=inventory, baseline_file="b.json") assert (unchanged["comparison_status"], unchanged["has_drift"]) == ("comparable", False) @@ -817,7 +818,7 @@ def test_a_current_baseline_compares_mappings_and_validates_against_the_schemas( drift = build_host_drift_payload(baseline=baseline, inventory=host_audit_inventory(tmp_path), baseline_file="b.json") assert drift["comparison_status"] == "comparable" and drift["has_drift"] is True assert len(drift["changes"]) == 1 and drift["expansion_signals"] == [] - Draft202012Validator(json.loads((ROOT / "docs/host-grants-drift-schema.v0.6.json").read_text())).validate(drift) + Draft202012Validator(json.loads((ROOT / "docs/host-grants-drift-schema.v0.7.json").read_text())).validate(drift) def test_a_saved_baseline_listing_mappings_out_of_order_still_compares_equal(tmp_path): diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py new file mode 100644 index 000000000..e77ae0a5e --- /dev/null +++ b/tests/test_workflow_agent_launches.py @@ -0,0 +1,838 @@ +"""#823: how a coding agent is launched inside a workflow job is read as text. + +The workflow grant already read triggers, token permissions, reusable calls and +step `uses:` references (#771), and nothing that says how an agent is started. +It now lists each documented agent action's permission inputs, the permission +flags of a literal `claude -p` / `codex exec` run step, and each +`actions/checkout` step's `with.ref`. Nothing is executed, fetched or +evaluated. Only a documented rule a job's launches gain widens; every other +edit is `changed`; a shape this reader does not read is a named limit, never a +row that claims an effect; and a workflow row that runs an agent names the job +facts beside it. +""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +import pytest +import yaml +from typer.testing import CliRunner + +from agents_shipgate.cli.main import app +from agents_shipgate.core.capability_diff_rows import capability_diff_rows +from agents_shipgate.core.host_grants import ( + _workflow_grant, + diff_host_grants, + host_grant_expansion_signals, + uncompared_agent_launch_texts, +) + +ROOT = Path(__file__).resolve().parents[1] +SOURCE = ".github/workflows/agent.yml" +CLAUDE = "anthropics/claude-code-action@v1" +HEAD_SHA = "${{ github.event.pull_request.head.sha }}" + + +def _workflow(*steps, trigger="pull_request", permissions=None, jobs=None, env=None): + data = { + "on": trigger, + "permissions": permissions if permissions is not None else {"contents": "read", "pull-requests": "read"}, + "jobs": jobs if jobs is not None else {"review": {"runs-on": "ubuntu-latest", "steps": list(steps)}}, + } + if env is not None: + data["env"] = env + return data + + +def _agent(claude_args='--allowedTools "Read"', **extra): + return {"uses": CLAUDE, "with": {"claude_args": claude_args, **extra}} + + +def _reproduction(trigger="pull_request", pr="read", claude_args='--allowedTools "Read"', run="echo done", ref=None): + """The workflow of #823's reproduction, as its `wf` shell function writes it.""" + + checkout = {"uses": "actions/checkout@v4", **({"with": {"ref": ref}} if ref else {})} + return _workflow( + checkout, _agent(claude_args), {"run": run}, + trigger=trigger, permissions={"contents": "read", "pull-requests": pr}, + ) + + +def _grant(value): + return _workflow_grant(value, source=SOURCE) + + +def _changes(before, after): + return diff_host_grants({"grants": [_grant(before)]}, {"grants": [_grant(after)]}) + + +def _rows(before, after): + changes = _changes(before, after) + payload = {"changes": changes, "expansion_signals": host_grant_expansion_signals(changes)} + return capability_diff_rows(payload) + + +def _launches(value): + return _grant(value).get("agent_launches", []) + + +# --- the four cases of the reproduction -------------------------------------------- + + +def test_args_gaining_bypass_permissions_is_one_widened_row_naming_job_step_and_both_values(): + row, = _rows( + _reproduction(), + _reproduction(claude_args='--permission-mode bypassPermissions --allowedTools "Bash(*)"'), + ) + + assert row.subject == f"github {SOURCE}" + assert (row.direction, row.expands) == ("widened", True) + assert 'review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools "Read"' in row.before + assert ( + 'review/steps[1]: runs anthropics/claude-code-action with claude_args: ' + '--permission-mode bypassPermissions --allowedTools "Bash(*)"' + ) in row.after + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[1])" in row.why + + +def test_a_literal_claude_run_step_is_one_changed_row_with_its_permission_flags(): + row, = _rows( + _reproduction(), + _reproduction(run='claude -p --permission-mode acceptEdits --allowedTools "Bash(*)" "Summarize this change"'), + ) + + assert (row.direction, row.expands) == ("changed", False) + assert "review/steps[2]" not in row.before + # A variadic flag reads every following word up to the next flag, as the + # CLI reads it, so the trailing quoted word is part of --allowedTools. + assert ( + "review/steps[2]: runs claude -p with --allowedTools 'Bash(*)' 'Summarize this change'; " + "--permission-mode acceptEdits" + ) in row.after + assert "a step now launches an agent (review/steps[2])" in row.why + assert "not counted as a widening" in row.why + + +def test_a_head_ref_checkout_is_one_changed_row_naming_the_default_and_the_new_ref(): + row, = _rows( + _reproduction(trigger="pull_request_target"), + _reproduction(trigger="pull_request_target", ref=HEAD_SHA), + ) + + assert (row.direction, row.expands) == ("changed", False) + assert "review/steps[0]: checkout of the default ref" in row.before + assert f"review/steps[0]: checkout of ref {HEAD_SHA}" in row.after + assert "a checkout's declared ref changed (review/steps[0])" in row.why + assert ( + "an agent runs at review/steps[1] (anthropics/claude-code-action) beside the " + "untrusted-input trigger pull_request_target and a checkout of pull request code (review/steps[0])" + ) in row.why + + +def test_an_untrusted_trigger_with_a_write_scope_names_the_agent_step_it_now_reaches(): + row, = _rows(_reproduction(), _reproduction(trigger="issue_comment", pr="write")) + + assert (row.direction, row.expands) == ("widened", True) + assert row.why.startswith("grants write permissions to workflow jobs") + assert ( + "an agent runs at review/steps[1] (anthropics/claude-code-action) beside the " + "untrusted-input trigger issue_comment and the write scope pull-requests" + ) in row.why + + +# --- what is read ------------------------------------------------------------------ + + +def test_only_the_documented_inputs_of_a_known_action_are_listed(): + launch, = _launches(_workflow({ + "uses": "Anthropics/Claude-Code-Action@0123456789abcdef0123456789abcdef01234567", + "with": { + "prompt": "Review this", + "anthropic_api_key": "${{ secrets.ANTHROPIC_API_KEY }}", + "claude_args": "--max-turns 5", + "Allowed_Non_Write_Users": "octocat", + "use_sticky_comment": True, + }, + })) + + assert launch["agent"] == "anthropics/claude-code-action" + assert (launch["job"], launch["step"], launch["form"]) == ("review", "steps[0]", "read") + assert launch["settings"] == [ + {"name": "allowed_non_write_users", "value": "octocat", "unresolved_reason": None}, + {"name": "claude_args", "value": "--max-turns 5", "unresolved_reason": None}, + ] + assert launch["job_secrets"] == ["ANTHROPIC_API_KEY"] + + +def test_the_codex_action_inputs_are_listed(): + launch, = _launches(_workflow({ + "uses": "openai/codex-action@v1", + "with": {"sandbox": "workspace-write", "safety-strategy": "drop-sudo", "prompt": "x", "allow-bots": False}, + })) + + assert launch["agent"] == "openai/codex-action" + assert launch["settings"] == [ + {"name": "allow-bots", "value": "false", "unresolved_reason": None}, + {"name": "safety-strategy", "value": "drop-sudo", "unresolved_reason": None}, + {"name": "sandbox", "value": "workspace-write", "unresolved_reason": None}, + ] + + +def test_an_action_step_with_no_inputs_is_still_an_agent_launch(): + launch, = _launches(_workflow({"uses": "anthropics/claude-code-base-action@beta"})) + assert (launch["agent"], launch["form"], launch["settings"]) == ("anthropics/claude-code-base-action", "read", []) + + +def test_a_literal_claude_command_publishes_its_permission_flags_under_their_primary_spelling(): + launch, = _launches(_workflow({ + "name": "Review", + "run": ( + "claude --print --allowed-tools=Read --disallowedTools 'Bash(rm *)' " + "--model sonnet --dangerously-skip-permissions --add-dir ../docs \"secret prompt text\"" + ), + })) + + assert (launch["agent"], launch["step"], launch["form"]) == ("claude", "Review", "read") + assert launch["settings"] == [ + {"name": "--add-dir", "value": "../docs 'secret prompt text'", "unresolved_reason": None}, + {"name": "--allowedTools", "value": "Read", "unresolved_reason": None}, + {"name": "--dangerously-skip-permissions", "value": None, "unresolved_reason": None}, + {"name": "--disallowedTools", "value": "'Bash(rm *)'", "unresolved_reason": None}, + ] + assert "sonnet" not in json.dumps(launch) + + +def test_a_literal_codex_exec_command_publishes_its_permission_flags(): + launch, = _launches(_workflow({"run": "codex e -s danger-full-access --yolo -c model=o3 'fix it'"})) + + assert (launch["agent"], launch["form"]) == ("codex", "read") + assert launch["settings"] == [ + {"name": "--config", "value": "model=o3", "unresolved_reason": None}, + {"name": "--dangerously-bypass-approvals-and-sandbox", "value": None, "unresolved_reason": None}, + {"name": "--sandbox", "value": "danger-full-access", "unresolved_reason": None}, + ] + + +def test_literal_assignments_before_the_command_are_skipped_and_never_published(): + launch, = _launches(_workflow({"run": "CI=true claude -p --permission-mode plan 'go'"})) + assert launch["form"] == "read" + assert launch["settings"] == [{"name": "--permission-mode", "value": "plan", "unresolved_reason": None}] + + +@pytest.mark.parametrize( + "run", + [ + "claude mcp add github -- npx server", + "claude --version", + "codex login --api-key sk-test", + 'echo "claude -p --dangerously-skip-permissions"', + "echo claude -p done", + "npm test", + ], + ids=["claude-mcp", "claude-version", "codex-login", "echo-quoted", "echo-bare", "unrelated"], +) +def test_a_command_that_launches_no_headless_agent_is_not_listed(run): + assert _launches(_workflow({"run": run})) == [] + + +@pytest.mark.parametrize( + ("run", "reason"), + [ + ("npm ci && claude -p --dangerously-skip-permissions 'go'", "compound_command"), + ("npm ci\nclaude -p 'go'", "compound_command"), + ("claude -p 'go' | tee review.md", "compound_command"), + ("cat < prompt.md\nIt's broken\nEOF\nclaude -p --dangerously-skip-permissions 'go'", "compound_command"), + ("claude -p --dangerously-skip-permissions \"go", "compound_command"), + ], + ids=["and", "lines", "pipe", "heredoc", "variable", "substitution", "expression", "unbalanced-heredoc", + "unbalanced"], +) +def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(run, reason): + launch, = _launches(_workflow({"run": run})) + + assert (launch["agent"], launch["form"], launch["unresolved_reason"]) == ("claude", "unresolved", reason) + assert launch["settings"] == [] + assert "dangerously" not in json.dumps(launch) and "go" not in json.dumps(launch["settings"]) + limit, = uncompared_agent_launch_texts(_grant(_workflow({"run": run}))) + assert limit.startswith("the agent launch at review/steps[0] (claude) is ") + assert "not reported" in limit + + +def test_single_quoted_dollars_are_literal_and_do_not_stop_the_read(): + launch, = _launches(_workflow({"run": "claude -p --allowedTools 'Bash(echo $HOME)' 'go'"})) + assert launch["form"] == "read" + assert launch["settings"][0]["value"] == "'Bash(echo $HOME)' go" + + +def test_inputs_that_are_not_a_mapping_are_unresolved(): + launch, = _launches(_workflow({"uses": CLAUDE, "with": ["claude_args"]})) + assert (launch["form"], launch["unresolved_reason"], launch["settings"]) == ( + "unresolved", "inputs_not_a_mapping", [], + ) + + +def test_every_checkout_step_records_its_declared_ref(): + grant = _grant(_workflow( + {"uses": "actions/checkout@v4"}, + {"id": "head", "uses": "actions/checkout@v4", "with": {"ref": HEAD_SHA, "fetch-depth": 0}}, + {"uses": "actions/checkout@v4", "with": {"ref": ""}}, + {"uses": "actions/checkout@v4", "with": {"ref": ["main"]}}, + {"uses": "actions/setup-node@v4", "with": {"ref": "main"}}, + )) + + assert grant["checkout_refs"] == [ + {"job": "review", "step": "steps[0]", "ref": None, "unresolved_reason": None}, + {"job": "review", "step": "head", "ref": HEAD_SHA, "unresolved_reason": None}, + {"job": "review", "step": "steps[2]", "ref": None, "unresolved_reason": None}, + {"job": "review", "step": "steps[3]", "ref": None, "unresolved_reason": "not_a_string"}, + ] + + +@pytest.mark.parametrize( + ("ref", "pull_request_code"), + [ + (HEAD_SHA, True), + ("${{github.event.pull_request.head.ref}}", True), + ("${{ github.event.pull_request.merge_commit_sha }}", True), + ("${{ github.head_ref }}", True), + ("${{ github.event.workflow_run.head_sha }}", True), + ("refs/pull/${{ github.event.issue.number }}/head", True), + ("refs/pull/123/merge", True), + (None, False), + ("main", False), + ("${{ github.sha }}", False), + ("${{ github.event.pull_request.base.sha }}", False), + ("${{ steps.pr.outputs.sha }}", False), + ], +) +def test_pull_request_code_is_the_documented_head_refs_only(ref, pull_request_code): + from agents_shipgate.core.host_grants import pull_request_code_ref + + assert pull_request_code_ref(ref) is pull_request_code + + +def test_a_workflow_without_agents_or_checkouts_keeps_its_v0_6_shape(): + grant = _grant(_workflow({"run": "make test"}, {"uses": "actions/setup-python@v5"})) + assert "agent_launches" not in grant and "checkout_refs" not in grant + + +def test_job_secrets_name_what_the_agent_job_and_the_workflow_env_reference(): + workflows = _workflow( + jobs={ + "review": {"env": {"GH": "${{ secrets.REVIEW_TOKEN }}"}, "steps": [ + {"run": "echo ${{ secrets.DEPLOY_KEY }}"}, + _agent(anthropic_api_key="${{ secrets.ANTHROPIC_API_KEY }}"), + ]}, + "other": {"steps": [{"run": "echo ${{ secrets.OTHER_JOB_ONLY }}"}]}, + }, + env={"SHARED": "${{ secrets.WORKFLOW_ENV }}"}, + ) + launch, = _launches(workflows) + assert launch["job_secrets"] == ["ANTHROPIC_API_KEY", "DEPLOY_KEY", "REVIEW_TOKEN", "WORKFLOW_ENV"] + + +# --- what is compared -------------------------------------------------------------- + + +@pytest.mark.parametrize( + "after", + [ + # the prompt and an undocumented flag are not compared + _workflow({"run": "claude -p 'a different prompt' --model opus --allowedTools Read"}), + # renamed, respelled and reordered flags + _workflow({"name": "Renamed", "run": "claude --allowed-tools=Read --print 'review'"}), + ], + ids=["prompt-and-model", "rename-and-reorder"], +) +def test_an_edit_outside_the_compared_flags_is_quiet(after): + before = _workflow({"run": "claude -p 'review' --allowedTools Read"}) + assert _rows(before, after) == [] + + +def test_a_word_after_a_variadic_flag_is_read_as_its_value_as_the_cli_reads_it(): + before = _workflow({"run": "claude -p 'review' --allowedTools Read"}) + after = _workflow({"run": "claude -p --allowedTools Read 'review'"}) + + row, = _rows(before, after) + assert "--allowedTools Read review" in row.after and "--allowedTools Read," in row.before + "," + + +def test_an_edit_inside_an_unresolved_command_is_quiet_and_named_as_a_limit(): + before = _workflow({"run": "npm ci && claude -p --allowedTools Read 'go'"}) + after = _workflow({"run": "npm ci && claude -p --dangerously-skip-permissions 'go'"}) + + assert _rows(before, after) == [] + assert uncompared_agent_launch_texts(_grant(after)) + + +def test_adding_an_unresolved_launch_is_a_row_that_claims_no_effect(): + row, = _rows(_workflow({"run": "npm test"}), _workflow({"run": "npm ci && claude -p --yolo 'go'"})) + + assert (row.direction, row.expands) == ("changed", False) + assert "review/steps[0]: runs claude -p (unresolved: compound command)" in row.after + assert "a step launches an agent in a form this audit does not read (review/steps[0])" in row.why + assert "does not say what that agent may do" in row.why + + +def test_an_action_ref_bump_is_a_step_reference_change_only(): + row, = _rows(_workflow(_agent()), _workflow({**_agent(), "uses": "anthropics/claude-code-action@v2"})) + + assert "action reference changed (review/steps[0])" in row.why + assert "agent launch" not in row.why.split("; an agent runs at")[0] + assert "claude_args" not in row.before + row.after + + +def test_renaming_or_moving_an_agent_step_within_its_job_is_quiet(): + before = _workflow({"run": "make"}, _agent()) + after = _workflow({**_agent(), "name": "Claude review"}, {"run": "make"}) + assert _rows(before, after) == [] + + +# --- direction --------------------------------------------------------------------- + + +@pytest.mark.parametrize( + ("before", "after", "rule"), + [ + (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions")), "skips permission checks"), + (_workflow({"run": "claude -p 'x'"}), _workflow({"run": "claude -p --permission-mode=bypassPermissions 'x'"}), + "skips permission checks"), + (_workflow(_agent(allowed_non_write_users="octocat")), _workflow(_agent(allowed_non_write_users="octocat, *")), + "accepts runs triggered by any user (allowed_non_write_users: *)"), + (_workflow(_agent()), _workflow(_agent(allowed_bots="*")), + "accepts runs triggered by any user (allowed_bots: *)"), + (_workflow({"uses": "openai/codex-action@v1", "with": {"sandbox": "read-only"}}), + _workflow({"uses": "openai/codex-action@v1", "with": {"sandbox": "danger-full-access"}}), + "runs without a sandbox (danger-full-access)"), + (_workflow({"uses": "openai/codex-action@v1"}), + _workflow({"uses": "openai/codex-action@v1", "with": {"safety-strategy": "unsafe"}}), + "runs without privilege restrictions"), + (_workflow({"uses": "openai/codex-action@v1"}), + _workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": '["--yolo"]'}}), + "bypasses approvals and the sandbox"), + (_workflow({"uses": "openai/codex-action@v1", "with": {"allow-users": "a"}}), + _workflow({"uses": "openai/codex-action@v1", "with": {"allow-users": "*"}}), + "accepts runs triggered by any user (allow-users: *)"), + (_workflow({"run": "codex exec 'x'"}), _workflow({"run": "codex exec --sandbox danger-full-access 'x'"}), + "runs without a sandbox"), + ], + ids=["skip-flag", "mode-flag", "gate", "bots", "codex-sandbox", "codex-unsafe", "codex-args", "codex-users", + "codex-cli"], +) +def test_a_documented_rule_gained_is_a_widening(before, after, rule): + changes = _changes(before, after) + assert host_grant_expansion_signals(changes) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert rule in row.why + + +@pytest.mark.parametrize( + ("before", "after"), + [ + # one rule, two spellings + (_workflow(_agent("--dangerously-skip-permissions")), _workflow(_agent("--permission-mode bypassPermissions"))), + # the same rule moved from the CLI to the action + (_workflow({"run": "claude -p --dangerously-skip-permissions 'x'"}), + _workflow(_agent("--dangerously-skip-permissions"))), + # narrowed + (_workflow(_agent("--dangerously-skip-permissions")), _workflow(_agent('--allowedTools "Read"'))), + (_workflow(_agent(allowed_non_write_users="*")), _workflow(_agent(allowed_non_write_users="octocat"))), + # an expression is text, never a rule + (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions ${{ inputs.extra }}"))), + # widened by a tool rule, which is #824's to rate + (_workflow(_agent('--allowedTools "Read"')), _workflow(_agent('--allowedTools "Bash(*)"'))), + (_workflow({"run": "claude -p --permission-mode default 'x'"}), + _workflow({"run": "claude -p --permission-mode acceptEdits 'x'"})), + ], + ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "expression", "tool-rule", "accept-edits"], +) +def test_any_other_edit_is_changed(before, after): + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + + +def test_a_new_workflow_that_bypasses_permissions_is_an_added_widening(): + after = _grant(_workflow(_agent("--dangerously-skip-permissions"))) + changes = diff_host_grants({"grants": []}, {"grants": [after]}) + + assert host_grant_expansion_signals(changes) == [f"workflow_agent_widened_added: {SOURCE}"] + row, = capability_diff_rows({"changes": changes, "expansion_signals": host_grant_expansion_signals(changes)}) + assert (row.direction, row.expands) == ("added", True) + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[0])" in row.why + + +def test_access_and_risk_still_describe_the_token_and_triggers_alone(): + plain, bypass = (_grant(_workflow(_agent(args))) for args in ("", "--dangerously-skip-permissions")) + assert (plain["access"], plain["risk"]) == (bypass["access"], bypass["risk"]) + + +# --- negative controls ------------------------------------------------------------- + + +def test_a_non_agent_actions_inputs_are_not_read(): + before = _workflow({"uses": "someone/ai-review@v1", "with": {"claude_args": "--allowedTools Read"}}) + after = _workflow({"uses": "someone/ai-review@v1", "with": {"claude_args": "--dangerously-skip-permissions"}}) + + assert _launches(after) == [] + assert _rows(before, after) == [] + + +def test_a_run_that_mentions_claude_in_an_echo_is_no_row(): + before = _workflow({"run": "echo done"}) + after = _workflow({"run": 'echo "claude -p --dangerously-skip-permissions"'}) + assert _rows(before, after) == [] + + +def test_a_pull_request_workflow_with_a_default_checkout_claims_no_pull_request_code(): + row, = _rows(_reproduction(), _reproduction(pr="write")) + + assert "checkout" not in row.why + assert "untrusted-input" not in row.why + assert row.why.endswith( + "an agent runs at review/steps[1] (anthropics/claude-code-action) beside the write scope pull-requests" + ) + + +@pytest.mark.parametrize( + "step", + [ + {"run": "./scripts/claude-review.sh --dangerously-skip-permissions"}, + {"uses": "./.github/actions/claude-review", "with": {"claude_args": "--dangerously-skip-permissions"}}, + {"run": "npx @anthropic-ai/claude-code -p --dangerously-skip-permissions 'go'"}, + {"run": "timeout 600 claude -p --dangerously-skip-permissions 'go'"}, + ], + ids=["script", "composite", "npx", "wrapped"], +) +def test_an_unread_surface_is_neither_a_launch_nor_a_row(step): + assert _launches(_workflow(step)) == [] + assert _rows(_workflow({"run": "echo"}), _workflow(step)) == [] + + +def test_a_removed_workflow_gets_no_note(): + before = _grant(_reproduction(trigger="issue_comment", pr="write")) + changes = diff_host_grants({"grants": [before]}, {"grants": []}) + row, = capability_diff_rows({"changes": changes, "expansion_signals": []}) + assert row.direction == "removed" + assert "an agent runs" not in row.why + + +# --- redaction (#802) -------------------------------------------------------------- + + +def test_a_credential_in_an_input_value_is_neither_published_nor_compared(): + value = '--mcp-config \'{"headers":{"Authorization":"Bearer sk-CANARYTOKEN123"}}\'' + grant = _grant(_workflow(_agent(value))) + launch, = grant["agent_launches"] + + assert launch["settings"] == [{"name": "claude_args", "value": None, "unresolved_reason": "redacted"}] + assert "CANARY" not in json.dumps(grant) + limit, = uncompared_agent_launch_texts(grant) + assert "claude_args value of the agent launch at review/steps[0]" in limit + assert "credential-shaped" in limit + + +def test_a_token_in_a_run_line_is_never_published(): + run = "claude -p --settings ghp_" + "A" * 36 + " 'review token=SECRETCANARY'" + grant = _grant(_workflow({"run": run})) + launch, = grant["agent_launches"] + + assert launch["settings"] == [{"name": "--settings", "value": None, "unresolved_reason": "redacted"}] + text = json.dumps(grant) + assert "ghp_" not in text and "SECRETCANARY" not in text + + +def test_token_shaped_job_and_step_labels_are_redacted_in_every_entry(): + job = "ghp_" + "B" * 36 + grant = _grant(_workflow(jobs={job: {"steps": [ + {"name": "Pull docker://ci:hunter2@gcr.io/x", "uses": "actions/checkout@v4"}, + {"name": "Run ghp_" + "C" * 36, "run": "claude -p 'go'"}, + ]}})) + + text = json.dumps({key: grant[key] for key in ("agent_launches", "checkout_refs")}) + assert "ghp_" not in text and "hunter2" not in text + assert grant["checkout_refs"][0]["step"] == "Pull docker://@gcr.io/x" + assert grant["agent_launches"][0]["job"] == "[REDACTED:github_token]" + + +# --- saved baselines --------------------------------------------------------------- + + +def _git(repo: Path, *args: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *args], check=True, capture_output=True, text=True + ).stdout.strip() + + +def _repo(tmp_path: Path, files: dict[str, str]) -> Path: + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "-q", "-b", "main") + _git(repo, "config", "user.name", "Fixture") + _git(repo, "config", "user.email", "fixture@example.invalid") + _write(repo, {".gitignore": "agents-shipgate-reports/\n", **files}) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "base") + return repo + + +def _write(repo: Path, files: dict[str, str]) -> None: + for name, text in files.items(): + path = repo / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + + +def _yaml(value) -> str: + return yaml.safe_dump(value, sort_keys=False) + + +def _v06_baseline(workspace: Path): + from agents_shipgate.cli.host_audit import host_audit_inventory + from agents_shipgate.core.host_grants import build_host_grants_baseline, host_grants_sha256 + + current = host_audit_inventory(workspace) + legacy = build_host_grants_baseline(current) + legacy["host_grants_schema_version"] = "0.6" + for grant in legacy["inventory"]["grants"]: + grant.pop("agent_launches", None) + grant.pop("checkout_refs", None) + legacy["inventory_sha256"] = host_grants_sha256(legacy["inventory"]) + return current, legacy + + +def test_a_v0_6_baseline_holding_a_workflow_does_not_assert_no_agent_launches(tmp_path): + from agents_shipgate.core.host_grants import build_host_drift_payload, load_host_grants_baseline + + _write(tmp_path, {SOURCE: _yaml(_reproduction())}) + current, legacy = _v06_baseline(tmp_path) + path = tmp_path / "baseline.json" + path.write_text(json.dumps(legacy)) + + drift = build_host_drift_payload( + baseline=load_host_grants_baseline(path), inventory=current, baseline_file=str(path) + ) + assert drift["comparison_status"] == "incomparable" + assert drift["incomparable_reasons"] == ["baseline_workflow_agent_launches_unavailable"] + assert drift["has_drift"] is None and drift["changes"] == [] + + +def test_a_v0_6_baseline_without_a_workflow_stays_comparable_and_saving_over_it_is_refused(tmp_path): + from agents_shipgate.core.host_grants import build_host_drift_payload + + _write(tmp_path, {".claude/settings.json": json.dumps({"permissions": {"allow": ["Read(**)"]}})}) + current, legacy = _v06_baseline(tmp_path) + drift = build_host_drift_payload(baseline=legacy, inventory=current, baseline_file="b.json") + assert (drift["comparison_status"], drift["has_drift"]) == ("comparable", False) + + path = tmp_path / ".agents-shipgate" / "host-grants.json" + path.parent.mkdir() + original = json.dumps(legacy, indent=2, sort_keys=True) + "\n" + path.write_text(original) + audit = ["audit", "--host", "--workspace", str(tmp_path), "--baseline-file", str(path)] + refused = CliRunner().invoke(app, [*audit, "--save-baseline"]) + assert refused.exit_code == 2 + assert path.read_text() == original + + path.rename(path.with_name("host-grants.v0.6.json")) + resaved = CliRunner().invoke(app, [*audit, "--save-baseline"]) + assert resaved.exit_code == 0, resaved.output + assert json.loads(path.read_text())["host_grants_schema_version"] == "0.7" + + +def test_the_documented_migration_from_a_v0_6_baseline_holding_a_workflow(tmp_path): + from tests.test_preflight import _workspace + + root = _workspace(tmp_path) + _write(root, {SOURCE: _yaml(_reproduction())}) + _, legacy = _v06_baseline(root) + path = root / ".agents-shipgate" / "host-grants.json" + path.parent.mkdir(parents=True, exist_ok=True) + original = json.dumps(legacy, indent=2, sort_keys=True) + "\n" + path.write_text(original) + audit = ["audit", "--host", "--workspace", str(root), "--baseline-file", str(path)] + + payload = json.loads(CliRunner().invoke(app, [*audit, "--drift", "--json"]).stdout) + assert payload["comparison_status"] == "incomparable" + assert payload["incomparable_reasons"] == ["baseline_workflow_agent_launches_unavailable"] + assert payload["has_drift"] is None and payload["next_action"] is None + assert CliRunner().invoke(app, [*audit, "--drift", "--fail-on-drift", "--json"]).exit_code == 20 + + preflight = CliRunner().invoke(app, ["preflight", "--workspace", str(root), "--json"]) + assert preflight.exit_code == 0, preflight.output + signal, = [item for item in json.loads(preflight.stdout)["signals"] if item["kind"] == "host_grant_drift"] + assert (signal["severity"], signal["actor"]) == ("high", "human") + assert "baseline_workflow_agent_launches_unavailable" in json.dumps(signal) + + refused = CliRunner().invoke(app, [*audit, "--save-baseline"]) + assert refused.exit_code == 2 + assert "unsupported_baseline_schema" in refused.output + (refused.stderr or "") + assert path.read_text() == original + + path.rename(path.with_name("host-grants.v0.6.json")) + resaved = CliRunner().invoke(app, [*audit, "--save-baseline"]) + assert resaved.exit_code == 0, resaved.output + after = json.loads(CliRunner().invoke(app, [*audit, "--drift", "--json"]).stdout) + assert (after["comparison_status"], after["has_drift"]) == ("comparable", False) + assert path.with_name("host-grants.v0.6.json").read_text() == original + + +def test_a_current_baseline_compares_agent_launches_and_validates_against_the_schemas(tmp_path): + from jsonschema import Draft202012Validator + + from agents_shipgate.cli.host_audit import host_audit_inventory + from agents_shipgate.core.host_grants import ( + build_host_drift_payload, + build_host_grants_baseline, + ) + + path = tmp_path / SOURCE + path.parent.mkdir(parents=True) + path.write_text(_yaml(_reproduction(run="npm ci && claude -p 'x'", ref=HEAD_SHA))) + inventory = host_audit_inventory(tmp_path) + baseline = build_host_grants_baseline(inventory) + assert baseline["host_grants_schema_version"] == "0.7" + for name, payload in (("inventory", inventory), ("baseline", baseline)): + schema = json.loads((ROOT / f"docs/host-grants-{name}-schema.v0.7.json").read_text()) + Draft202012Validator(schema).validate(payload) + + path.write_text(_yaml(_reproduction(claude_args="--dangerously-skip-permissions", ref=HEAD_SHA))) + drift = build_host_drift_payload(baseline=baseline, inventory=host_audit_inventory(tmp_path), baseline_file="b.json") + assert (drift["comparison_status"], drift["has_drift"]) == ("comparable", True) + assert drift["expansion_signals"] == [f"workflow_agent_widened_changed: {SOURCE}"] + schema = json.loads((ROOT / "docs/host-grants-drift-schema.v0.7.json").read_text()) + Draft202012Validator(schema).validate(drift) + + +def test_an_unresolved_launch_is_a_non_blocking_limit_that_leaves_coverage_complete(tmp_path): + from agents_shipgate.cli.host_audit import host_audit_inventory + + _write(tmp_path, {SOURCE: _yaml(_workflow({"run": "npm ci && claude -p 'go'"}))}) + inventory = host_audit_inventory(tmp_path) + + github, = [item for item in inventory["host_coverage"] if item["host"] == "github"] + assert github["status"] == "complete" + issue, = [item for item in inventory["issues"] if item["host"] == "github"] + assert (issue["kind"], issue["blocking"]) == ("unsupported", False) + assert "review/steps[0] (claude)" in issue["message"] + + +# --- the same row on every route --------------------------------------------------- + + +@pytest.fixture +def pr(tmp_path): + repo = _repo(tmp_path, {SOURCE: _yaml(_reproduction())}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(_reproduction(claude_args='--permission-mode bypassPermissions --allowedTools "Bash(*)"'))}) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "bypass permissions") + return repo + + +def _assert_the_row(row: dict) -> None: + assert row["subject"] == f"github {SOURCE}" + assert 'review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools "Read"' in row["before"] + assert "--permission-mode bypassPermissions" in row["after"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert "skips permission checks" in row["why"] + + +def _diff(repo: Path, *args: str) -> dict: + result = CliRunner().invoke(app, ["diff", "--workspace", str(repo), "--base", "main", *args, "--json"]) + assert result.exit_code == 0, result.output + return json.loads(result.output) + + +def test_diff_names_the_widening_in_json_and_text(pr): + payload = _diff(pr) + assert payload["comparison_status"] == "comparable" + row, = payload["rows"] + _assert_the_row(row) + + text = CliRunner().invoke(app, ["diff", "--workspace", str(pr), "--base", "main"]) + assert text.exit_code == 0, text.output + assert "⚠" in text.output and "review/steps[1]" in text.output + assert "1 widening what the agent may do" in text.output + + +def test_manifest_free_verify_and_its_pr_comment_name_the_widening(pr): + args = ["verify", "--workspace", str(pr), "--base", "main", "--head", "HEAD", "--format", "text"] + result = CliRunner().invoke(app, args) + assert result.exit_code == 0, result.output + assert "bypassPermissions" in result.output + + comment = (pr / "agents-shipgate-reports/pr-comment.md").read_text() + assert "bypassPermissions" in comment and "review/steps[1]" in comment + verifier = json.loads((pr / "agents-shipgate-reports/verifier.json").read_text()) + row, = verifier["host_comparison"]["rows"] + _assert_the_row(row) + + +def test_check_and_the_control_envelope_carry_the_same_row(pr): + args = ["check", "--workspace", str(pr), "--base", "main", "--head", "HEAD"] + machine = CliRunner().invoke(app, [*args, "--format", "agent-boundary-json"]) + assert machine.exit_code == 0, machine.output + row, = json.loads(machine.output)["rows"] + _assert_the_row(row) + + control = CliRunner().invoke(app, [*args, "--format", "agent-control-json"]) + assert control.exit_code == 0, control.output + envelope_row, = json.loads(control.output)["capability_rows"]["rows"] + _assert_the_row(envelope_row) + + +def test_the_stop_hook_announces_the_widening(pr, tmp_path): + from tests.test_install_hooks import _host_diff_workspace, _run_hook + + payload = _diff(pr) + hooked = tmp_path / "hooked" + hooked.mkdir() + _host_diff_workspace(hooked) + result = _run_hook(hooked, "verify", {}, diff_payload=json.dumps(payload)) + + assert result.returncode == 0, result.stderr + message = json.loads(result.stdout)["systemMessage"] + assert "These rows widen what the agent can do" in message + assert SOURCE in message + + +def test_no_canary_reaches_any_published_output(tmp_path): + canary = "sk-ant-api03-" + "Z" * 40 + job = "ghp_" + "D" * 36 + base = _workflow(jobs={job: {"steps": [_agent()]}}) + head = _workflow(jobs={job: {"steps": [ + {"name": "Pull docker://ci:" + "p4ssCANARY" + "@gcr.io/x", "uses": "actions/checkout@v4", + "with": {"ref": "token=REFCANARY"}}, + _agent(f"--mcp-config '{{\"headers\":{{\"Authorization\":\"Bearer {canary}\"}}}}'"), + {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read 'go'"}, + ]}}) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "canaries") + + outputs = [json.dumps(_diff(repo))] + for args in ( + ["diff", "--workspace", str(repo), "--base", "main"], + ["audit", "--host", "--workspace", str(repo), "--json"], + ["check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "agent-boundary-json"], + ["verify", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "text"], + ): + result = CliRunner().invoke(app, args) + outputs.append(result.output) + outputs.append((repo / "agents-shipgate-reports/pr-comment.md").read_text()) + outputs.append((repo / "agents-shipgate-reports/verifier.json").read_text()) + joined = "\n".join(outputs) + + for secret in (canary, "p4ssCANARY", "REFCANARY", job): + assert secret not in joined + assert "runs claude -p with --allowedTools Read" in joined diff --git a/tests/test_workflow_step_action_references.py b/tests/test_workflow_step_action_references.py index 242b299fd..61eaf2827 100644 --- a/tests/test_workflow_step_action_references.py +++ b/tests/test_workflow_step_action_references.py @@ -394,6 +394,9 @@ def _legacy_baseline(tmp_path: Path, version: str): for grant in legacy["inventory"]["grants"]: if grant["kind"] == "workflow": del grant["step_actions"] + # Nor did they read agent launches or checkout refs (#823). + grant.pop("agent_launches", None) + grant.pop("checkout_refs", None) legacy["inventory_sha256"] = host_grants_sha256(legacy["inventory"]) return current, legacy @@ -415,7 +418,10 @@ def test_a_legacy_baseline_holding_a_workflow_does_not_assert_no_step_references drift = build_host_drift_payload(baseline=loaded, inventory=current, baseline_file=str(baseline_path)) assert drift["comparison_status"] == "incomparable" - assert drift["incomparable_reasons"] == ["baseline_workflow_step_actions_unavailable"] + assert drift["incomparable_reasons"] == [ + "baseline_workflow_agent_launches_unavailable", + "baseline_workflow_step_actions_unavailable", + ] assert drift["has_drift"] is None and drift["changes"] == [] assert baseline_path.read_text() == original @@ -454,7 +460,7 @@ def test_a_current_baseline_compares_step_references(tmp_path): path.parent.mkdir(parents=True) path.write_text(_yaml({"uses": f"actions/checkout@{PINNED}"})) baseline = build_host_grants_baseline(host_audit_inventory(tmp_path)) - assert baseline["host_grants_schema_version"] == "0.6" + assert baseline["host_grants_schema_version"] == "0.7" path.write_text(_yaml({"uses": "actions/checkout@main"})) drift = build_host_drift_payload(baseline=baseline, inventory=host_audit_inventory(tmp_path), baseline_file="b.json") @@ -821,7 +827,7 @@ def test_saving_over_a_legacy_baseline_without_a_workflow_is_refused(tmp_path, v path.rename(path.with_name(f"host-grants.v{version}.json")) resaved = CliRunner().invoke(app, [*audit, "--save-baseline"]) assert resaved.exit_code == 0, _output(resaved) - assert json.loads(path.read_text())["host_grants_schema_version"] == "0.6" + assert json.loads(path.read_text())["host_grants_schema_version"] == "0.7" def _output(result) -> str: @@ -915,6 +921,8 @@ def test_the_documented_migration_from_a_legacy_baseline_holding_a_workflow(tmp_ for grant in legacy["inventory"]["grants"]: if grant["kind"] == "workflow": grant.pop("step_actions", None) + grant.pop("agent_launches", None) + grant.pop("checkout_refs", None) legacy["inventory_sha256"] = host_grants_sha256(legacy["inventory"]) path = root / ".agents-shipgate" / "host-grants.json" path.parent.mkdir(parents=True, exist_ok=True) @@ -925,7 +933,10 @@ def test_the_documented_migration_from_a_legacy_baseline_holding_a_workflow(tmp_ drift = CliRunner().invoke(app, [*audit, "--drift", "--json"]) payload = json.loads(drift.stdout) assert payload["comparison_status"] == "incomparable" - assert payload["incomparable_reasons"] == ["baseline_workflow_step_actions_unavailable"] + assert payload["incomparable_reasons"] == [ + "baseline_workflow_agent_launches_unavailable", + "baseline_workflow_step_actions_unavailable", + ] assert payload["has_drift"] is None and payload["next_action"] is None assert CliRunner().invoke(app, [*audit, "--drift", "--fail-on-drift", "--json"]).exit_code == 20 @@ -945,7 +956,7 @@ def test_the_documented_migration_from_a_legacy_baseline_holding_a_workflow(tmp_ path.rename(path.with_name("host-grants.v0.5.json")) resaved = CliRunner().invoke(app, [*audit, "--save-baseline"]) assert resaved.exit_code == 0, resaved.output - assert json.loads(path.read_text())["host_grants_schema_version"] == "0.6" + assert json.loads(path.read_text())["host_grants_schema_version"] == "0.7" after = json.loads(CliRunner().invoke(app, [*audit, "--drift", "--json"]).stdout) assert (after["comparison_status"], after["has_drift"]) == ("comparable", False) assert path.with_name("host-grants.v0.5.json").read_text() == original From 45795a00bfd4aff3329e88d80ae75f3b7de00852 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Tue, 22 Sep 2026 12:26:13 -0700 Subject: [PATCH 02/14] Read no widening rule from an agent input that holds an expression (#823) The support page and STABILITY say a documented rule is read from literal values only, so a value holding `${{ }}` never meets one. The `*` user gate did not apply that: `allowed_non_write_users: "${{ vars.USERS }}, *"` raised workflow_agent_widened_changed. Every rule now skips a value holding an expression, as the claude_args and codex-args rules already did, and a test pins the gate case. --- src/agents_shipgate/core/host_grants.py | 3 +++ tests/test_workflow_agent_launches.py | 4 +++- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 26a6ebfb6..168d94a5b 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -2100,10 +2100,13 @@ def agent_widening_rules(entry: dict[str, Any]) -> set[tuple[str, str]]: if entry.get("form") != "read": return set() agent = str(entry["agent"]) + # Only literal values: GitHub substitutes a `${{ }}` expression into the + # input before the action reads it, so what it holds is not known here. readable = { str(setting["name"]): str(setting["value"]) for setting in entry.get("settings", []) if setting.get("unresolved_reason") is None and setting.get("value") is not None + and "${{" not in str(setting["value"]) } rules: set[tuple[str, str]] = set() if agent in {"claude", "codex"}: diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index e77ae0a5e..c4eff3ac7 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -447,12 +447,14 @@ def test_a_documented_rule_gained_is_a_widening(before, after, rule): (_workflow(_agent(allowed_non_write_users="*")), _workflow(_agent(allowed_non_write_users="octocat"))), # an expression is text, never a rule (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions ${{ inputs.extra }}"))), + (_workflow(_agent()), _workflow(_agent(allowed_non_write_users="${{ vars.USERS }}, *"))), # widened by a tool rule, which is #824's to rate (_workflow(_agent('--allowedTools "Read"')), _workflow(_agent('--allowedTools "Bash(*)"'))), (_workflow({"run": "claude -p --permission-mode default 'x'"}), _workflow({"run": "claude -p --permission-mode acceptEdits 'x'"})), ], - ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "expression", "tool-rule", "accept-edits"], + ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "expression", "gate-expression", "tool-rule", + "accept-edits"], ) def test_any_other_edit_is_changed(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] From d4b2d7ae78112ecffa386e506d1c8f739c2bd258 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Tue, 22 Sep 2026 13:53:51 -0700 Subject: [PATCH 03/14] Address review cycle 1 on agent launches in CI (#823) F1. `claude_args` and `codex-args` were split with the `run:` shell tokenizer, which gave up on a newline or an unquoted `(`, so a widening written the way the actions document it gave `changed`: `claude_args: |` on several lines, `--allowedTools Bash(git:*) --dangerously-skip-permissions`, a `# comment` line, or a multi-line `codex-args`. Each input is now split the way its action splits it. The Claude actions' parse-sdk-options.ts drops full `#` lines, makes `()|&;<>` literal and reads the rest with shell-quote (newlines are whitespace, `$NAME` is empty, an unquoted `#` ends the input, a `--` word is always a flag); codex-action reads a JSON array of strings or string-argv. The `run:` tokenizer is kept for `run:` steps only. The published `claude_args` is the text the action parses, so a full-line comment is neither published nor compared. F2. Any URL path made a whole setting `redacted`, so a `--dangerously-skip-permissions` beside `https://example.com/style-guide` gave no row, and every `plugin_marketplaces` value was never compared, under a limit that wrongly called it credential-shaped. The documented rules are now decided from the declared text when the workflow is read, before anything is withheld, and published on the launch as `widening_rules` (rule and setting), which the comparator keys on, so redaction never hides a rule. A URL publishes its scheme and host with ``, as an MCP server URL does (#723), and the rest of the value is published and compared. Other credential-shaped text in a setting or checkout ref (a token shape, an assignment, a bearer or header value, URL userinfo) is published redacted and makes the workflow a blocking limit through `_uncompared_workflow_text`, as a redacted step reference does (#767). F3. JSON in `settings`, `mcp_config`, `--settings`, `--mcp-config` or any argument word was published verbatim, env values and apiKeyHelper included. A JSON object now publishes what `.claude/settings.json` and `.mcp.json` publish: key names, with `env`/`headers` values, apiKeyHelper and secret-named values ``, as canonical JSON. A codex `--config` override under env, headers or a secret-named key publishes ``. Text that starts like JSON and does not parse is withheld (`unparsed_json`, a non-blocking limit). The agent-launch canary sweep now carries JSON-shaped canaries and their digests across diff, audit, check, verify, the PR comment and verifier.json, and a second sweep covers the refused credential-shaped case. Nonblocking: a rule gained where the job launched that agent before only in an unread form is named and not claimed (the `unknown_before` rule); a quoted word starting with `#` no longer makes a `run:` compound; `anthropics/claude-code-action/base-action` is read as the base action; the STABILITY note says a prompt after a variadic flag is compared; and `__all__ =[` is spaced. Host-grants 0.7 is unreleased, so `widening_rules`, `unparsed_json` and the base-action agent value extend it in place; the 0.7 schema files are regenerated. The support page, STABILITY migration note, CHANGELOG Unreleased entry, contract summary, integrations Stop-hook sentence and the capability_diff distribution-surface row say the same. --- CHANGELOG.md | 2 +- STABILITY.md | 22 +- docs/agent-contract-current.md | 13 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 79 +- docs/host-grants-baseline-schema.v0.7.json | 44 +- docs/host-grants-inventory-schema.v0.7.json | 44 +- docs/integrations.md | 4 +- llms-full.txt | 13 +- .../core/capability_diff_rows.py | 17 +- src/agents_shipgate/core/host_grants.py | 758 +++++++++++++++--- src/agents_shipgate/schemas/host_grants.py | 68 +- tests/test_distribution_surface_parity.py | 3 +- tests/test_workflow_agent_launches.py | 429 +++++++++- 14 files changed, 1283 insertions(+), 215 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5e17ea3ee..052ec75bc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,7 +11,7 @@ - **What it is not.** Never a row, a widening, a `check` violation or a claim that a host loads the file. Nothing is fetched or run, only plugin manifests and marketplaces are read, and at most 32 candidates are examined; the rest, and any whose rule needed a file that was not read or did not parse, are counted as not examined, on a line that names both causes. The rules are listed in `docs/host-boundary-support.md` under *Changed inputs named but not read*. - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text; every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A setting publishes what the host readers would: a JSON object its key names, with `env` and `headers` values and `apiKeyHelper` withheld, and a URL its scheme and host; other credential-shaped text in a setting or checkout ref is published redacted and refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index 958af0f0c..295a28f8e 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -27,9 +27,14 @@ or the permission flags of a `run:` that is one literal `claude -p` or `checkout_refs[]`, each `actions/checkout` step's `with.ref`. Only a documented rule a job's launches gain widens (`workflow_agent_widened_`); every other edit is a `changed` row naming `job/step`, and a workflow row that -runs an agent ends with the job facts beside each agent step. A compound -command, an expansion or an expression is `unresolved`, a named non-blocking -limit that leaves coverage complete. A `0.4`–`0.6` baseline holding a workflow +runs an agent ends with the job facts beside each agent step. The rules each +launch meets are read from its declared text and published as +`widening_rules`, and `claude_args` and `codex-args` are split as each action +splits them. A setting publishes what the host readers would: a JSON object its +key names, a URL its scheme and host. A compound command, an expansion or an +expression is `unresolved`, a named non-blocking limit that leaves coverage +complete; a setting or checkout ref holding credential-shaped text is published +redacted and refuses as a redacted step reference does. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and `minimum_control_contract_version` `21` are unchanged. See @@ -332,6 +337,7 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i "settings": [ {"name": "claude_args", "value": "--permission-mode bypassPermissions --allowedTools \"Bash(*)\"", "unresolved_reason": null} ], + "widening_rules": [{"rule": "bypass_permissions", "setting": "claude_args"}], "job_secrets": ["CLAUDE_CODE_OAUTH_TOKEN"] } ], @@ -341,11 +347,13 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i } ``` -- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). -- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. The prompt, `--model` and any undocumented flag of a CLI launch are not compared; an agent action's `claude_args` or `codex-args` is compared whole. -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. It is read from literal values only, so a value holding `${{ }}` never meets a rule. The grant then earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). +- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings`, `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. +- **What is withheld.** A setting publishes what the host readers would publish for the same text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value, or any word of `claude_args` or `codex-args` — publishes its key names, with `env` and `headers` values, `apiKeyHelper` and every secret-named value ``, as canonical JSON; a codex `--config` override under `env`, `headers` or a secret-named key publishes `` for its value. A URL publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to a URL's path or query is not reported. Text that starts like JSON and does not parse is withheld whole (`unparsed_json`). +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal values only, so a value holding `${{ }}` never meets a rule. A rule gained where the job launched that agent before only in a form this audit does not read is named in the `why` and not claimed, as for a job whose permissions were not explicit. The grant then earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. -- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting or ref the label redaction rewrites (`redacted`) or that is not a string (`not_a_string`) has `value`/`ref: null`. Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry is still a row, one that claims no effect. Only an edit inside it is not reported. +- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. +- **Credential-shaped values refuse, as a redacted step reference does (#767).** Other text the #802 label redaction rewrites in a setting or a checkout ref — a token shape, a credential assignment, a bearer or header value, a URL's userinfo — is published redacted with `unresolved_reason: redacted` and makes the workflow a blocking `unsupported` limit, because two values that redact alike cannot be compared apart: a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. Its rules are still read from the declared text. - **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, a path), and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. **Compatibility.** diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index 1bae576af..2e5820ad3 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -53,16 +53,21 @@ An agent launch is a step whose `uses:` is a documented agent action `openai/codex-action`) with the permission inputs it declares, or a `run:` that is one literal `claude -p` / `codex exec` command with its documented permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`) and -`job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, -`null` for the default. Values are compared as text and never executed. Only a +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`), +`widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is +each `actions/checkout` step's `with.ref`, `null` for the default. Values are +compared as text and never executed; `claude_args` and `codex-args` are split +as each action splits them, and a setting publishes what the host readers +would, a JSON object by its key names and a URL by its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`; every other edit is `changed`, and a workflow row that runs an agent ends its `why` with the job facts beside each agent step. A compound `run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit. A `0.4`–`0.6` baseline holding a workflow grant is incomparable +limit; a setting or checkout ref holding credential-shaped text is published +redacted and a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and `minimum_control_contract_version` `21` are unchanged. See diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index b65ae8912..fa439b9c1 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`), with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch or unreadable value is named only by the host inventory and `audit --host`, as for an unread secret value (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`), with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch or unreadable value is named only by the host inventory and `audit --host`, as for an unread secret value, and a setting or checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index c094c9360..69114b48d 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -125,9 +125,10 @@ redacted and refuses the same way a step reference does, so two values that redact alike never compare as unchanged. How a coding agent is launched inside a job is read (#823). Every value is -compared as declared text; no action is fetched, no command is run and no -expression is evaluated. Three things are listed on the workflow grant, each -naming its `job/step` (the step's `id`, else its `name`, else `steps[N]`): +compared as the text it declares, less what the host readers withhold (below); +no action is fetched, no command is run and no expression is evaluated. Three +things are listed on the workflow grant, each naming its `job/step` (the step's +`id`, else its `name`, else `steps[N]`): - **A documented agent action** — a step whose `uses:` is one of these `owner/repo` references, at any ref and in any letter case — with the inputs @@ -137,9 +138,22 @@ naming its `job/step` (the step's `id`, else its `name`, else `steps[N]`): | Action | Inputs compared as text | Documented widening | |---|---|---| | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions`; `allowed_bots` or `allowed_non_write_users` gains a `*` entry | - | `anthropics/claude-code-base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` as above | + | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` as above | | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access`; `allow-users` gains a `*` entry | + No shell reads `claude_args` or `codex-args`: each action splits its own + input, and a rule is met only by the words the action passes on. The + Claude actions (`base-action/src/parse-sdk-options.ts`) drop each line whose + first non-blank character is `#`, which is then neither published nor + compared, and split the rest with shell-quote, taking `()|&;<>` literally: + newlines separate words as spaces do, so `claude_args: |` on several lines + reads as it would on one; quotes and backslashes work as in a shell; + `$NAME` reads as empty; an unquoted `#` later in the input ends it; and a + word starting with `--` is always a flag, never another flag's value. + `openai/codex-action` reads `codex-args` as a JSON array of strings or, when + it does not start with `[`, as words separated by whitespace, newlines + included, with a quoted string kept together (string-argv). + - **A literal agent CLI command in `run:`** — only when the whole `run:` is one simple command, after any literal `NAME=value` assignments (which are skipped and never published), that starts with `claude` and passes `-p`/`--print`, or @@ -175,10 +189,19 @@ naming its `job/step` (the step's `id`, else its `name`, else `steps[N]`): Each job's multiset of launches and of checkout refs is compared, so renaming or reordering steps is quiet. An added, removed or changed launch or ref is a `changed` row on the workflow naming `job/step` and the value on each side. -Direction is claimed only by the documented rules above, read from literal -values: when a job's launches gain one, the workflow earns +Direction is claimed only by the documented rules above. The rules each launch +meets are decided when the workflow is read, from the declared text before +anything is withheld for publication, and published on the launch as +`widening_rules` (each rule and the setting it was read from), so redaction +never hides one. Only literal values meet a rule: GitHub substitutes a +`${{ }}` expression into an input before the action reads it, so a value +holding one meets none. When a job's launches gain one, the workflow earns `workflow_agent_widened_`, the row is `widened` and its `why` -names the rule and step. Any other edit — `--allowedTools "Read"` to +names the rule and step. A rule gained where the job launched that agent before +only in a form this audit does not read — a compound `run:` that became a +literal one — is named in the `why` and not claimed, because the unread launch +may already have met it, as a job whose permissions were not explicit may +already have held a write scope. Any other edit — `--allowedTools "Read"` to `--allowedTools "Bash(*)"`, `acceptEdits`, a new plugin, a value holding `${{ }}`, a head-ref checkout — is `changed`; rating a tool rule's reach is a job for #824. `access` and `risk` still describe the token and triggers alone. @@ -194,14 +217,40 @@ code (`github.event.pull_request.head.sha`, `.head.ref` or `.merge_commit_sha`, direction, and `if:` conditions and the default checkout of a `pull_request` event are not read into it. A removed workflow gets no note. -An unresolved launch, a setting whose value the label redaction rewrites or -that is not a string, and a checkout ref of either kind publish nothing of the -value and record a **non-blocking** `unsupported` coverage issue naming the -`job/step`, printed under `audit --host` → Coverage issues. GitHub coverage -stays complete, so `check`, baselines and every other row are unaffected, and -adding, removing or re-forming such an entry is still a row that claims no -effect; only an edit inside it is not reported. `diff`, `verify` and `check` -carry no limit for it, as for an unread secret value (#693). +A setting publishes what the host readers would publish for the same text. A +JSON object — a `settings` or `mcp_config` value, a `--settings` or +`--mcp-config` value, or any word of `claude_args` or `codex-args` — publishes +as `.claude/settings.json` and `.mcp.json` do: its key names, with `env` and +`headers` values, `apiKeyHelper` and every other secret-named value read as +``, in canonical JSON, so rotating an `env` value or reordering keys +compares as unchanged and adding a key is a change. A codex `--config` +override under `env`, `headers` or a secret-named key publishes `` for +its value, and a table or array value by the same JSON rule. A URL publishes +its scheme, host and port, with `` for any path and no query, as +an MCP server's URL does (#723), and the rest of the setting is compared, so a change +only to a URL's path or query — which repository a `plugin_marketplaces` URL +names, for one — is not reported; a zero-row result says redacted values are +not compared. + +Any other text the #802 label redaction rewrites in a setting or a checkout ref +is credential-shaped: a token shape, a credential assignment such as `token=…`, +a bearer or header value, a URL's userinfo. The value is published redacted +with `unresolved_reason: redacted` and refuses the same way a step reference +does, because two values that redact alike cannot be compared apart: GitHub +coverage is `partial`, a changed workflow's comparison is refused, and an +unchanged one is named in `unchanged_limits`. Its rules are still read from the +declared text. + +An unresolved launch, a setting that is not a string (`not_a_string`) or that +holds text starting like JSON that does not parse (`unparsed_json`, +whose values cannot be told from its keys), and a checkout ref that is not a +string or whose `with:` is not a mapping publish nothing of the value and +record a **non-blocking** `unsupported` coverage issue naming the `job/step`, +printed under `audit --host` → Coverage issues. GitHub coverage stays complete, +so `check`, baselines and every other row are unaffected, and adding, removing +or re-forming such an entry, or its gaining a documented rule, is still a row; +only an edit inside it that gains no rule is not reported. `diff`, `verify` and +`check` carry no limit for it, as for an unread secret value (#693). A workflow's labels are published redacted (#802). A job id, a step's `id` or `name`, a trigger and a permission scope name go through the same redaction as diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 21e8c7208..dc781c14f 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1284,12 +1284,13 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with:\n``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read``\nlists the documented permission inputs or flags the step declares in\n``settings``. ``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ "anthropics/claude-code-action", "anthropics/claude-code-base-action", + "anthropics/claude-code-action/base-action", "openai/codex-action", "claude", "codex" @@ -1344,6 +1345,13 @@ ], "default": null, "title": "Unresolved Reason" + }, + "widening_rules": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentRuleV7" + }, + "title": "Widening Rules", + "type": "array" } }, "required": [ @@ -1355,9 +1363,36 @@ "title": "HostWorkflowAgentLaunchV7", "type": "object" }, + "HostWorkflowAgentRuleV7": { + "additionalProperties": false, + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only a\nliteral value meets one: a value holding ``${{ }}`` meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "properties": { + "rule": { + "enum": [ + "bypass_permissions", + "bypass_approvals_and_sandbox", + "danger_full_access", + "unsafe_safety_strategy", + "open_gate" + ], + "title": "Rule", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + } + }, + "required": [ + "rule", + "setting" + ], + "title": "HostWorkflowAgentRuleV7", + "type": "object" + }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, published through the workflow\nlabel redaction (#802); a flag that takes no value has ``null``. A value\nthe redaction rewrites, or one that is not a string, is ``null`` with\n``unresolved_reason``, and records a non-blocking coverage issue naming\nits ``job/step``: it is neither published nor compared.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). The rest is published through the workflow label\nredaction (#802). A value it rewrites beyond that is credential-shaped: it\nis published redacted with ``unresolved_reason: redacted`` and makes the\nworkflow a blocking limit, as a redacted step reference does (#767). A\nvalue that is not a string (``not_a_string``), or one holding text that\nstarts like JSON and does not parse (``unparsed_json``), is\n``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.", "properties": { "name": { "title": "Name", @@ -1368,7 +1403,8 @@ { "enum": [ "not_a_string", - "redacted" + "redacted", + "unparsed_json" ], "type": "string" }, @@ -1400,7 +1436,7 @@ }, "HostWorkflowCheckoutRefV7": { "additionalProperties": false, - "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites, a value that is not a string, or ``with:`` that is not a mapping\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767). A value that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", "properties": { "job": { "title": "Job", diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index 2f9b6c6f2..6d6dc4b4f 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1342,12 +1342,13 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with:\n``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read``\nlists the documented permission inputs or flags the step declares in\n``settings``. ``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ "anthropics/claude-code-action", "anthropics/claude-code-base-action", + "anthropics/claude-code-action/base-action", "openai/codex-action", "claude", "codex" @@ -1402,6 +1403,13 @@ ], "default": null, "title": "Unresolved Reason" + }, + "widening_rules": { + "items": { + "$ref": "#/$defs/HostWorkflowAgentRuleV7" + }, + "title": "Widening Rules", + "type": "array" } }, "required": [ @@ -1413,9 +1421,36 @@ "title": "HostWorkflowAgentLaunchV7", "type": "object" }, + "HostWorkflowAgentRuleV7": { + "additionalProperties": false, + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only a\nliteral value meets one: a value holding ``${{ }}`` meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "properties": { + "rule": { + "enum": [ + "bypass_permissions", + "bypass_approvals_and_sandbox", + "danger_full_access", + "unsafe_safety_strategy", + "open_gate" + ], + "title": "Rule", + "type": "string" + }, + "setting": { + "title": "Setting", + "type": "string" + } + }, + "required": [ + "rule", + "setting" + ], + "title": "HostWorkflowAgentRuleV7", + "type": "object" + }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, published through the workflow\nlabel redaction (#802); a flag that takes no value has ``null``. A value\nthe redaction rewrites, or one that is not a string, is ``null`` with\n``unresolved_reason``, and records a non-blocking coverage issue naming\nits ``job/step``: it is neither published nor compared.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). The rest is published through the workflow label\nredaction (#802). A value it rewrites beyond that is credential-shaped: it\nis published redacted with ``unresolved_reason: redacted`` and makes the\nworkflow a blocking limit, as a redacted step reference does (#767). A\nvalue that is not a string (``not_a_string``), or one holding text that\nstarts like JSON and does not parse (``unparsed_json``), is\n``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.", "properties": { "name": { "title": "Name", @@ -1426,7 +1461,8 @@ { "enum": [ "not_a_string", - "redacted" + "redacted", + "unparsed_json" ], "type": "string" }, @@ -1458,7 +1494,7 @@ }, "HostWorkflowCheckoutRefV7": { "additionalProperties": false, - "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites, a value that is not a string, or ``with:`` that is not a mapping\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767). A value that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", "properties": { "job": { "title": "Job", diff --git a/docs/integrations.md b/docs/integrations.md index 8e55adec6..485ee999a 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -227,7 +227,9 @@ non-widening row, so the hook stays quiet about it; `diff` and the PR comment still show it. The same holds for an agent launch in a workflow whose settings change without gaining a documented widening rule, and for a checkout's ref. One that gains a rule, such as `claude_args` gaining -`--dangerously-skip-permissions`, widens, and the hook announces it (#823). +`--dangerously-skip-permissions` on any of its lines, widens, and the hook +announces it (#823), unless the job launched that agent before only in a form +the audit does not read. It names each widening row once, and repeats the announcement only when the change or its rows change. A missing base ref, an incomparable inventory or unparsed output is never diff --git a/llms-full.txt b/llms-full.txt index 3b780438f..5476338d7 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1627,16 +1627,21 @@ An agent launch is a step whose `uses:` is a documented agent action `openai/codex-action`) with the permission inputs it declares, or a `run:` that is one literal `claude -p` / `codex exec` command with its documented permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`) and -`job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, -`null` for the default. Values are compared as text and never executed. Only a +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`), +`widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is +each `actions/checkout` step's `with.ref`, `null` for the default. Values are +compared as text and never executed; `claude_args` and `codex-args` are split +as each action splits them, and a setting publishes what the host readers +would, a JSON object by its key names and a URL by its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`; every other edit is `changed`, and a workflow row that runs an agent ends its `why` with the job facts beside each agent step. A compound `run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit. A `0.4`–`0.6` baseline holding a workflow grant is incomparable +limit; a setting or checkout ref holding credential-shaped text is published +redacted and a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and `minimum_control_contract_version` `21` are unchanged. See diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index f5f016324..87eca1869 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -30,6 +30,7 @@ AGENT_WIDENING_RULES, UNTRUSTED_INPUT_TRIGGERS, agent_launch_key, + agent_widenings_unread_before, checkout_ref_key, gained_agent_widenings, hook_loading_basis, @@ -323,9 +324,11 @@ def _agent_launch_reasons( """What changed in how an agent is launched, and which of it widens (#823). Only a documented rule gained by a job's agent launches is called a - widening, and the sentence says which rule and where. Every other - agent-launch or checkout edit is a change: its settings are compared as - declared text, and nothing here ranks one value against another. + widening, and the sentence says which rule and where. A rule gained where + the job's launch was unread before is named and not called a widening, as + the engine claims no expansion for it. Every other agent-launch or + checkout edit is a change: its settings are compared as published text, + and nothing here ranks one value against another. """ def where(item: dict[str, Any]) -> str: @@ -337,6 +340,14 @@ def where(item: dict[str, Any]) -> str: widened_at.add(where(entry)) what = AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") reasons.append(f"an agent launch now {what} ({where(entry)})") + for _job, rule, detail, entry in agent_widenings_unread_before(before, after): + widened_at.add(where(entry)) + what = AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") + reasons.append( + f"an agent launch now {what} ({where(entry)}), which is not counted as a widening: " + "before, this job launched the agent in a form this audit does not read, which may " + "already have done the same" + ) # The step label only words the sentence; what changed was decided by the # comparator's key, which never reads it. old = {where(item) for item in gone} diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 168d94a5b..54988a6a0 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -1633,10 +1633,21 @@ class _AgentAction: modes: tuple[tuple[str, str], ...] = () +_CLAUDE_BASE_ACTION = _AgentAction( + family="claude", + inputs=( + "allowed_tools", "claude_args", "disallowed_tools", "mcp_config", + "plugin_marketplaces", "plugins", "settings", + ), + args="claude_args", +) + #: The documented agent actions, by ``owner/repo`` (matched case-insensitively, #: at any ref). The input names are those the actions' own `action.yml` #: declare; `allowed_tools`, `disallowed_tools` and `mcp_config` are the -#: Claude actions' earlier inputs, still read when a workflow sets them. +#: Claude actions' earlier inputs, still read when a workflow sets them. The +#: base action is published both as its own repository and as the +#: `base-action` directory of `anthropics/claude-code-action`. _AGENT_ACTIONS: dict[str, _AgentAction] = { "anthropics/claude-code-action": _AgentAction( family="claude", @@ -1648,14 +1659,8 @@ class _AgentAction: gates=("allowed_bots", "allowed_non_write_users"), args="claude_args", ), - "anthropics/claude-code-base-action": _AgentAction( - family="claude", - inputs=( - "allowed_tools", "claude_args", "disallowed_tools", "mcp_config", - "plugin_marketplaces", "plugins", "settings", - ), - args="claude_args", - ), + "anthropics/claude-code-base-action": _CLAUDE_BASE_ACTION, + "anthropics/claude-code-action/base-action": _CLAUDE_BASE_ACTION, "openai/codex-action": _AgentAction( family="codex", inputs=( @@ -1782,6 +1787,34 @@ def _has_shell_expansion(text: str) -> bool: return False +def _has_shell_comment(text: str) -> bool: + """An unquoted ``#`` that starts a word: the shell reads the rest of the line as a comment. + + Read off the raw text because the word splitter keeps ``#`` as an + ordinary character, so a quoted ``"#123 review"`` is one word, not a comment. + """ + + quote: str | None = None + escaped = False + previous = " " + for char in text: + if escaped: + # An escaped character, a space included, is part of the word. + escaped = False + previous = "\\" + continue + if char == "\\" and quote != "'": + escaped = True + elif quote is not None: + quote = None if char == quote else quote + elif char in {"'", '"'}: + quote = char + elif char == "#" and (previous.isspace() or previous in _SHELL_OPERATOR_CHARS): + return True + previous = char + return False + + def _agent_command(words: list[str]) -> tuple[str, list[str]] | None: """The agent CLI one simple command launches headless, and its arguments. @@ -1814,14 +1847,17 @@ def _line_agent(line: str) -> str | None: return (_agent_command(line.split()) or (None, None))[0] -def _read_flags(words: list[str], table: dict[str, tuple[str, int | None]]) -> list[tuple[str, str | None]]: - """The documented permission flags among ``words``, each with its declared value. +def _read_flags( + words: list[str] | tuple[str, ...], table: dict[str, tuple[str, int | None]] +) -> list[tuple[str, int | None, list[str] | None]]: + """The documented permission flags among ``words``: primary spelling, arity and value words. A flag is read by name wherever it is a word of its own; every other word - — the prompt, ``--model`` and any undocumented flag — is not compared. + — the prompt, ``--model`` and any undocumented flag — is not compared. A + flag that takes no value, or is given none, has ``None``. """ - flags: list[tuple[str, str | None]] = [] + flags: list[tuple[str, int | None, list[str] | None]] = [] index = 0 while index < len(words): word = words[index] @@ -1832,7 +1868,7 @@ def _read_flags(words: list[str], table: dict[str, tuple[str, int | None]]) -> l continue primary, arity = spec if arity == 0: - flags.append((primary, None)) + flags.append((primary, arity, None)) continue values = [attached] if equals else [] if arity == 1 and not equals and index < len(words): @@ -1842,36 +1878,392 @@ def _read_flags(words: list[str], table: dict[str, tuple[str, int | None]]) -> l while index < len(words) and not words[index].startswith("-"): values.append(words[index]) index += 1 - if not values: - flags.append((primary, None)) - else: - flags.append((primary, values[0] if arity == 1 else shlex.join(values))) + flags.append((primary, arity, values or None)) return flags -def _published_setting(name: str, value: Any) -> dict[str, Any]: - """One setting as it may be published: declared text, or why it is not.""" +# --- how an agent action splits its argument input (#823) ----------------------------- +# +# `claude_args` and `codex-args` are not shell text: no shell reads them. Each +# action splits its input itself, and a widening rule is read from the words +# the action passes on, so these follow the actions' own parsers rather than +# the `run:` tokenizer above. + +#: One word of shell-quote's chunker once the Claude actions have made +#: ``()|&;<>`` literal: unquoted non-space characters (a backslash escaping a +#: quote or a blank), a double-quoted run or a single-quoted run, adjacent. +#: An unbalanced quote matches none of them, so shell-quote skips it. +_SHELL_QUOTE_CHUNK_RE = re.compile(r"""(?:(?:\\['" \t]|[^\s'"])+|"(?:\\"|[^"])*?"|'[^']*?')+""") + +#: string-argv's pattern, which `openai/codex-action` splits a shell-like +#: `codex-args` with: a word holding quotes keeps them, a quoted string alone +#: is its content, and whitespace, newlines included, separates the rest. +_STRING_ARGV_RE = re.compile( + r"""([^\s'"]([^\s'"]*(['"])([^\x03]*?)\3)+[^\s'"]*)|[^\s'"]+|(['"])([^\x03]*?)\5""" +) - if isinstance(value, bool): - text = "true" if value else "false" - elif value is None: - text = "" - elif isinstance(value, (int, float)): - text = str(value) - elif isinstance(value, str): - text = value.strip() + +def _shell_quote_variable(chunk: str, index: int) -> tuple[str, int]: + """shell-quote's ``parseEnvVar`` with no environment, at the ``$`` at ``index``. + + Returns the value, empty for any name and ``$`` for none, and the index of + the last character the variable took. Raises ``ValueError`` for the "Bad + substitution" shell-quote throws, which fails the action. + """ + + index += 1 + char = chunk[index:index + 1] + if char == "{": + index += 1 + if chunk[index:index + 1] == "}": + raise ValueError("bad substitution") + depth, end = 1, index + while depth > 0 and end < len(chunk): + if chunk[end] == "{" and chunk[end - 1] == "$": + depth += 1 + elif chunk[end] == "}": + depth -= 1 + end += 1 + if depth != 0: + raise ValueError("bad substitution") + name, index = chunk[index:end - 1], end - 1 + elif char and char in "*@#?$!_-": + # shell-quote steps past the name and then past one more character. + name, index = char, index + 1 else: - return {"name": name, "value": None, "unresolved_reason": "not_a_string"} - shown = published_workflow_label(text) - if shown != text: - return {"name": name, "value": None, "unresolved_reason": "redacted"} - return {"name": name, "value": text, "unresolved_reason": None} + match = re.search(r"[^A-Za-z0-9_]", chunk[index:]) + if match is None: + name, index = chunk[index:], len(chunk) + else: + name, index = chunk[index:index + match.start()], index + match.start() - 1 + return ("" if name else "$"), index + + +def _shell_quote_word(chunk: str, *, comments: bool) -> tuple[str, bool]: + """One chunk as shell-quote reads it: the word, and whether an unquoted ``#`` ended the input. + + Quotes and backslashes work as in a shell and ``$NAME`` reads as empty. + With ``comments``, an unquoted ``#`` ends the word and every word after + it, as it does for the action; without, ``#`` is an ordinary character. + """ + + out: list[str] = [] + quote = "" + escaped = False + index = 0 + while index < len(chunk): + char = chunk[index] + if escaped: + out.append(char) + escaped = False + elif quote: + if char == quote: + quote = "" + elif quote == "'": + out.append(char) + elif char == "\\": + index += 1 + following = chunk[index:index + 1] + out.append(following if following and following in "\"\\$" else "\\" + following) + elif char == "$": + value, index = _shell_quote_variable(chunk, index) + out.append(value) + else: + out.append(char) + elif char in {'"', "'"}: + quote = char + elif char == "#" and comments: + return "".join(out), True + elif char == "\\": + escaped = True + elif char == "$": + value, index = _shell_quote_variable(chunk, index) + out.append(value) + else: + out.append(char) + index += 1 + return "".join(out), False + + +@dataclass(frozen=True) +class _ArgumentInput: + """An agent action's argument input as the action splits it (#823). + + ``text`` is what the action parses. ``spans`` is every word with where it + sits in ``text``, for publication. ``words`` is what the action passes on, + or ``None`` when the action refuses the input and the agent does not run. + """ + + text: str + spans: tuple[tuple[str, int, int], ...] + words: tuple[str, ...] | None + + +def _claude_argument_input(value: str) -> _ArgumentInput: + """``claude_args`` split as ``base-action/src/parse-sdk-options.ts`` splits it. + + Each line whose first non-blank character is ``#`` is dropped, ``()|&;<>`` + are literal, and the rest is read by shell-quote with no environment: + whitespace, newlines included, separates words; quotes and backslashes work + as in a shell; ``$NAME`` reads as empty; and an unquoted ``#`` later in the + input ends it. + """ + + text = "\n".join( + line for line in value.split("\n") if not line.strip().startswith("#") + ).strip() + spans: list[tuple[str, int, int]] = [] + words: list[str] | None = [] + ended = False + for match in _SHELL_QUOTE_CHUNK_RE.finditer(text): + chunk = match.group() + if words is not None and not ended: + try: + word, ended = _shell_quote_word(chunk, comments=True) + except ValueError: + words = None + else: + if word or not ended: + words.append(word) + try: + shown = _shell_quote_word(chunk, comments=False)[0] + except ValueError: + shown = chunk + spans.append((shown, match.start(), match.end())) + return _ArgumentInput(text, tuple(spans), None if words is None else tuple(words)) + + +def _codex_argument_input(value: str) -> _ArgumentInput: + """``codex-args`` read as `openai/codex-action` reads it: a JSON array of strings, or string-argv. + + A value starting with ``[`` that is not a JSON array of strings makes the + action refuse it. The array form has no spans: it is published whole. + """ + + if value.startswith("["): + try: + loaded = json.loads(value) + except (ValueError, RecursionError): + return _ArgumentInput(value, (), None) + if isinstance(loaded, list) and all(isinstance(item, str) for item in loaded): + return _ArgumentInput(value, (), tuple(loaded)) + return _ArgumentInput(value, (), None) + spans = tuple( + ( + next(group for group in (match.group(1), match.group(6), match.group(0)) if group is not None), + match.start(), + match.end(), + ) + for match in _STRING_ARGV_RE.finditer(value) + ) + return _ArgumentInput(value, spans, tuple(word for word, _start, _end in spans)) + + +def _argument_input(family: str, value: str) -> _ArgumentInput: + return _claude_argument_input(value) if family == "claude" else _codex_argument_input(value) + + +# --- what an agent setting publishes (#823, #802) -------------------------------------- + +#: A codex ``--config`` override under one of these keys carries values the +#: codex host reader never publishes, as ``.mcp.json`` ``env``/``headers`` do not. +_CONFIG_WITHHELD_KEYS = frozenset({"env", "headers", "http_headers", "env_http_headers"}) + + +def _withheld_json(value: Any) -> str | None: + """A JSON value as the host readers publish one: key names, secret-bearing values replaced. + + ``env`` and ``headers`` keep their keys with every value ````, + ``apiKeyHelper`` and every other secret-named key's value is ````, + and strings go through the host sanitizer (URLs, bearer and header + values), exactly as `.claude/settings.json` and `.mcp.json` are read. + Canonical, so reformatting or reordering keys changes nothing. + """ + + try: + return json.dumps( + _redact_secret_values(value), sort_keys=True, separators=(",", ":"), + ensure_ascii=False, default=str, + ) + except (RecursionError, TypeError, ValueError): + return None + + +def _withheld_word(word: str) -> str | None: + """One word as it may be published: a JSON object by :func:`_withheld_json`, else as written. + + ``None`` for a word that starts like a JSON object and does not parse: what + it holds cannot be told apart, so none of it may be published. + """ + + if not word.lstrip().startswith("{"): + return word + try: + loaded = json.loads(word) + except (ValueError, RecursionError): + return None + return _withheld_json(loaded) + + +def _withheld_config(text: str) -> str | None: + """A codex ``--config key=value`` override, its secret-bearing value withheld. + + A key path through ``env``, ``headers`` or a secret-named key publishes + ```` for its value; a table or array value, parsed as TOML as + codex parses it, publishes by :func:`_withheld_json`. Anything else is kept. + """ + + key, equals, value = text.partition("=") + if not equals: + return text + segments = [segment.strip().strip("\"'") for segment in key.split(".")] + if any(segment in _CONFIG_WITHHELD_KEYS or _is_secret_key(segment) for segment in segments): + return f"{key}=" + try: + loaded = tomllib.loads(f"value = {value}").get("value") + except (tomllib.TOMLDecodeError, RecursionError): + return text + if isinstance(loaded, (dict, list)): + shown = _withheld_json(loaded) + return None if shown is None else f"{key}={shown}" + return text + + +def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> list[str] | None: + """Each argument word as it may be published, or ``None`` when one cannot be.""" + + shown: list[str] = [] + config = False + for word in words: + if config: + item = _withheld_config(word) + elif family == "codex" and word.startswith("--config="): + rest = _withheld_config(word.removeprefix("--config=")) + item = None if rest is None else f"--config={rest}" + else: + item = _withheld_word(word) + if item is None: + return None + shown.append(item) + config = family == "codex" and word in {"-c", "--config"} + return shown + +def _withheld_arguments(family: str, value: str) -> str | None: + """An argument input as it may be published: the text the action parses, JSON words withheld. -def _published_flag(name: str, value: str | None) -> dict[str, Any]: + A word the host readers would not publish is replaced, quoted, by what they + would; every other character stays as declared. A JSON array + ``codex-args`` publishes as its array of withheld words. + """ + + parsed = _argument_input(family, value) + if family == "codex" and value.startswith("["): + if parsed.words is None: + # The action refuses it; what it holds is still withheld as JSON. + try: + return _withheld_json(json.loads(value)) + except (ValueError, RecursionError): + return None + shown = _withheld_words(parsed.words, family=family) + return None if shown is None else json.dumps(shown, separators=(",", ":"), ensure_ascii=False) + shown = _withheld_words([word for word, _start, _end in parsed.spans], family=family) + if shown is None: + return None + pieces: list[str] = [] + cursor = 0 + for (word, start, end), published in zip(parsed.spans, shown, strict=True): + if published != word: + pieces.extend((parsed.text[cursor:start], shlex.quote(published))) + cursor = end + pieces.append(parsed.text[cursor:]) + return "".join(pieces) + + +def _url_withheld(url: str) -> str: + """A URL with its path and query withheld as the host sanitizer withholds them. + + One holding userinfo is kept as written, so the label redaction still + finds the credential in it. + """ + + try: + netloc = urlsplit(url).netloc + except ValueError: + return url + return url if "@" in netloc else _sanitize_url(url) + + +def _published_value(text: str) -> tuple[str, bool]: + """``text`` as it may be published, and whether credential-shaped text had to be redacted. + + A URL's path and query are withheld the way the host sanitizer withholds + them from an MCP server URL (#723), and the rest of the text is published + and compared: a URL path is not a credential. Anything else the #802 label + redaction rewrites — a token shape, a credential assignment, a bearer or + header value, a URL's userinfo — is credential-shaped text: the value is + published redacted, and two values that redact alike cannot be compared + apart, so :func:`_uncompared_workflow_text` makes the workflow a blocking limit. + """ + + withheld = _URL_RE.sub(lambda match: _url_withheld(match.group(0)), text) + if published_workflow_label(withheld) == withheld: + return withheld, False + return published_workflow_label(text), True + + +def _setting_text(value: Any) -> str | None: + """A ``with:`` value as the text GitHub passes, or ``None`` when it is not a scalar.""" + + if isinstance(value, bool): + return "true" if value else "false" if value is None: + return "" + if isinstance(value, (int, float)): + return str(value) + if isinstance(value, str): + return value.strip() + return None + + +def _published_text(name: str, text: str | None) -> dict[str, Any]: + """A setting's withheld text as it may be published; ``None`` text is ``unparsed_json``.""" + + if text is None: + return {"name": name, "value": None, "unresolved_reason": "unparsed_json"} + shown, redacted = _published_value(text) + return {"name": name, "value": shown, "unresolved_reason": "redacted" if redacted else None} + + +def _published_setting(name: str, value: Any, *, arguments: str | None = None) -> dict[str, Any]: + """One action input as it may be published: its text, or why it is not. + + ``arguments`` names the family whose action splits this input + (``claude_args``, ``codex-args``); every other input is one value, a JSON + object published by :func:`_withheld_json`. + """ + + text = _setting_text(value) + if text is None: + return {"name": name, "value": None, "unresolved_reason": "not_a_string"} + return _published_text( + name, _withheld_word(text) if arguments is None else _withheld_arguments(arguments, text) + ) + + +def _published_flag(family: str, name: str, arity: int | None, values: list[str] | None) -> dict[str, Any]: + """One CLI flag as it may be published: its value words, each withheld as an input's are.""" + + if values is None: return {"name": name, "value": None, "unresolved_reason": None} - return _published_setting(name, value) + if family == "codex" and name == "--config": + overrides = [_withheld_config(value) for value in values] + shown = None if None in overrides else [str(value) for value in overrides] + else: + shown = _withheld_words(values, family=family) + if shown is None: + return _published_text(name, None) + return _published_text(name, shown[0] if arity == 1 else shlex.join(shown)) def _setting_key(setting: dict[str, Any]) -> tuple[str, str, str]: @@ -1919,6 +2311,12 @@ def _action_identity(uses: Any) -> str | None: def _action_launch(job: str, step_label: str, agent: str, step: dict[Any, Any]) -> dict[str, Any]: + """A documented agent action's launch: its documented inputs as published, and the rules they meet. + + The rules are read from each input's declared text before any of it is + withheld for publication, so what redaction hides never hides a rule. + """ + spec = _AGENT_ACTIONS[agent] entry: dict[str, Any] = { "job": job, "step": step_label, "agent": agent, "form": "read", @@ -1930,12 +2328,18 @@ def _action_launch(job: str, step_label: str, agent: str, step: dict[Any, Any]) if not isinstance(inputs, dict): return {**entry, "form": "unresolved", "unresolved_reason": "inputs_not_a_mapping"} wanted = {name.casefold(): name for name in spec.inputs} - settings = [ - _published_setting(wanted[str(key).casefold()], value) + declared = [ + (wanted[str(key).casefold()], value) for key, value in inputs.items() if str(key).casefold() in wanted ] - return {**entry, "settings": sorted(settings, key=_setting_key)} + settings = [ + _published_setting(name, value, arguments=spec.family if name == spec.args else None) + for name, value in declared + ] + return _with_rules( + {**entry, "settings": sorted(settings, key=_setting_key)}, _action_rules(spec, declared) + ) def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: @@ -1972,8 +2376,12 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: launches = [launch for command in commands if (launch := _agent_command(command))] if not launches: return [] - single = len([command for command in commands if command]) == 1 and not any( - _is_operator(word) or word.startswith("#") for word in words + # A comment is an unquoted `#` at the start of a word, read off the raw + # text: the splitter keeps `#` literal, so a quoted "#123 review" is one word. + single = ( + len([command for command in commands if command]) == 1 + and not any(_is_operator(word) for word in words) + and not _has_shell_comment(run) ) if "${{" in run: reason: str | None = "expression" @@ -1990,11 +2398,15 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: for agent in agents ] (agent, arguments), = launches - settings = [_published_flag(name, value) for name, value in _read_flags(arguments, _AGENT_FLAG_TABLES[agent])] - return [{ - **base, "agent": agent, "form": "read", "unresolved_reason": None, - "settings": sorted(settings, key=_setting_key), - }] + flags = _read_flags(arguments, _AGENT_FLAG_TABLES[agent]) + settings = [_published_flag(agent, name, arity, values) for name, arity, values in flags] + return [_with_rules( + { + **base, "agent": agent, "form": "read", "unresolved_reason": None, + "settings": sorted(settings, key=_setting_key), + }, + _flag_rules(agent, flags), + )] def _step_agent_launches(job: str, step: dict[Any, Any], index: int) -> list[dict[str, Any]]: @@ -2026,14 +2438,21 @@ def _checkout_ref(job: str, step: dict[Any, Any], index: int) -> dict[str, Any] return {**entry, "unresolved_reason": "inputs_not_a_mapping"} if "ref" not in inputs: return entry - setting = _published_setting("ref", inputs["ref"]) - if setting["unresolved_reason"] is not None: - return {**entry, "unresolved_reason": setting["unresolved_reason"]} - return {**entry, "ref": setting["value"] or None} + text = _setting_text(inputs["ref"]) + if text is None: + return {**entry, "unresolved_reason": "not_a_string"} + shown, redacted = _published_value(text) + # A redacted ref is published redacted, as a step reference is, and makes + # the workflow a blocking limit (#767): two refs may redact alike. + return {**entry, "ref": shown or None, "unresolved_reason": "redacted" if redacted else None} def agent_launch_key(entry: dict[str, Any]) -> tuple[Any, ...]: - """What an agent launch is compared by: its job, agent, form and settings, not its step.""" + """What an agent launch is compared by: its job, agent, form, settings and rules, not its step. + + The widening rules are part of it: they are read from the declared text, + so a rule gained where redaction withholds the text is still a change. + """ return ( str(entry["job"]), @@ -2041,6 +2460,9 @@ def agent_launch_key(entry: dict[str, Any]) -> tuple[Any, ...]: str(entry["form"]), str(entry.get("unresolved_reason") or ""), tuple(sorted(_setting_key(setting) for setting in entry.get("settings", []))), + tuple(sorted( + (str(item["rule"]), str(item["setting"])) for item in entry.get("widening_rules", []) + )), ) @@ -2054,79 +2476,104 @@ def checkout_ref_key(entry: dict[str, Any]) -> tuple[str, str, str]: ) -def _argument_words(value: str, *, json_list: bool) -> list[str] | None: - """An arguments input's words, or ``None`` when they cannot be read literally.""" - - if "${{" in value: - return None - if json_list and value.startswith("["): - try: - loaded = json.loads(value) - except json.JSONDecodeError: - return None - if isinstance(loaded, list) and all(isinstance(item, str) for item in loaded): - return loaded - return None - words = _shell_words(value) - if words is None or any(_is_operator(word) for word in words): - return None - return words +def _claude_action_rules(words: tuple[str, ...]) -> set[str]: + """The widening rules the words a Claude action passes on meet. + Read as ``parse-sdk-options.ts`` reads them: a word starting with ``--`` + is always a flag and never another flag's value, so + ``--dangerously-skip-permissions`` counts wherever it stands, and + ``--permission-mode`` takes the next word unless that starts with ``--``. + """ -def _flag_rules(family: str, flags: list[tuple[str, str | None]]) -> set[str]: rules: set[str] = set() - for name, value in flags: + for index, word in enumerate(words): + name, equals, attached = word.partition("=") + if name == "--dangerously-skip-permissions": + rules.add("bypass_permissions") + elif name == "--permission-mode": + following = words[index + 1] if index + 1 < len(words) else "" + mode = attached if equals else ("" if following.startswith("--") else following) + if mode == "bypassPermissions": + rules.add("bypass_permissions") + return rules + + +def _flag_rules( + family: str, flags: list[tuple[str, int | None, list[str] | None]] +) -> set[tuple[str, str]]: + """The widening rules a CLI's documented flags meet, each with the flag that met it.""" + + rules: set[tuple[str, str]] = set() + for name, _arity, values in flags: + value = values[0] if values else None if family == "claude" and ( name == "--dangerously-skip-permissions" or (name == "--permission-mode" and value == "bypassPermissions") ): - rules.add("bypass_permissions") + rules.add(("bypass_permissions", name)) if family == "codex" and name == "--dangerously-bypass-approvals-and-sandbox": - rules.add("bypass_approvals_and_sandbox") + rules.add(("bypass_approvals_and_sandbox", name)) if family == "codex" and name == "--sandbox" and value == "danger-full-access": - rules.add("danger_full_access") + rules.add(("danger_full_access", name)) + return rules + + +def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tuple[str, str]]: + """The documented widening rules an agent action's declared inputs meet, read from the raw text. + + Read here, before anything is withheld for publication, so redaction never + hides a rule. Only literal values: GitHub substitutes a ``${{ }}`` + expression into the input before the action reads it, so what it holds is + not known here and it meets no rule. Each rule names the input it was read from. + """ + + literal = [ + (name, text) for name, value in declared + if (text := _setting_text(value)) is not None and "${{" not in text + ] + rules: set[tuple[str, str]] = set() + for name, text in literal: + if name in spec.gates and "*" in {part.strip() for part in text.split(",")}: + rules.add(("open_gate", name)) + for mode, widening in spec.modes: + if name == mode and text == widening: + rules.add(("danger_full_access" if mode == "sandbox" else "unsafe_safety_strategy", name)) + if name == spec.args: + words = _argument_input(spec.family, text).words + if words is None: + continue + if spec.family == "claude": + found = _claude_action_rules(words) + else: + found = {rule for rule, _flag in _flag_rules("codex", _read_flags(words, _CODEX_FLAGS))} + rules.update((rule, name) for rule in found) return rules +def _with_rules(entry: dict[str, Any], rules: set[tuple[str, str]]) -> dict[str, Any]: + """``entry`` with the rules it meets, omitted when none, as the schema omits them.""" + + if not rules: + return entry + return {**entry, "widening_rules": [{"rule": rule, "setting": setting} for rule, setting in sorted(rules)]} + + def agent_widening_rules(entry: dict[str, Any]) -> set[tuple[str, str]]: """The documented widening rules one published agent launch meets (#823). Each is ``(rule, detail)``: ``detail`` names the input for an opened gate - (``allowed_non_write_users``) and is empty otherwise. Read from published - settings only, so a redacted value, a value holding an expression, an - unresolved launch and an unknown flag meet no rule. + (``allowed_non_write_users``) and is empty otherwise, so one rule spelled + two ways, or moved between the CLI and an action, is one rule. Read from + ``widening_rules``, which the reader decided from the declared text before + any of it was withheld; an unresolved launch meets none. """ if entry.get("form") != "read": return set() - agent = str(entry["agent"]) - # Only literal values: GitHub substitutes a `${{ }}` expression into the - # input before the action reads it, so what it holds is not known here. - readable = { - str(setting["name"]): str(setting["value"]) - for setting in entry.get("settings", []) - if setting.get("unresolved_reason") is None and setting.get("value") is not None - and "${{" not in str(setting["value"]) + return { + (str(item["rule"]), str(item["setting"]) if item["rule"] == "open_gate" else "") + for item in entry.get("widening_rules", []) } - rules: set[tuple[str, str]] = set() - if agent in {"claude", "codex"}: - flags = [(str(setting["name"]), setting.get("value")) for setting in entry.get("settings", []) - if setting.get("unresolved_reason") is None] - rules.update((rule, "") for rule in _flag_rules(agent, flags)) - return rules - spec = _AGENT_ACTIONS[agent] - for name in spec.gates: - if name in readable and "*" in {part.strip() for part in readable[name].split(",")}: - rules.add(("open_gate", name)) - for name, value in spec.modes: - if readable.get(name) == value: - rules.add(("danger_full_access" if name == "sandbox" else "unsafe_safety_strategy", "")) - if spec.args and spec.args in readable: - words = _argument_words(readable[spec.args], json_list=spec.family == "codex") - if words is not None: - flags = _read_flags(words, _AGENT_FLAG_TABLES[spec.family]) - rules.update((rule, "") for rule in _flag_rules(spec.family, flags)) - return rules def agent_family(agent: str) -> str: @@ -2135,16 +2582,20 @@ def agent_family(agent: str) -> str: return _AGENT_ACTIONS[agent].family if agent in _AGENT_ACTIONS else agent -def gained_agent_widenings( - before: dict[str, Any] | None, after: dict[str, Any] | None -) -> list[tuple[str, str, str, dict[str, Any]]]: - """Documented widening rules a job's agent launches meet at ``after`` and not at ``before``. +AgentWidening = tuple[str, str, str, dict[str, Any]] - Keyed by job, agent family and rule, so moving a launch between steps or - spellings (``--dangerously-skip-permissions`` and - ``--permission-mode bypassPermissions`` are one rule) gains nothing. Each - result is ``(job, rule, detail, entry)`` for the first launch at ``after`` - that meets it. + +def _agent_rule_gains( + before: dict[str, Any] | None, after: dict[str, Any] | None +) -> tuple[list[AgentWidening], list[AgentWidening]]: + """Rules a job's launches meet at ``after`` and not at ``before``: claimed, and not claimed. + + A gain is not claimed where the job launched that agent at ``before`` only + in a form this reader does not read, such as a compound ``run:`` that + became a literal one: that launch may have met the rule already, as a job + whose permissions were not explicit may already have held a write scope + (``unknown_before``). A job that also launched the agent in a form that + was read claims the gain. """ def met(grant: dict[str, Any] | None) -> dict[tuple[str, str, str, str], dict[str, Any]]: @@ -2155,10 +2606,41 @@ def met(grant: dict[str, Any] | None) -> dict[tuple[str, str, str, str], dict[st found.setdefault(key, entry) return found + read_before: dict[tuple[str, str], bool] = {} + for entry in (before or {}).get("agent_launches", []): + key = (str(entry["job"]), agent_family(str(entry["agent"]))) + read_before[key] = read_before.get(key, False) or entry.get("form") == "read" + unread = {key for key, read in read_before.items() if not read} old = met(before) - return [ - (key[0], key[2], key[3], entry) for key, entry in met(after).items() if key not in old - ] + claimed: list[AgentWidening] = [] + unclaimed: list[AgentWidening] = [] + for key, entry in met(after).items(): + if key in old: + continue + (unclaimed if key[:2] in unread else claimed).append((key[0], key[2], key[3], entry)) + return claimed, unclaimed + + +def gained_agent_widenings(before: dict[str, Any] | None, after: dict[str, Any] | None) -> list[AgentWidening]: + """Documented widening rules a job's agent launches meet at ``after`` and not at ``before``. + + Keyed by job, agent family and rule, so moving a launch between steps or + spellings (``--dangerously-skip-permissions`` and + ``--permission-mode bypassPermissions`` are one rule) gains nothing, and a + rule the job's unread launch at ``before`` may already have met is not + claimed. Each result is ``(job, rule, detail, entry)`` for the first + launch at ``after`` that meets it. + """ + + return _agent_rule_gains(before, after)[0] + + +def agent_widenings_unread_before( + before: dict[str, Any] | None, after: dict[str, Any] | None +) -> list[AgentWidening]: + """Rules a job's launches now meet that are not claimed, because its launch before was unread.""" + + return _agent_rule_gains(before, after)[1] #: How an unresolved agent launch or checkout reads in the limit that names it. @@ -2174,8 +2656,11 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: """One message per agent launch setting or checkout ref a workflow does not compare (#823). Not blocking, like an unread secret value (#693): the launch's job, agent, - form and reason are still compared, so adding, removing or re-forming one - is a row. Only an edit inside what is named here is not reported. + form, reason and widening rules are still compared, so adding, removing or + re-forming one, or gaining a documented rule, is a row. Only an edit inside + what is named here is not reported. A redacted value is not named here: it + is published redacted, and :func:`_uncompared_workflow_text` makes it a + blocking limit, as a redacted step reference is (#767). """ texts: list[str] = [] @@ -2190,19 +2675,26 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: ) for setting in entry.get("settings", []): unread = setting.get("unresolved_reason") - if unread: - what = "contains credential-shaped text" if unread == "redacted" else "is not a string" + what = { + "not_a_string": "is not a string", + "unparsed_json": ( + "holds text that starts like JSON and does not parse, so the " + "values it may hold cannot be told apart from its key names" + ), + }.get(str(unread)) + if what: texts.append( f"the {setting['name']} value of the agent launch at {where} {what}; it is " - "neither published nor compared, so an edit to it is not reported" + "neither published nor compared, so an edit to it that gains no documented " + "widening rule is not reported" ) for entry in grant.get("checkout_refs", []): unread = entry.get("unresolved_reason") - if unread: - what = { - "redacted": "a ref that contains credential-shaped text", - "not_a_string": "a ref that is not a string", - }.get(str(unread), "a `with:` that is not a mapping") + what = { + "not_a_string": "a ref that is not a string", + "inputs_not_a_mapping": "a `with:` that is not a mapping", + }.get(str(unread)) + if what: texts.append( f"the checkout at {entry['job']}/{entry['step']} declares {what}; it is " "neither published nor compared, so an edit to it is not reported" @@ -2335,10 +2827,12 @@ def _uncompared_workflow_text( """Why part of a workflow grant is published but cannot be compared, or ``None``. One rule for every compared workflow text (#767, #693). A redacted step - reference, reusable target, or secret name could publish the same text as - a different one, so comparing the display would read a change as equal. - That is a blocking limit: a changed workflow refuses, and an unchanged one - is named (#721). + reference, reusable target, secret name, agent launch setting or checkout + ref (#823) could publish the same text as a different one, so comparing + the display would read a change as equal. That is a blocking limit: a + changed workflow refuses, and an unchanged one is named (#721). A URL path + an agent setting withholds is not credential-shaped and is not counted + here, as an MCP server URL's path is not (#723). A job id, trigger or permission scope name is compared by its published label (#802). One redacted label is still a distinct label, so it refuses @@ -2358,6 +2852,14 @@ def _uncompared_workflow_text( ("a reusable workflow secret name", any( entry["unresolved_reason"] == "redacted" for entry in mappings )), + # An agent setting and a checkout ref are compared text too (#823). + ("an agent launch setting", any( + setting.get("unresolved_reason") == "redacted" + for launch in grant.get("agent_launches", []) for setting in launch.get("settings", []) + )), + ("a checkout ref", any( + item.get("unresolved_reason") == "redacted" for item in grant.get("checkout_refs", []) + )), ) if present ] merged = [kind for kind in _LABEL_KINDS if kind in collided] diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index f23c41481..0782299da 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -626,28 +626,66 @@ class HostWorkflowAgentSettingV7(BaseModel): ``name`` is the documented input (``claude_args``, ``sandbox``, …) or the flag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too). - ``value`` is the declared text, stripped, published through the workflow - label redaction (#802); a flag that takes no value has ``null``. A value - the redaction rewrites, or one that is not a string, is ``null`` with - ``unresolved_reason``, and records a non-blocking coverage issue naming - its ``job/step``: it is neither published nor compared. + ``value`` is the declared text, stripped, as it may be published; a flag + that takes no value has ``null``. ``claude_args`` is the text the Claude + actions parse, without the full-line ``#`` comments they drop. What the + host readers withhold stays withheld: a JSON object (a ``settings`` or + ``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any + argument word) publishes its key names with ``env`` and ``headers`` + values, ``apiKeyHelper`` and every secret-named value ````, a + codex ``--config`` override under such a key publishes ````, and + a URL publishes its scheme and host with ```` for its path + and query (#723). The rest is published through the workflow label + redaction (#802). A value it rewrites beyond that is credential-shaped: it + is published redacted with ``unresolved_reason: redacted`` and makes the + workflow a blocking limit, as a redacted step reference does (#767). A + value that is not a string (``not_a_string``), or one holding text that + starts like JSON and does not parse (``unparsed_json``), is + ``null`` and records a non-blocking coverage issue naming its + ``job/step``: it is neither published nor compared. """ model_config = ConfigDict(extra="forbid") name: str value: str | None - unresolved_reason: Literal["not_a_string", "redacted"] | None = None + unresolved_reason: Literal["not_a_string", "redacted", "unparsed_json"] | None = None + + +class HostWorkflowAgentRuleV7(BaseModel): + """One documented widening rule an agent launch meets, and the setting it was read from (#823). + + Decided when the workflow is read, from the declared text, before any of + it is withheld for publication, so redaction never hides a rule. Only a + literal value meets one: a value holding ``${{ }}`` meets none. + ``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``, + …) or the CLI flag's primary spelling. One rule compares as one whatever + setting meets it, except ``open_gate``, which is one rule per gate input. + """ + + model_config = ConfigDict(extra="forbid") + + rule: Literal[ + "bypass_permissions", + "bypass_approvals_and_sandbox", + "danger_full_access", + "unsafe_safety_strategy", + "open_gate", + ] + setting: str class HostWorkflowAgentLaunchV7(BaseModel): """A step that launches a known coding agent, read as text and never run (#823). ``agent`` is a documented action reference's ``owner/repo`` (the step's - ``uses:`` at any ref) or a known agent CLI a literal ``run:`` starts with: - ``claude`` with ``-p``/``--print``, or ``codex exec``. ``form: read`` - lists the documented permission inputs or flags the step declares in - ``settings``. ``form: unresolved`` names why the step's settings were not + ``uses:`` at any ref; the Claude base action also as the ``base-action`` + directory of ``anthropics/claude-code-action``) or a known agent CLI a + literal ``run:`` starts with: ``claude`` with ``-p``/``--print``, or + ``codex exec``. ``form: read`` lists the documented permission inputs or + flags the step declares in ``settings``, and the documented widening + rules they meet in ``widening_rules``, omitted when none. + ``form: unresolved`` names why the step's settings were not read — a ``run:`` holding more than one command or quoting that does not balance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is not a mapping — with no @@ -664,6 +702,7 @@ class HostWorkflowAgentLaunchV7(BaseModel): agent: Literal[ "anthropics/claude-code-action", "anthropics/claude-code-base-action", + "anthropics/claude-code-action/base-action", "openai/codex-action", "claude", "codex", @@ -676,6 +715,9 @@ class HostWorkflowAgentLaunchV7(BaseModel): "inputs_not_a_mapping", ] | None = None settings: list[HostWorkflowAgentSettingV7] = Field(default_factory=list) + widening_rules: list[HostWorkflowAgentRuleV7] = Field( + default_factory=list, exclude_if=lambda value: not value, + ) job_secrets: list[str] = Field(default_factory=list, exclude_if=lambda value: not value) @@ -684,7 +726,9 @@ class HostWorkflowCheckoutRefV7(BaseModel): ``ref`` is ``null`` when the step declares none, or an empty one: the checkout's default for the triggering event. A ref the label redaction - rewrites, a value that is not a string, or ``with:`` that is not a mapping + rewrites is published redacted with ``unresolved_reason: redacted`` and + makes the workflow a blocking limit, as a redacted step reference does + (#767). A value that is not a string, or ``with:`` that is not a mapping, is ``null`` with ``unresolved_reason`` and records a non-blocking coverage issue. The ref is never resolved or fetched. """ @@ -761,4 +805,4 @@ class HostGrantsDriftArtifactV7(RootModel[HostGrantsDriftV7]): root: HostGrantsDriftV7 -__all__ =[name for name in globals() if name.startswith("Host") or name.startswith("HOST_")] +__all__ = [name for name in globals() if name.startswith("Host") or name.startswith("HOST_")] diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index a78a02f51..98684db06 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -229,7 +229,8 @@ def paths(self) -> list[Path]: # `tests/test_partial_host_comparison.py`. # Its agent-launch cells and note (#823) # restate no answer: direction comes from the engine's - # `workflow_agent_widened_*` expansion signal, and the note reads the + # `workflow_agent_widened_*` expansion signal, itself read off the + # `widening_rules` the engine published on each launch, and the note reads the # triggers, write scopes, secrets and checkout refs the engine already # published on the grant, so it adds no claim; # `tests/test_workflow_agent_launches.py` holds diff, verify, the PR diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index c4eff3ac7..6f34aa5b0 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -13,6 +13,7 @@ from __future__ import annotations +import hashlib import json import subprocess from pathlib import Path @@ -24,6 +25,9 @@ from agents_shipgate.cli.main import app from agents_shipgate.core.capability_diff_rows import capability_diff_rows from agents_shipgate.core.host_grants import ( + _claude_argument_input, + _codex_argument_input, + _uncompared_workflow_text, _workflow_grant, diff_host_grants, host_grant_expansion_signals, @@ -265,6 +269,29 @@ def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(r assert "not reported" in limit +def test_a_quoted_word_that_starts_with_a_hash_is_not_a_comment(): + launch, = _launches(_workflow({"run": 'claude -p --allowedTools Read "#123 review"'})) + assert launch["form"] == "read" + assert launch["settings"] == [{"name": "--allowedTools", "value": "Read '#123 review'", "unresolved_reason": None}] + + commented, = _launches(_workflow({"run": "claude -p --allowedTools Read # review"})) + assert (commented["form"], commented["unresolved_reason"]) == ("unresolved", "compound_command") + + +def test_the_base_action_directory_of_the_claude_action_is_read_as_the_base_action(): + launch, = _launches(_workflow({ + "uses": "anthropics/claude-code-action/base-action@v1", + "with": {"claude_args": "--dangerously-skip-permissions", "allowed_bots": "*"}, + })) + + assert (launch["agent"], launch["form"]) == ("anthropics/claude-code-action/base-action", "read") + # `allowed_bots` is not a base-action input, so it is neither listed nor a rule. + assert launch["settings"] == [ + {"name": "claude_args", "value": "--dangerously-skip-permissions", "unresolved_reason": None}, + ] + assert launch["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] + + def test_single_quoted_dollars_are_literal_and_do_not_stop_the_read(): launch, = _launches(_workflow({"run": "claude -p --allowedTools 'Bash(echo $HOME)' 'go'"})) assert launch["form"] == "read" @@ -338,6 +365,129 @@ def test_job_secrets_name_what_the_agent_job_and_the_workflow_env_reference(): assert launch["job_secrets"] == ["ANTHROPIC_API_KEY", "DEPLOY_KEY", "REVIEW_TOKEN", "WORKFLOW_ENV"] +# --- how an agent action splits its argument input (#823 review cycle 1) ----------------- +# +# `claude_args` and `codex-args` are not shell text. The Claude actions split +# `claude_args` with shell-quote after dropping full `#` lines and making +# `()|&;<>` literal (base-action/src/parse-sdk-options.ts); `openai/codex-action` +# reads `codex-args` as a JSON array of strings or with string-argv. A widening +# rule is read from the words the action passes on. + + +@pytest.mark.parametrize( + ("before", "after"), + [ + ("--max-turns 5\n--allowedTools Read", "--max-turns 5\n--dangerously-skip-permissions"), + ("--allowedTools Bash(git:*)", "--allowedTools Bash(git:*) --dangerously-skip-permissions"), + ("# review agent\n--max-turns 5", "# review agent\n--dangerously-skip-permissions"), + ("--max-turns 5", "--max-turns 5\n--permission-mode\nbypassPermissions"), + # A word starting with `--` is always a flag to the action, never a value. + ("--allowedTools Read", "--settings --dangerously-skip-permissions"), + ], + ids=["several-lines", "unquoted-parentheses", "comment-line", "mode-over-lines", "never-a-value"], +) +def test_claude_args_are_split_as_the_claude_actions_split_them(before, after): + changes = _changes(_workflow(_agent(before)), _workflow(_agent(after))) + assert host_grant_expansion_signals(changes) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(_workflow(_agent(before)), _workflow(_agent(after))) + + assert (row.direction, row.expands) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[0])" in row.why + assert "not counted as a widening" not in row.why + + +@pytest.mark.parametrize( + "after", + [ + "# --dangerously-skip-permissions\n--max-turns 5", + " # an indented comment line is dropped too\n--max-turns 5", + # An unquoted `#` later in the input ends it, as shell-quote reads it. + "--max-turns 5 # --dangerously-skip-permissions", + "--max-turns 5 notes#--dangerously-skip-permissions", + ], + ids=["comment-line", "indented-comment", "inline-comment", "hash-in-a-word"], +) +def test_a_flag_the_action_drops_as_a_comment_meets_no_rule(after): + launch, = _launches(_workflow(_agent(after))) + assert "widening_rules" not in launch + assert host_grant_expansion_signals(_changes(_workflow(_agent("--max-turns 5")), _workflow(_agent(after)))) == [] + + +def test_editing_only_a_comment_line_the_action_drops_is_quiet(): + before = _workflow(_agent("# reviewer: alice\n--max-turns 5")) + after = _workflow(_agent("# reviewer: bob\n# --dangerously-skip-permissions\n--max-turns 5")) + + assert _launches(after)[0]["settings"] == [ + {"name": "claude_args", "value": "--max-turns 5", "unresolved_reason": None}, + ] + assert _rows(before, after) == [] + + +@pytest.mark.parametrize( + ("value", "words"), + [ + ('--allowedTools "Bash(git status)" \'Read\'', ["--allowedTools", "Bash(git status)", "Read"]), + ("--allowedTools Bash(gh:*)|Read;x", ["--allowedTools", "Bash(gh:*)|Read;x"]), + ('cost$5 "$HOME/x" $', ["cost", "/x", "$"]), + ('--append-system-prompt "unbalanced --dangerously-skip-permissions', + ["--append-system-prompt", "unbalanced", "--dangerously-skip-permissions"]), + ('a\\ b "c\\"d"', ["a b", 'c"d']), + ("--x a#b --y", ["--x", "a"]), + ("--x ${}", None), + ], + ids=["quotes", "metacharacters", "variables", "unbalanced-quote", "escapes", "hash", "bad-substitution"], +) +def test_the_claude_args_splitter_reads_as_shell_quote_does(value, words): + assert _claude_argument_input(value).words == (None if words is None else tuple(words)) + + +@pytest.mark.parametrize( + ("codex_args", "rule"), + [ + ("--json\n--dangerously-bypass-approvals-and-sandbox", "bypasses approvals and the sandbox"), + ("--full-auto --yolo", "bypasses approvals and the sandbox"), + ('["--json", "--yolo"]', "bypasses approvals and the sandbox"), + ("-s 'danger-full-access'", "runs without a sandbox (danger-full-access)"), + ("--json\n--sandbox=danger-full-access", "runs without a sandbox (danger-full-access)"), + ], + ids=["several-lines", "yolo", "json-array", "quoted-short-sandbox", "attached-sandbox"], +) +def test_codex_args_are_read_as_the_codex_action_reads_them(codex_args, rule): + before = _workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": "--json"}}) + after = _workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}) + + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert rule in row.why + + +@pytest.mark.parametrize( + ("value", "words"), + [ + ("--json\n--full-auto", ["--json", "--full-auto"]), + ("--flag=\"a b\" 'c d' e", ['--flag="a b"', "c d", "e"]), + ('["-c", "x=1"]', ["-c", "x=1"]), + ("[1, 2]", None), + ("[not json", None), + ], + ids=["lines", "string-argv-quotes", "json-array", "not-strings", "invalid-json"], +) +def test_the_codex_args_reader_reads_as_the_codex_action_does(value, words): + assert _codex_argument_input(value).words == (None if words is None else tuple(words)) + + +def test_the_multi_line_widening_reaches_diff_and_the_review_summary(tmp_path): + repo = _repo(tmp_path, {SOURCE: _yaml(_workflow(_agent("--max-turns 5\n--allowedTools Read")))}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(_workflow(_agent("--max-turns 5\n--dangerously-skip-permissions")))}) + _git(repo, "commit", "-qam", "bypass on its own line") + + payload = _diff(repo) + row, = payload["rows"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert payload["review"]["summary"]["widenings"] == 1 + + # --- what is compared -------------------------------------------------------------- @@ -462,6 +612,29 @@ def test_any_other_edit_is_changed(before, after): assert (row.direction, row.expands) == ("changed", False) +def test_a_rule_gained_where_the_job_launched_the_agent_only_unread_before_is_not_claimed(): + before = _workflow({"run": "npm ci && claude -p --dangerously-skip-permissions 'review'"}) + after = _workflow({"run": "claude -p --dangerously-skip-permissions 'review'"}) + + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert ( + "an agent launch now skips permission checks (bypassPermissions) (review/steps[0]), which is not " + "counted as a widening: before, this job launched the agent in a form this audit does not read" + ) in row.why + + +def test_a_rule_gained_by_a_launch_read_on_both_sides_widens_beside_an_unread_one(): + unread = {"run": "npm ci && claude -p 'x'"} + before = _workflow(unread, _agent()) + after = _workflow(unread, _agent("--dangerously-skip-permissions")) + + assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + + def test_a_new_workflow_that_bypasses_permissions_is_an_added_widening(): after = _grant(_workflow(_agent("--dangerously-skip-permissions"))) changes = diff_host_grants({"grants": []}, {"grants": [after]}) @@ -530,26 +703,165 @@ def test_a_removed_workflow_gets_no_note(): # --- redaction (#802) -------------------------------------------------------------- -def test_a_credential_in_an_input_value_is_neither_published_nor_compared(): - value = '--mcp-config \'{"headers":{"Authorization":"Bearer sk-CANARYTOKEN123"}}\'' +SETTINGS_JSON = json.dumps({ + "env": {"DB_PASSWORD": "hunter2-canary", "INTERNAL_KEY": "canary-9f8e7d"}, + "apiKeyHelper": "echo canary-helper-value", + "permissions": {"allow": ["Bash(npm test)"]}, +}) +MCP_JSON = json.dumps({"mcpServers": {"db": { + "command": "db-mcp", "env": {"DB_API_TOKEN": "canary-tok-123"}, "headers": {"X-API-Key": "canary-hdr-456"}, +}}}) +#: What the host readers publish for them: key names, every such value withheld. +SETTINGS_PUBLISHED = ( + '{"apiKeyHelper":"","env":{"DB_PASSWORD":"","INTERNAL_KEY":""},' + '"permissions":{"allow":["Bash(npm test)"]}}' +) +MCP_PUBLISHED = ( + '{"mcpServers":{"db":{"command":"db-mcp","env":{"DB_API_TOKEN":""},' + '"headers":{"X-API-Key":""}}}}' +) +JSON_CANARIES = ("hunter2-canary", "canary-9f8e7d", "canary-helper-value", "canary-tok-123", "canary-hdr-456") + + +def test_a_json_value_publishes_only_what_the_host_readers_publish(): + grant = _grant(_workflow( + _agent(f"--mcp-config '{MCP_JSON}' --allowedTools Read", settings=SETTINGS_JSON, mcp_config=MCP_JSON), + {"run": f"claude -p --mcp-config '{MCP_JSON}' --settings '{SETTINGS_JSON}' 'go'"}, + )) + action, cli = grant["agent_launches"] + + assert {item["name"]: item["value"] for item in action["settings"]} == { + "claude_args": f"--mcp-config '{MCP_PUBLISHED}' --allowedTools Read", + "mcp_config": MCP_PUBLISHED, + "settings": SETTINGS_PUBLISHED, + } + assert {item["name"]: item["value"] for item in cli["settings"]} == { + "--mcp-config": f"'{MCP_PUBLISHED}'", + "--settings": SETTINGS_PUBLISHED, + } + text = json.dumps(grant) + for canary in JSON_CANARIES: + assert canary not in text + assert hashlib.sha256(canary.encode()).hexdigest() not in text + assert uncompared_agent_launch_texts(grant) == [] and _uncompared_workflow_text(grant) is None + + +def test_a_withheld_json_value_compares_as_the_host_readers_compare_it(): + def settings(env): + return _workflow(_agent(settings=json.dumps({"env": env}))) + + # An env value is not compared, as in `.claude/settings.json`; an added key is. + assert _rows(settings({"DB": "one"}), settings({"DB": "two"})) == [] + row, = _rows(settings({"DB": "one"}), settings({"DB": "one", "EXTRA": "three"})) + assert '"EXTRA":""' in row.after + assert "three" not in row.after + + +def test_a_codex_config_override_withholds_env_header_and_secret_values(): + run = ( + "codex exec -c 'mcp_servers.db.env.TOKEN=\"canary-cfg\"' " + "-c 'mcp_servers.gh={command=\"gh\", env={GH_TOKEN=\"canary-inline\"}}' -c model=o3 'go'" + ) + action_args = '["-c", "mcp_servers.db.env.TOKEN=\\"canary-array\\"", "--yolo"]' + grant = _grant(_workflow( + {"run": run}, {"uses": "openai/codex-action@v1", "with": {"codex-args": action_args}}, + )) + cli, action = grant["agent_launches"] + + assert cli["settings"] == [ + {"name": "--config", "value": "mcp_servers.db.env.TOKEN=", "unresolved_reason": None}, + {"name": "--config", "value": 'mcp_servers.gh={"command":"gh","env":{"GH_TOKEN":""}}', + "unresolved_reason": None}, + {"name": "--config", "value": "model=o3", "unresolved_reason": None}, + ] + assert action["settings"] == [{ + "name": "codex-args", "value": '["-c","mcp_servers.db.env.TOKEN=","--yolo"]', + "unresolved_reason": None, + }] + assert action["widening_rules"] == [{"rule": "bypass_approvals_and_sandbox", "setting": "codex-args"}] + assert "canary" not in json.dumps(grant) + + +def test_text_that_starts_like_json_and_does_not_parse_is_withheld_and_named(): + # shell-quote strips the double quotes of an unquoted JSON word, so the + # action reads it as a path; its values cannot be told from its keys. + value = '--mcp-config {"mcpServers":{"db":{"env":{"T":"canary-unquoted"}}}} --dangerously-skip-permissions' grant = _grant(_workflow(_agent(value))) launch, = grant["agent_launches"] - assert launch["settings"] == [{"name": "claude_args", "value": None, "unresolved_reason": "redacted"}] - assert "CANARY" not in json.dumps(grant) + assert launch["settings"] == [{"name": "claude_args", "value": None, "unresolved_reason": "unparsed_json"}] + # The rule is read from the declared text, so withholding it hides no rule. + assert launch["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] + assert "canary-unquoted" not in json.dumps(grant) limit, = uncompared_agent_launch_texts(grant) - assert "claude_args value of the agent launch at review/steps[0]" in limit - assert "credential-shaped" in limit + assert limit.startswith("the claude_args value of the agent launch at review/steps[0] (anthropics/claude-code-action)") + assert "starts like JSON and does not parse" in limit + assert _uncompared_workflow_text(grant) is None -def test_a_token_in_a_run_line_is_never_published(): - run = "claude -p --settings ghp_" + "A" * 36 + " 'review token=SECRETCANARY'" - grant = _grant(_workflow({"run": run})) - launch, = grant["agent_launches"] +def test_a_url_path_is_withheld_while_the_rest_of_the_setting_and_a_rule_beside_it_are_read(): + before = _workflow(_agent("--append-system-prompt 'Follow https://example.com/style-guide' --allowedTools Read")) + after = _workflow(_agent( + "--append-system-prompt 'Follow https://example.com/style-guide' --dangerously-skip-permissions" + )) + + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert ( + "claude_args: --append-system-prompt 'Follow https://example.com/' " + "--dangerously-skip-permissions" + ) in row.after + assert "style-guide" not in row.before + row.after + assert uncompared_agent_launch_texts(_grant(after)) == [] + assert _uncompared_workflow_text(_grant(after)) is None + + +def test_a_marketplace_url_compares_by_scheme_and_host_as_an_mcp_server_url_does(): + def marketplace(url): + return _workflow(_agent(plugin_marketplaces=url)) - assert launch["settings"] == [{"name": "--settings", "value": None, "unresolved_reason": "redacted"}] + launch, = _launches(marketplace("https://github.com/anthropics/claude-code.git")) + assert { + "name": "plugin_marketplaces", "value": "https://github.com/", "unresolved_reason": None, + } in launch["settings"] + row, = _rows(marketplace("https://github.com/org/a.git"), marketplace("https://gitlab.example.com/org/a.git")) + assert "plugin_marketplaces: https://gitlab.example.com/" in row.after + # The path is withheld as an MCP server URL's is (#723), so a change only + # there is not reported; the support page says so. + assert _rows(marketplace("https://github.com/org/a.git"), marketplace("https://github.com/org/b.git")) == [] + + +def test_credential_shaped_text_is_published_redacted_and_blocks_the_workflow_as_a_step_reference_does(): + grant = _grant(_workflow( + _agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions", + plugin_marketplaces="https://robot:PWCANARY@github.com/org/repo.git"), + {"uses": "actions/checkout@v4", "with": {"ref": "token=REFCANARY"}}, + {"run": "claude -p --settings ghp_" + "A" * 36 + " 'review token=SECRETCANARY'"}, + )) + action, cli = grant["agent_launches"] + + assert action["settings"] == [ + {"name": "claude_args", + "value": "--append-system-prompt 'use token=' --dangerously-skip-permissions", + "unresolved_reason": "redacted"}, + {"name": "plugin_marketplaces", "value": "https://github.com/", "unresolved_reason": "redacted"}, + ] + # The rule is read from the declared text, so redaction does not hide it. + assert action["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] + assert cli["settings"] == [ + {"name": "--settings", "value": "[REDACTED:github_token]", "unresolved_reason": "redacted"}, + ] + assert grant["checkout_refs"] == [ + {"job": "review", "step": "steps[1]", "ref": "token=", "unresolved_reason": "redacted"}, + ] text = json.dumps(grant) - assert "ghp_" not in text and "SECRETCANARY" not in text + for canary in ("ARGCANARY", "PWCANARY", "REFCANARY", "SECRETCANARY", "ghp_"): + assert canary not in text + assert uncompared_agent_launch_texts(grant) == [] + assert _uncompared_workflow_text(grant) == ( + "an agent launch setting and a checkout ref contain credential-shaped text; " + "they are published redacted and cannot be compared" + ) def test_token_shaped_job_and_step_labels_are_redacted_in_every_entry(): @@ -807,34 +1119,91 @@ def test_the_stop_hook_announces_the_widening(pr, tmp_path): assert SOURCE in message -def test_no_canary_reaches_any_published_output(tmp_path): - canary = "sk-ant-api03-" + "Z" * 40 - job = "ghp_" + "D" * 36 - base = _workflow(jobs={job: {"steps": [_agent()]}}) - head = _workflow(jobs={job: {"steps": [ - {"name": "Pull docker://ci:" + "p4ssCANARY" + "@gcr.io/x", "uses": "actions/checkout@v4", - "with": {"ref": "token=REFCANARY"}}, - _agent(f"--mcp-config '{{\"headers\":{{\"Authorization\":\"Bearer {canary}\"}}}}'"), - {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read 'go'"}, - ]}}) - repo = _repo(tmp_path, {SOURCE: _yaml(base)}) - _git(repo, "checkout", "-qb", "change") - _write(repo, {SOURCE: _yaml(head)}) - _git(repo, "commit", "-qam", "canaries") +def _published_outputs(repo: Path) -> str: + """Every route's output for the change on ``repo``: diff, audit, check, verify and its reports.""" - outputs = [json.dumps(_diff(repo))] + outputs = [] for args in ( + ["diff", "--workspace", str(repo), "--base", "main", "--json"], ["diff", "--workspace", str(repo), "--base", "main"], ["audit", "--host", "--workspace", str(repo), "--json"], + ["audit", "--host", "--workspace", str(repo)], ["check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "agent-boundary-json"], + ["check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "agent-control-json"], ["verify", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "text"], ): result = CliRunner().invoke(app, args) outputs.append(result.output) outputs.append((repo / "agents-shipgate-reports/pr-comment.md").read_text()) outputs.append((repo / "agents-shipgate-reports/verifier.json").read_text()) - joined = "\n".join(outputs) + return "\n".join(outputs) + + +def _assert_absent(joined: str, canaries) -> None: + for canary in canaries: + assert canary not in joined, canary + digest = hashlib.sha256(canary.encode()).hexdigest() + for prefix in (digest, digest[:24], digest[:12]): + assert prefix not in joined, canary + + +def test_no_canary_reaches_any_published_output(tmp_path): + """The #802 sweep for agent launches, JSON-shaped canaries included (#823 review F3).""" + + canary = "sk-ant-api03-" + "Z" * 40 + job = "ghp_" + "D" * 36 + mcp = json.dumps({"mcpServers": {"db": { + "command": "db-mcp", "env": {"DB_API_TOKEN": "canary-tok-123"}, + "headers": {"X-API-Key": "canary-hdr-456", "Authorization": f"Bearer {canary}"}, + }}}) + codex_args = "-c 'mcp_servers.db.env.TOKEN=\"canary-cfg-789\"' --full-auto" + base = _workflow(jobs={job: {"steps": [_agent()]}}) + head = _workflow(jobs={job: {"steps": [ + {"name": "Pull docker://ci:" + "p4ssCANARY" + "@gcr.io/x", "uses": "actions/checkout@v4"}, + _agent( + f"--mcp-config '{mcp}' --dangerously-skip-permissions", + settings=SETTINGS_JSON, mcp_config=mcp, + plugin_marketplaces="https://github.com/canary-org/canary-repo.git", + ), + {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read --mcp-config '{mcp}' 'go'"}, + {"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}, + ]}}) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "canaries") - for secret in (canary, "p4ssCANARY", "REFCANARY", job): - assert secret not in joined + payload = _diff(repo) + assert payload["comparison_status"] == "comparable" + row, = payload["rows"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert '"headers":{"Authorization":"","X-API-Key":""}' in row["after"] + assert f"settings: {SETTINGS_PUBLISHED}" in row["after"] + joined = _published_outputs(repo) + + _assert_absent(joined, ( + canary, "p4ssCANARY", job, "canary-org", "canary-repo", "canary-cfg-789", *JSON_CANARIES, + )) assert "runs claude -p with --allowedTools Read" in joined + assert "mcp_servers.db.env.TOKEN=" in joined + + +def test_a_credential_shaped_setting_or_ref_refuses_a_changed_workflow_and_publishes_no_canary(tmp_path): + base = _workflow(_agent()) + head = _workflow( + {"uses": "actions/checkout@v4", "with": {"ref": "token=REFCANARY"}}, + _agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions"), + ) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "credential-shaped values") + + payload = _diff(repo) + assert payload["comparison_status"] == "incomparable" and payload["rows"] == [] + assert "head_inventory_incomplete" in payload["incomparable_reasons"] + audit = json.loads(CliRunner().invoke(app, ["audit", "--host", "--workspace", str(repo), "--json"]).stdout) + issue, = [item for item in audit["issues"] if item["host"] == "github"] + assert (issue["kind"], issue["blocking"]) == ("unsupported", True) + assert "an agent launch setting and a checkout ref contain credential-shaped text" in issue["message"] + _assert_absent(_published_outputs(repo), ("ARGCANARY", "REFCANARY")) From aa3bc9963964a2e524b37bd3605f1b69cd06703b Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Tue, 22 Sep 2026 19:00:45 -0700 Subject: [PATCH 04/14] Address review cycle 1 on agent launches in CI (#823) The second review of #850 at 25c13ce8 found two P1 and four P2 defects. The branch is rebased onto origin/main daa4ad5f. Attached JSON values are withheld (F1). A --settings={...} or --mcp-config={...} word inside claude_args, and codex's attached -c / -c=, published their env, header and apiKeyHelper values verbatim, because only a word starting with "{" was withheld. _withheld_words now splits a --name=value word and withholds the value, and reads codex's attached -c through _withheld_config, as clap reads it. The canary sweep carries both spellings. Redacted prose no longer refuses the comparison (F2). The #802 label redaction rewrites ordinary prose ("never print bearer tokens", "Authorization: headers"), and a redacted agent setting made the workflow a blocking limit. That hid every row beside it and made every check incomparable while the workflow existed. The rules are already read from the declared text, so a redacted setting is now compared by its published text and its widening_rules. uncompared_agent_launch_texts names it as a non-blocking limit, and the row cell shows the redacted text. A redacted checkout ref still refuses, as a redacted step reference does (#767), because it names the code a job runs. A renamed job's launch moves its rules (F3). _agent_rule_gains keyed a rule on its job, so renaming a job that launches a bypassing agent was a widening. agent_rule_gains now pairs a rule one job gains with the same rule another job lost, when the launch that met it left that job: the job no longer launches that agent, or the same launch (agent_launch_key less the job) now runs in the gaining job. The why names the move. A second job gaining a rule, or a different launch gaining one while the first job still launches that agent, still widens. An expression no longer turns off every rule, and the row says what it leaves unread (F4). Rules are read from literal text a ${{ }} expression cannot reach: - the words of claude_args or codex-args before the first expression, less the word it touches and any quoted run still open at it; - the elements of a JSON-array codex-args before the one holding it; - the gate entries that hold none. A setting holding an expression is published with holds_expression (the unreleased 0.7 schema extends in place), and a row that changes it says the text the expression reaches is not read. A gain where the job's launch held an expression before, in the input the rule is read from, is named and not claimed, as unknown_before is. That also fixes a false widening at 25c13ce8: replacing --model ${{ vars.CLAUDE_MODEL }} with --model opus beside --dangerously-skip-permissions was reported as gaining the bypass. Two non-blocking fixes: - _published_value reads each expression as one word, so an expression in a URL's userinfo is withheld with it rather than garbling the URL and publishing its path. - A codex --config value that starts like a table and does not parse, as string-argv leaves a quoted one, is withheld as unparsed_json. Rebase (F5). CHANGELOG keeps #853's #778 line beside #823's under github_action row. llms.txt and ai-search-summary state the source tree as contract 41, unreleased, ahead of the published v1.1.0 (contract 40), as test_public_surface_contract requires while the two differ. The pilot ledger's source-tree column was re-taken on the rebased tree, through ./shipgate beside the engine of e3c6cb0c, the commit v1.1.0 was cut from: - the only differences are contract 40 -> 41 and inventory schema 0.6 -> 0.7; - diff rows are byte-identical; - check JSON differs only in the launcher path its next action names. The support page, the STABILITY preamble and migration note, the CHANGELOG entry, agent-contract-current, the Stop hook sentence in integrations, the capability_diff row in distribution-surfaces with its parity comment, the 0.7 schema files and llms-full.txt are updated to match. --- CHANGELOG.md | 3 +- STABILITY.md | 24 +- docs/agent-contract-current.md | 13 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 112 +++-- docs/host-grants-baseline-schema.v0.7.json | 11 +- docs/host-grants-inventory-schema.v0.7.json | 11 +- docs/integrations.md | 3 +- llms-full.txt | 13 +- .../core/capability_diff_rows.py | 77 +++- src/agents_shipgate/core/host_grants.py | 380 +++++++++++++---- src/agents_shipgate/schemas/host_grants.py | 36 +- tests/test_distribution_surface_parity.py | 8 +- tests/test_workflow_agent_launches.py | 400 +++++++++++++++++- 14 files changed, 891 insertions(+), 202 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 052ec75bc..cb26d2274 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,7 +11,8 @@ - **What it is not.** Never a row, a widening, a `check` violation or a claim that a host loads the file. Nothing is fetched or run, only plugin manifests and marketplaces are read, and at most 32 candidates are examined; the rest, and any whose rule needed a file that was not read or did not parse, are counted as not examined, on a line that names both causes. The rules are listed in `docs/host-boundary-support.md` under *Changed inputs named but not read*. - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A setting publishes what the host readers would: a JSON object its key names, with `env` and `headers` values and `apiKeyHelper` withheld, and a URL its scheme and host; other credential-shaped text in a setting or checkout ref is published redacted and refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) + +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A setting publishes what the host readers would: a JSON object its key names, with `env` and `headers` values and `apiKeyHelper` withheld however it is attached to its flag, and a URL its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index 295a28f8e..c1211aedc 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -30,11 +30,15 @@ every other edit is a `changed` row naming `job/step`, and a workflow row that runs an agent ends with the job facts beside each agent step. The rules each launch meets are read from its declared text and published as `widening_rules`, and `claude_args` and `codex-args` are split as each action -splits them. A setting publishes what the host readers would: a JSON object its -key names, a URL its scheme and host. A compound command, an expansion or an -expression is `unresolved`, a named non-blocking limit that leaves coverage -complete; a setting or checkout ref holding credential-shaped text is published -redacted and refuses as a redacted step reference does. A `0.4`–`0.6` baseline holding a workflow +splits them, and only from literal text a `${{ }}` expression cannot reach (a +setting holding one says so with `holds_expression`); a rule a launch already +met in a job it left is moved, not gained. A setting publishes what the host +readers would: a JSON object its key names, however it is attached to its flag, +a URL its scheme and host. A compound command, an expansion or an expression in +`run:` is `unresolved`, a named non-blocking limit that leaves coverage +complete; a setting holding credential-shaped text, prose included, is +published redacted, compared as published and named the same way, while a +checkout ref holding it refuses as a redacted step reference does. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and `minimum_control_contract_version` `21` are unchanged. See @@ -348,12 +352,12 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i ``` - **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). -- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings`, `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. -- **What is withheld.** A setting publishes what the host readers would publish for the same text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value, or any word of `claude_args` or `codex-args` — publishes its key names, with `env` and `headers` values, `apiKeyHelper` and every secret-named value ``, as canonical JSON; a codex `--config` override under `env`, `headers` or a secret-named key publishes `` for its value. A URL publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to a URL's path or query is not reported. Text that starts like JSON and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal values only, so a value holding `${{ }}` never meets a rule. A rule gained where the job launched that agent before only in a form this audit does not read is named in the `why` and not claimed, as for a job whose permissions were not explicit. The grant then earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. +- **What is withheld.** A setting publishes what the host readers would publish for the same text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — publishes its key names, with `env` and `headers` values, `apiKeyHelper` and every secret-named value ``, as canonical JSON; a codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. A URL publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox` or `safety-strategy` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. -- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. -- **Credential-shaped values refuse, as a redacted step reference does (#767).** Other text the #802 label redaction rewrites in a setting or a checkout ref — a token shape, a credential assignment, a bearer or header value, a URL's userinfo — is published redacted with `unresolved_reason: redacted` and makes the workflow a blocking `unsupported` limit, because two values that redact alike cannot be compared apart: a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. Its rules are still read from the declared text. +- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. +- **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. - **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, a path), and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. **Compatibility.** diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index 2e5820ad3..f89d3882a 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -53,7 +53,8 @@ An agent launch is a step whose `uses:` is a documented agent action `openai/codex-action`) with the permission inputs it declares, or a `run:` that is one literal `claude -p` / `codex exec` command with its documented permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`), +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, +`holds_expression`), `widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, `null` for the default. Values are compared as text and never executed; `claude_args` and `codex-args` are split @@ -62,11 +63,15 @@ would, a JSON object by its key names and a URL by its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the -row `widened`; every other edit is `changed`, and a workflow row that runs an +row `widened`. A rule is read only from literal text a `${{ }}` expression +cannot reach, and one a launch already met in a job it left, or where the job's +launch before was unread or held an expression the rule is read from, is named +and not claimed. Every other edit is `changed`, and a workflow row that runs an agent ends its `why` with the job facts beside each agent step. A compound `run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit; a setting or checkout ref holding credential-shaped text is published -redacted and a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +limit; a setting holding credential-shaped text, prose included, is published +redacted, compared as published and a named non-blocking limit, and a checkout +ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index fa439b9c1..0def93f90 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`), with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch or unreadable value is named only by the host inventory and `audit --host`, as for an unread secret value, and a setting or checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from literal text a `${{ }}` expression cannot reach, and a gain the engine does not claim (a rule moved in from a job the launch left, or one the job's unread or expression-holding launch before may already have met) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch, an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 69114b48d..658ec1173 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -193,18 +193,44 @@ Direction is claimed only by the documented rules above. The rules each launch meets are decided when the workflow is read, from the declared text before anything is withheld for publication, and published on the launch as `widening_rules` (each rule and the setting it was read from), so redaction -never hides one. Only literal values meet a rule: GitHub substitutes a -`${{ }}` expression into an input before the action reads it, so a value -holding one meets none. When a job's launches gain one, the workflow earns +never hides one. + +Only literal text meets a rule. GitHub substitutes a `${{ }}` expression into +an input before the action reads it, and the substituted text may be anything +— more flags, a quote that closes one opened before it, a `#` that ends +`claude_args` — so a rule is read only from text it cannot reach: the words of +`claude_args` or `codex-args` before the first expression, less the word it +touches and any quoted run still open at it (in a JSON-array `codex-args`, the +elements before the one holding it), and the entries of a user gate that hold +no expression, so `"${{ vars.USERS }}, *"` opens the gate. A `sandbox` or +`safety-strategy` value holding one meets none. Such a setting is published +with `holds_expression: true`, and a row that changes it says the text the +expression reaches is not read for a rule, rather than that none was gained. + +When a job's launches gain a rule, the workflow earns `workflow_agent_widened_`, the row is `widened` and its `why` -names the rule and step. A rule gained where the job launched that agent before -only in a form this audit does not read — a compound `run:` that became a -literal one — is named in the `why` and not claimed, because the unread launch -may already have met it, as a job whose permissions were not explicit may -already have held a write scope. Any other edit — `--allowedTools "Read"` to -`--allowedTools "Bash(*)"`, `acceptEdits`, a new plugin, a value holding -`${{ }}`, a head-ref checkout — is `changed`; rating a tool rule's reach is a -job for #824. `access` and `risk` still describe the token and triggers alone. +names the rule and step. Three gains are named in the `why` and not claimed: + +- where the job launched that agent before only in a form this audit does not + read — a compound `run:` that became a literal one — because the unread + launch may already have met it, as a job whose permissions were not explicit + may already have held a write scope; +- where the job's launch of that agent held a `${{ }}` expression before in an + input the rule is read from (`claude_args` for bypassed permission checks, + the gate for a `*` entry), because the substituted text may already have met + it — so replacing `--model ${{ vars.M }}` with `--model opus` beside + `--dangerously-skip-permissions` is not a widening; +- where the rule moved between jobs: another job met it before and the launch + that met it left that job — the job no longer exists or no longer launches + that agent, as when a job is renamed or an agent step moves, or the same + launch now runs in this job — as a step reference moved between jobs adds no + scope. A second job gaining a rule a first job keeps, or a different launch + gaining it while the first job still launches that agent, is claimed. + +Any other edit — `--allowedTools "Read"` to `--allowedTools "Bash(*)"`, +`acceptEdits`, a new plugin, a flag after a `${{ }}` expression, a head-ref +checkout — is `changed`; rating a tool rule's reach is a job for #824. `access` +and `risk` still describe the token and triggers alone. A workflow row whose workflow runs an agent ends its `why` with the job facts beside each agent step, whatever else the row is about: an untrusted-input @@ -219,38 +245,46 @@ event are not read into it. A removed workflow gets no note. A setting publishes what the host readers would publish for the same text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or -`--mcp-config` value, or any word of `claude_args` or `codex-args` — publishes +`--mcp-config` value written as its own word or as `--settings={…}`, or any +word of `claude_args` or `codex-args` — publishes as `.claude/settings.json` and `.mcp.json` do: its key names, with `env` and `headers` values, `apiKeyHelper` and every other secret-named value read as ``, in canonical JSON, so rotating an `env` value or reordering keys compares as unchanged and adding a key is a change. A codex `--config` -override under `env`, `headers` or a secret-named key publishes `` for -its value, and a table or array value by the same JSON rule. A URL publishes -its scheme, host and port, with `` for any path and no query, as -an MCP server's URL does (#723), and the rest of the setting is compared, so a change -only to a URL's path or query — which repository a `plugin_marketplaces` URL -names, for one — is not reported; a zero-row result says redacted values are -not compared. - -Any other text the #802 label redaction rewrites in a setting or a checkout ref -is credential-shaped: a token shape, a credential assignment such as `token=…`, -a bearer or header value, a URL's userinfo. The value is published redacted -with `unresolved_reason: redacted` and refuses the same way a step reference -does, because two values that redact alike cannot be compared apart: GitHub -coverage is `partial`, a changed workflow's comparison is refused, and an -unchanged one is named in `unchanged_limits`. Its rules are still read from the -declared text. - -An unresolved launch, a setting that is not a string (`not_a_string`) or that -holds text starting like JSON that does not parse (`unparsed_json`, -whose values cannot be told from its keys), and a checkout ref that is not a -string or whose `with:` is not a mapping publish nothing of the value and -record a **non-blocking** `unsupported` coverage issue naming the `job/step`, -printed under `audit --host` → Coverage issues. GitHub coverage stays complete, -so `check`, baselines and every other row are unaffected, and adding, removing -or re-forming such an entry, or its gaining a documented rule, is still a row; -only an edit inside it that gains no rule is not reported. `diff`, `verify` and -`check` carry no limit for it, as for an unread secret value (#693). +override — `-c key=value`, `--config=key=value`, `-ckey=value` or +`-c=key=value` — under `env`, `headers` or a secret-named key publishes +`` for its value, and a table or array value by the same JSON rule. A +URL publishes its scheme, host and port, with `` for any path and +no query, as an MCP server's URL does (#723), and the rest of the setting is +compared, so a change only to a URL's path or query — which repository a +`plugin_marketplaces` URL names, for one — is not reported; a zero-row result +says redacted values are not compared. A `${{ }}` expression is one word while +this is decided, so one inside a URL's userinfo is withheld with it. + +Any other text the #802 label redaction rewrites is credential-shaped: a token +shape, a credential assignment such as `token=…`, a bearer or header value, a +URL's userinfo — and ordinary prose too, such as "never print bearer tokens" in +a system prompt. In a setting, the value is published redacted with +`unresolved_reason: redacted` and compared as published, beside the rules read +from its declared text, so a permission change or a rule gained beside it is +still a row; only an edit inside what is redacted that gains no rule is not +reported, and that is named as a non-blocking limit below. A checkout ref names +the code a job runs, as a step reference does, so a redacted ref refuses the +same way a step reference does (#767), because two refs that redact alike +cannot be compared apart: GitHub coverage is `partial`, a changed workflow's +comparison is refused, and an unchanged one is named in `unchanged_limits`. + +An unresolved launch, a setting that is not a string (`not_a_string`), that +holds text starting like JSON or a codex `--config` table or array that does +not parse (`unparsed_json`, whose values cannot be told from its keys), or that +is published redacted (`redacted`), and a checkout ref that is not a string or +whose `with:` is not a mapping record a **non-blocking** `unsupported` coverage +issue naming the `job/step`, printed under `audit --host` → Coverage issues; +all but a redacted setting publish nothing of the value. GitHub coverage stays +complete, so `check`, baselines and every other row are unaffected, and adding, +removing or re-forming such an entry, or its gaining a documented rule, is +still a row; only an edit inside it that gains no rule is not reported. `diff`, +`verify` and `check` carry no limit for it, as for an unread secret value (#693). A workflow's labels are published redacted (#802). A job id, a step's `id` or `name`, a trigger and a permission scope name go through the same redaction as diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index dc781c14f..2deac0c4f 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1365,7 +1365,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only a\nliteral value meets one: a value holding ``${{ }}`` meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode input holding one meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1392,8 +1392,13 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). The rest is published through the workflow label\nredaction (#802). A value it rewrites beyond that is credential-shaped: it\nis published redacted with ``unresolved_reason: redacted`` and makes the\nworkflow a blocking limit, as a redacted step reference does (#767). A\nvalue that is not a string (``not_a_string``), or one holding text that\nstarts like JSON and does not parse (``unparsed_json``), is\n``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). Each is withheld however it is attached to its flag:\n``--settings={\u2026}`` and ``-c`` as well as a separate word. The\nrest is published through the workflow label redaction (#802). A value it\nrewrites beyond that is credential-shaped \u2014 a token, but also prose such\nas \"never print bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", "properties": { + "holds_expression": { + "default": false, + "title": "Holds Expression", + "type": "boolean" + }, "name": { "title": "Name", "type": "string" @@ -1436,7 +1441,7 @@ }, "HostWorkflowCheckoutRefV7": { "additionalProperties": false, - "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767). A value that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767): a ref names the code the job runs, as a step reference does. A\nvalue that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", "properties": { "job": { "title": "Job", diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index 6d6dc4b4f..eec78d398 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1423,7 +1423,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only a\nliteral value meets one: a value holding ``${{ }}`` meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode input holding one meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1450,8 +1450,13 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). The rest is published through the workflow label\nredaction (#802). A value it rewrites beyond that is credential-shaped: it\nis published redacted with ``unresolved_reason: redacted`` and makes the\nworkflow a blocking limit, as a redacted step reference does (#767). A\nvalue that is not a string (``not_a_string``), or one holding text that\nstarts like JSON and does not parse (``unparsed_json``), is\n``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). Each is withheld however it is attached to its flag:\n``--settings={\u2026}`` and ``-c`` as well as a separate word. The\nrest is published through the workflow label redaction (#802). A value it\nrewrites beyond that is credential-shaped \u2014 a token, but also prose such\nas \"never print bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", "properties": { + "holds_expression": { + "default": false, + "title": "Holds Expression", + "type": "boolean" + }, "name": { "title": "Name", "type": "string" @@ -1494,7 +1499,7 @@ }, "HostWorkflowCheckoutRefV7": { "additionalProperties": false, - "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767). A value that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", + "description": "One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823).\n\n``ref`` is ``null`` when the step declares none, or an empty one: the\ncheckout's default for the triggering event. A ref the label redaction\nrewrites is published redacted with ``unresolved_reason: redacted`` and\nmakes the workflow a blocking limit, as a redacted step reference does\n(#767): a ref names the code the job runs, as a step reference does. A\nvalue that is not a string, or ``with:`` that is not a mapping,\nis ``null`` with ``unresolved_reason`` and records a non-blocking coverage\nissue. The ref is never resolved or fetched.", "properties": { "job": { "title": "Job", diff --git a/docs/integrations.md b/docs/integrations.md index 485ee999a..30accb76c 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -229,7 +229,8 @@ change without gaining a documented widening rule, and for a checkout's ref. One that gains a rule, such as `claude_args` gaining `--dangerously-skip-permissions` on any of its lines, widens, and the hook announces it (#823), unless the job launched that agent before only in a form -the audit does not read. +the audit does not read or with a `${{ }}` expression the rule is read from, +or the rule moved in from a job the launch left, as a renamed job's does. It names each widening row once, and repeats the announcement only when the change or its rows change. A missing base ref, an incomparable inventory or unparsed output is never diff --git a/llms-full.txt b/llms-full.txt index 5476338d7..c89be0865 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1627,7 +1627,8 @@ An agent launch is a step whose `uses:` is a documented agent action `openai/codex-action`) with the permission inputs it declares, or a `run:` that is one literal `claude -p` / `codex exec` command with its documented permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`), +with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, +`holds_expression`), `widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, `null` for the default. Values are compared as text and never executed; `claude_args` and `codex-args` are split @@ -1636,11 +1637,15 @@ would, a JSON object by its key names and a URL by its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the -row `widened`; every other edit is `changed`, and a workflow row that runs an +row `widened`. A rule is read only from literal text a `${{ }}` expression +cannot reach, and one a launch already met in a job it left, or where the job's +launch before was unread or held an expression the rule is read from, is named +and not claimed. Every other edit is `changed`, and a workflow row that runs an agent ends its `why` with the job facts beside each agent step. A compound `run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit; a setting or checkout ref holding credential-shaped text is published -redacted and a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +limit; a setting holding credential-shaped text, prose included, is published +redacted, compared as published and a named non-blocking limit, and a checkout +ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index 87eca1869..a94d5345b 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -27,12 +27,12 @@ from typing import Any from agents_shipgate.core.host_grants import ( + AGENT_RULE_INPUTS, AGENT_WIDENING_RULES, UNTRUSTED_INPUT_TRIGGERS, agent_launch_key, - agent_widenings_unread_before, + agent_rule_gains, checkout_ref_key, - gained_agent_widenings, hook_loading_basis, host_grant_expansion_signals, permission_rule_replacements, @@ -292,7 +292,8 @@ def _agent_launch_value(item: dict[str, Any]) -> str: for setting in item.get("settings") or []: unread = setting.get("unresolved_reason") name = str(setting["name"]) - if unread: + # A redacted value is compared as published, so the cell shows it (#823 review F2). + if unread and not (unread == "redacted" and setting.get("value") is not None): parts.append(f"{name} (unresolved: {str(unread).replace('_', ' ')})") elif setting.get("value") is None: parts.append(name) @@ -323,30 +324,49 @@ def _agent_launch_reasons( ) -> list[str]: """What changed in how an agent is launched, and which of it widens (#823). - Only a documented rule gained by a job's agent launches is called a - widening, and the sentence says which rule and where. A rule gained where - the job's launch was unread before is named and not called a widening, as - the engine claims no expansion for it. Every other agent-launch or - checkout edit is a change: its settings are compared as published text, - and nothing here ranks one value against another. + Only a documented rule the engine claims — ``agent_rule_gains(...).claimed``, + the rule behind ``workflow_agent_widened_*`` — is called a widening, and + the sentence says which rule and where. A rule gained where the job's + launch was unread before, or held a ``${{ }}`` expression the rule is read + from, or that moved in from another job, is named and not called a + widening, as the engine claims no expansion for it. Every other + agent-launch or checkout edit is a change: its settings are compared as + published text, and nothing here ranks one value against another. """ def where(item: dict[str, Any]) -> str: return f"{item['job']}/{item['step']}" + def meets(rule: str, detail: str) -> str: + return AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") + + gains = agent_rule_gains(before, after) reasons: list[str] = [] - widened_at: set[str] = set() - for _job, rule, detail, entry in gained_agent_widenings(before, after): - widened_at.add(where(entry)) - what = AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") - reasons.append(f"an agent launch now {what} ({where(entry)})") - for _job, rule, detail, entry in agent_widenings_unread_before(before, after): - widened_at.add(where(entry)) - what = AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") + # Launches a sentence about a rule already names. + named: set[str] = set() + for _job, rule, detail, entry in gains.claimed: + named.add(where(entry)) + reasons.append(f"an agent launch now {meets(rule, detail)} ({where(entry)})") + for _job, rule, detail, entry in gains.unread_before: + named.add(where(entry)) + reasons.append( + f"an agent launch now {meets(rule, detail)} ({where(entry)}), which is not counted as a " + "widening: before, this job launched the agent in a form this audit does not read, which " + "may already have done the same" + ) + for (_job, rule, detail, entry), setting in gains.expression_before: + named.add(where(entry)) + reasons.append( + f"an agent launch now {meets(rule, detail)} ({where(entry)}), which is not counted as a " + f"widening: before, this job's {setting} held a " + "`${{ }}`" + " expression, whose " + "substituted text this audit does not read and which may already have done the same" + ) + for (_job, rule, detail, entry), source in gains.moved: + named.update({where(entry), where(source)}) reasons.append( - f"an agent launch now {what} ({where(entry)}), which is not counted as a widening: " - "before, this job launched the agent in a form this audit does not read, which may " - "already have done the same" + f"an agent launch that {meets(rule, detail)} moved between jobs ({where(source)} → " + f"{where(entry)}), which is not counted as a widening: the launch already met that " + "rule in the job it left, and it now runs with the receiving job's token permissions" ) # The step label only words the sentence; what changed was decided by the # comparator's key, which never reads it. @@ -355,7 +375,7 @@ def where(item: dict[str, Any]) -> str: groups: dict[str, list[str]] = {} for item in (*new, *gone): label = where(item) - if label in widened_at: + if label in named: continue verb = "changed" if label in old and label in now else ("added" if label in now else "removed") groups.setdefault(verb, []) @@ -375,6 +395,21 @@ def where(item: dict[str, Any]) -> str: f"{_joined_words(phrases)}; agent launch settings are compared as declared text, " "and a change that gains no documented widening rule is not counted as a widening" ) + # A rule is read only from literal text a `${{ }}` expression cannot + # reach, so the row says where that leaves text unread (#823 review). + expressions = list(dict.fromkeys( + f"{setting['name']} at {where(item)}" + for item in new if item.get("form") == "read" + for setting in item.get("settings") or [] + if setting.get("holds_expression") and setting["name"] in AGENT_RULE_INPUTS + )) + if expressions: + reasons.append( + "an agent launch setting holds a `${{ }}` expression (" + ", ".join(expressions) + "), " + "which GitHub substitutes before the action reads it; documented widening rules are " + "read only from the literal text the expression cannot reach, so this row does not " + "say whether the text it reaches meets one" + ) unread = list(dict.fromkeys(where(item) for item in new if item.get("form") != "read")) if unread: reasons.append( diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 54988a6a0..72e100bc6 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -1708,6 +1708,13 @@ class _AgentAction: _AGENT_FLAG_TABLES = {"claude": _CLAUDE_FLAGS, "codex": _CODEX_FLAGS} +#: The agent action inputs a documented widening rule is read from. +AGENT_RULE_INPUTS: frozenset[str] = frozenset( + name + for spec in _AGENT_ACTIONS.values() + for name in (*spec.gates, *(mode for mode, _value in spec.modes), *((spec.args,) if spec.args else ())) +) + #: The widening each documented rule names, as a row's ``why`` says it. AGENT_WIDENING_RULES: dict[str, str] = { "bypass_permissions": "skips permission checks (bypassPermissions)", @@ -2063,6 +2070,83 @@ def _argument_input(family: str, value: str) -> _ArgumentInput: return _claude_argument_input(value) if family == "claude" else _codex_argument_input(value) +# --- what an expression in an input leaves readable (#823 review) --------------------- +# +# GitHub substitutes a `${{ }}` expression into an input before the action reads +# it, and the substituted text may be anything: more words, a quote that closes +# one opened before it, a `#` that ends `claude_args`. So a rule is read only +# from literal text the expression cannot reach. + +#: One ``${{ … }}`` expression, or an unterminated ``${{`` to the end of the text. +_EXPRESSION_SPAN_RE = re.compile(r"\$\{\{.*?(?:\}\}|\Z)", re.S) +#: What an expression reads as while literal text around it is split: a +#: private-use character, which no splitter here reads as a blank, a quote or +#: an operator, so the word the expression touches holds it and is set aside. +_EXPRESSION_MARK = "\ue000" + + +def holds_expression(text: str) -> bool: + """Whether declared text holds a ``${{ }}`` expression GitHub substitutes before the action reads it.""" + + return "${{" in text + + +def _skipped_quote(text: str, pattern: re.Pattern[str]) -> int | None: + """The first quote ``pattern`` leaves unmatched in ``text``, else ``None``. + + Both splitters skip a quote no word covers; before an expression, that is a + quoted run the substituted text may close, so nothing from it on is read. + """ + + covered = 0 + for match in (*pattern.finditer(text), None): + gap = text[covered:] if match is None else text[covered:match.start()] + quote = next((index for index, char in enumerate(gap) if char in "'\""), None) + if quote is not None: + return covered + quote + if match is not None: + covered = match.end() + return None + + +def _literal_argument_words(family: str, text: str) -> tuple[str, ...] | None: + """The words of an argument input no ``${{ }}`` expression in it can reach, as the action splits them. + + Without an expression, every word the action passes on. With one, the + words the action has finished reading before the first expression: the + word the expression touches, any quoted run still open at it, and + everything after it are not read. A JSON array ``codex-args`` gives the + elements before the one holding an expression. ``None`` when the action + refuses what is left. + """ + + start = text.find("${{") + if start == -1: + return _argument_input(family, text).words + if family == "codex" and text.startswith("["): + words = _argument_input(family, _EXPRESSION_SPAN_RE.sub(_EXPRESSION_MARK, text)).words + else: + prefix, pattern = text[:start], _STRING_ARGV_RE + if family == "claude": + # A line the action drops as a comment is dropped whatever the + # expression on it holds, and the lines before it are whole. + *whole, last = prefix.split("\n") + dropped = last.strip().startswith("#") + kept = [line for line in whole if not line.strip().startswith("#")] + prefix = "\n".join([*kept, ""] if dropped else [*kept, last]) + pattern = _SHELL_QUOTE_CHUNK_RE + quote = _skipped_quote(prefix, pattern) + words = _argument_input(family, prefix[:quote] + _EXPRESSION_MARK).words + if words is None: + return None + literal: list[str] = [] + for word in words: + if _EXPRESSION_MARK in word: + break + literal.append(word) + return tuple(literal) + + # --- what an agent setting publishes (#823, #802) -------------------------------------- #: A codex ``--config`` override under one of these keys carries values the @@ -2110,7 +2194,10 @@ def _withheld_config(text: str) -> str | None: A key path through ``env``, ``headers`` or a secret-named key publishes ```` for its value; a table or array value, parsed as TOML as - codex parses it, publishes by :func:`_withheld_json`. Anything else is kept. + codex parses it, publishes by :func:`_withheld_json`. A value that starts + like a table or array and does not parse — string-argv keeps the quotes of + ``--config='k={…}'`` — is ``None``: what it holds cannot be told apart from + its keys. Anything else is kept. """ key, equals, value = text.partition("=") @@ -2122,15 +2209,28 @@ def _withheld_config(text: str) -> str | None: try: loaded = tomllib.loads(f"value = {value}").get("value") except (tomllib.TOMLDecodeError, RecursionError): - return text + return None if value.strip().strip("\"'").startswith(("{", "[")) else text if isinstance(loaded, (dict, list)): shown = _withheld_json(loaded) return None if shown is None else f"{key}={shown}" return text +def _withheld_attached(prefix: str, value: str, withhold: Callable[[str], str | None]) -> str | None: + """``prefix`` and a value written attached to its flag, the value withheld by ``withhold``.""" + + shown = withhold(value) + return None if shown is None else f"{prefix}{shown}" + + def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> list[str] | None: - """Each argument word as it may be published, or ``None`` when one cannot be.""" + """Each argument word as it may be published, or ``None`` when one cannot be. + + A value is withheld however it is attached to its flag (#823 review): a + separate word, ``--name=value`` (``--settings={…}``, ``--mcp-config={…}``, + ``--config=…``), and codex's ``-c`` and ``-c=``, which clap + reads as ``-c ``. + """ shown: list[str] = [] config = False @@ -2138,8 +2238,13 @@ def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> list[ if config: item = _withheld_config(word) elif family == "codex" and word.startswith("--config="): - rest = _withheld_config(word.removeprefix("--config=")) - item = None if rest is None else f"--config={rest}" + item = _withheld_attached("--config=", word.removeprefix("--config="), _withheld_config) + elif family == "codex" and word.startswith("-c") and len(word) > 2: + prefix = "-c=" if word.startswith("-c=") else "-c" + item = _withheld_attached(prefix, word.removeprefix(prefix), _withheld_config) + elif word.startswith("--") and "=" in word: + name, _, value = word.partition("=") + item = _withheld_attached(f"{name}=", value, _withheld_word) else: item = _withheld_word(word) if item is None: @@ -2194,6 +2299,11 @@ def _url_withheld(url: str) -> str: return url if "@" in netloc else _sanitize_url(url) +#: Private-use characters that stand for one expression each while a value is redacted. +_EXPRESSION_SLOTS = range(0xE000, 0xF900) +_EXPRESSION_SLOT_RE = re.compile("[\ue000-\uf8ff]") + + def _published_value(text: str) -> tuple[str, bool]: """``text`` as it may be published, and whether credential-shaped text had to be redacted. @@ -2201,15 +2311,36 @@ def _published_value(text: str) -> tuple[str, bool]: them from an MCP server URL (#723), and the rest of the text is published and compared: a URL path is not a credential. Anything else the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or - header value, a URL's userinfo — is credential-shaped text: the value is - published redacted, and two values that redact alike cannot be compared - apart, so :func:`_uncompared_workflow_text` makes the workflow a blocking limit. + header value, a URL's userinfo — is credential-shaped text, and the value + is published redacted. The caller decides what that refuses. + + Each ``${{ }}`` expression is read as one word while this is decided, so an + expression inside a URL's userinfo is withheld with the userinfo rather + than splitting the URL (``https://x:${{ secrets.T }}@host/…`` publishes + ``https://host/``), and every expression left in the value is + published through the same label redaction. """ - withheld = _URL_RE.sub(lambda match: _url_withheld(match.group(0)), text) - if published_workflow_label(withheld) == withheld: - return withheld, False - return published_workflow_label(text), True + expressions = _EXPRESSION_SPAN_RE.findall(text) + if len(expressions) > len(_EXPRESSION_SLOTS) or _EXPRESSION_SLOT_RE.search(text): + # No free character to stand for each expression: read the text as written. + expressions = [] + def urls_withheld(value: str) -> str: + return _URL_RE.sub(lambda match: _url_withheld(match.group(0)), value) + + slots = iter(_EXPRESSION_SLOTS) + marked = _EXPRESSION_SPAN_RE.sub(lambda _match: chr(next(slots)), text) if expressions else text + withheld = urls_withheld(marked) + shown = published_workflow_label(withheld) + kept = [urls_withheld(expression) for expression in expressions] + labels = [published_workflow_label(expression) for expression in kept] + redacted = shown != withheld or labels != kept + + if expressions: + shown = _EXPRESSION_SLOT_RE.sub( + lambda match: labels[ord(match.group()) - _EXPRESSION_SLOTS.start], shown + ) + return shown, redacted def _setting_text(value: Any) -> str | None: @@ -2246,9 +2377,11 @@ def _published_setting(name: str, value: Any, *, arguments: str | None = None) - text = _setting_text(value) if text is None: return {"name": name, "value": None, "unresolved_reason": "not_a_string"} - return _published_text( + setting = _published_text( name, _withheld_word(text) if arguments is None else _withheld_arguments(arguments, text) ) + # Read off the declared text: redaction may rewrite the expression away. + return {**setting, "holds_expression": True} if holds_expression(text) else setting def _published_flag(family: str, name: str, arity: int | None, values: list[str] | None) -> dict[str, Any]: @@ -2522,24 +2655,29 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu """The documented widening rules an agent action's declared inputs meet, read from the raw text. Read here, before anything is withheld for publication, so redaction never - hides a rule. Only literal values: GitHub substitutes a ``${{ }}`` - expression into the input before the action reads it, so what it holds is - not known here and it meets no rule. Each rule names the input it was read from. + hides a rule. Only from literal text: GitHub substitutes a ``${{ }}`` + expression into the input before the action reads it, so a rule is read + only where the substituted text cannot reach — an argument input's words + before the first expression (:func:`_literal_argument_words`), a user + gate's entries that hold none — and a mode input holding one meets none. + Each rule names the input it was read from. """ - literal = [ - (name, text) for name, value in declared - if (text := _setting_text(value)) is not None and "${{" not in text - ] rules: set[tuple[str, str]] = set() - for name, text in literal: - if name in spec.gates and "*" in {part.strip() for part in text.split(",")}: - rules.add(("open_gate", name)) + for name, value in declared: + text = _setting_text(value) + if text is None: + continue + if name in spec.gates: + # The substituted text may add entries; it cannot remove a literal one. + entries = _EXPRESSION_SPAN_RE.sub(_EXPRESSION_MARK, text).split(",") + if "*" in {entry.strip() for entry in entries}: + rules.add(("open_gate", name)) for mode, widening in spec.modes: if name == mode and text == widening: rules.add(("danger_full_access" if mode == "sandbox" else "unsafe_safety_strategy", name)) if name == spec.args: - words = _argument_input(spec.family, text).words + words = _literal_argument_words(spec.family, text) if words is None: continue if spec.family == "claude": @@ -2583,64 +2721,126 @@ def agent_family(agent: str) -> str: AgentWidening = tuple[str, str, str, dict[str, Any]] +#: A rule one job's launches of one agent family meet: ``(job, family, rule, detail)``. +_RuleKey = tuple[str, str, str, str] + +#: The agent action inputs each documented rule is read from; a user gate's +#: rule is read from the gate it names. +_RULE_SETTINGS: dict[str, frozenset[str]] = { + "bypass_permissions": frozenset({"claude_args"}), + "bypass_approvals_and_sandbox": frozenset({"codex-args"}), + "danger_full_access": frozenset({"sandbox", "codex-args"}), + "unsafe_safety_strategy": frozenset({"safety-strategy"}), +} -def _agent_rule_gains( - before: dict[str, Any] | None, after: dict[str, Any] | None -) -> tuple[list[AgentWidening], list[AgentWidening]]: - """Rules a job's launches meet at ``after`` and not at ``before``: claimed, and not claimed. +def _expression_setting(entry: dict[str, Any], rule: str, detail: str) -> str | None: + """The input of ``entry`` holding a ``${{ }}`` expression whose substituted text may meet ``rule``.""" - A gain is not claimed where the job launched that agent at ``before`` only - in a form this reader does not read, such as a compound ``run:`` that - became a literal one: that launch may have met the rule already, as a job - whose permissions were not explicit may already have held a write scope - (``unknown_before``). A job that also launched the agent in a form that - was read claims the gain. + names = frozenset({detail}) if rule == "open_gate" else _RULE_SETTINGS.get(rule, frozenset()) + return next( + ( + str(setting["name"]) for setting in entry.get("settings", []) + if setting.get("holds_expression") and setting["name"] in names + ), + None, + ) + + +@dataclass(frozen=True) +class AgentRuleGains: + """The documented rules a workflow's agent launches meet at ``after`` and not at ``before`` (#823). + + Each widening is ``(job, rule, detail, entry)`` for the first launch at + ``after`` in that job that meets it. Only ``claimed`` is a widening: + + - ``unread_before``: the job launched that agent at ``before`` only in a + form this reader does not read, such as a compound ``run:`` that became a + literal one. That launch may have met the rule already, as a job whose + permissions were not explicit may already have held a write scope + (``unknown_before``). A job that also launched the agent in a form that + was read claims the gain. + - ``expression_before``: at ``before``, the job's launch of that agent held + a ``${{ }}`` expression in an input the rule is read from, and GitHub's + substituted text may already have met it. Each names that input. + - ``moved``: the rule left another job whose launch that met it left that + job — the job no longer exists or no longer launches that agent, or the + same launch now runs here — as when a job is renamed or an agent step + moves to another job (#823 review). Each names the launch it left, the + way a step reference moved between jobs adds no scope (#771). + """ + + claimed: list[AgentWidening] + unread_before: list[AgentWidening] + expression_before: list[tuple[AgentWidening, str]] + moved: list[tuple[AgentWidening, dict[str, Any]]] + + +def agent_rule_gains(before: dict[str, Any] | None, after: dict[str, Any] | None) -> AgentRuleGains: + """Which documented rules the workflow's agent launches gain, and which of them are claimed. + + Keyed by job, agent family and rule, so moving a launch between steps or + spellings (``--dangerously-skip-permissions`` and + ``--permission-mode bypassPermissions`` are one rule) gains nothing. """ - def met(grant: dict[str, Any] | None) -> dict[tuple[str, str, str, str], dict[str, Any]]: - found: dict[tuple[str, str, str, str], dict[str, Any]] = {} + def met(grant: dict[str, Any] | None) -> dict[_RuleKey, list[dict[str, Any]]]: + found: dict[_RuleKey, list[dict[str, Any]]] = {} for entry in (grant or {}).get("agent_launches", []): for rule, detail in sorted(agent_widening_rules(entry)): key = (str(entry["job"]), agent_family(str(entry["agent"])), rule, detail) - found.setdefault(key, entry) + found.setdefault(key, []).append(entry) return found - read_before: dict[tuple[str, str], bool] = {} - for entry in (before or {}).get("agent_launches", []): - key = (str(entry["job"]), agent_family(str(entry["agent"]))) - read_before[key] = read_before.get(key, False) or entry.get("form") == "read" - unread = {key for key, read in read_before.items() if not read} - old = met(before) - claimed: list[AgentWidening] = [] - unclaimed: list[AgentWidening] = [] - for key, entry in met(after).items(): - if key in old: - continue - (unclaimed if key[:2] in unread else claimed).append((key[0], key[2], key[3], entry)) - return claimed, unclaimed + def launches(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[str, Any]]]: + found: dict[tuple[str, str], list[dict[str, Any]]] = {} + for entry in (grant or {}).get("agent_launches", []): + found.setdefault((str(entry["job"]), agent_family(str(entry["agent"]))), []).append(entry) + return found + old, new = met(before), met(after) + launched_before, launched_after = launches(before), launches(after) + lost = [key for key in old if key not in new] -def gained_agent_widenings(before: dict[str, Any] | None, after: dict[str, Any] | None) -> list[AgentWidening]: - """Documented widening rules a job's agent launches meet at ``after`` and not at ``before``. - - Keyed by job, agent family and rule, so moving a launch between steps or - spellings (``--dangerously-skip-permissions`` and - ``--permission-mode bypassPermissions`` are one rule) gains nothing, and a - rule the job's unread launch at ``before`` may already have met is not - claimed. Each result is ``(job, rule, detail, entry)`` for the first - launch at ``after`` that meets it. - """ + def left(key: _RuleKey, arriving: list[dict[str, Any]]) -> bool: + # The launch that met the rule in the losing job left it: the job no + # longer launches that agent, or the same launch now runs elsewhere. + if (key[0], key[1]) not in launched_after: + return True + return any( + agent_launch_key(entry)[1:] == agent_launch_key(other)[1:] + for entry in old[key] for other in arriving + ) - return _agent_rule_gains(before, after)[0] + gains = AgentRuleGains(claimed=[], unread_before=[], expression_before=[], moved=[]) + for key, entries in new.items(): + if key in old: + continue + job, family, rule, detail = key + widening: AgentWidening = (job, rule, detail, entries[0]) + source = next((other for other in lost if other[1:] == key[1:] and left(other, entries)), None) + if source is not None: + lost.remove(source) + gains.moved.append((widening, old[source][0])) + continue + before_launches = launched_before.get((job, family), []) + if before_launches and not any(entry.get("form") == "read" for entry in before_launches): + gains.unread_before.append(widening) + continue + setting = next( + (name for entry in before_launches if (name := _expression_setting(entry, rule, detail))), None + ) + if setting is not None: + gains.expression_before.append((widening, setting)) + continue + gains.claimed.append(widening) + return gains -def agent_widenings_unread_before( - before: dict[str, Any] | None, after: dict[str, Any] | None -) -> list[AgentWidening]: - """Rules a job's launches now meet that are not claimed, because its launch before was unread.""" +def gained_agent_widenings(before: dict[str, Any] | None, after: dict[str, Any] | None) -> list[AgentWidening]: + """Documented widening rules the workflow's agent launches gain and claim (``AgentRuleGains.claimed``).""" - return _agent_rule_gains(before, after)[1] + return agent_rule_gains(before, after).claimed #: How an unresolved agent launch or checkout reads in the limit that names it. @@ -2658,9 +2858,11 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: Not blocking, like an unread secret value (#693): the launch's job, agent, form, reason and widening rules are still compared, so adding, removing or re-forming one, or gaining a documented rule, is a row. Only an edit inside - what is named here is not reported. A redacted value is not named here: it - is published redacted, and :func:`_uncompared_workflow_text` makes it a - blocking limit, as a redacted step reference is (#767). + what is named here is not reported. That includes a setting holding + credential-shaped text (#823 review): it is compared by its redacted text + and its rules, so only an edit inside what is redacted is not reported. A + redacted checkout ref is not named here: :func:`_uncompared_workflow_text` + makes it a blocking limit, as a redacted step reference is (#767). """ texts: list[str] = [] @@ -2675,11 +2877,20 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: ) for setting in entry.get("settings", []): unread = setting.get("unresolved_reason") + if unread == "redacted": + texts.append( + f"the {setting['name']} value of the agent launch at {where} contains " + "credential-shaped text; it is published redacted and compared as published, " + "so an edit inside what is redacted that gains no documented widening rule " + "is not reported" + ) + continue what = { "not_a_string": "is not a string", "unparsed_json": ( - "holds text that starts like JSON and does not parse, so the " - "values it may hold cannot be told apart from its key names" + "holds text that starts like JSON, or a codex `--config` table or array, " + "and does not parse, so the values it may hold cannot be told apart from " + "its key names" ), }.get(str(unread)) if what: @@ -2826,13 +3037,20 @@ def _uncompared_workflow_text( ) -> str | None: """Why part of a workflow grant is published but cannot be compared, or ``None``. - One rule for every compared workflow text (#767, #693). A redacted step - reference, reusable target, secret name, agent launch setting or checkout - ref (#823) could publish the same text as a different one, so comparing - the display would read a change as equal. That is a blocking limit: a - changed workflow refuses, and an unchanged one is named (#721). A URL path - an agent setting withholds is not credential-shaped and is not counted - here, as an MCP server URL's path is not (#723). + One rule for every compared workflow reference (#767, #693). A redacted + step reference, reusable target, secret name or checkout ref (#823) could + publish the same text as a different one, so comparing the display would + read a change to the code a job runs, or to where a secret goes, as equal. + That is a blocking limit: a changed workflow refuses, and an unchanged one + is named (#721). + + A redacted agent launch setting is not counted here (#823 review). Its + documented widening rules are read from the declared text before it is + redacted, so comparing its published text and its rules loses no + direction, and ordinary prose such as "never print bearer tokens" in a + system prompt is as credential-shaped to the label redaction as a token + is. :func:`uncompared_agent_launch_texts` names it, not blocking, as a + redacted label is compared by what it publishes (#802). A job id, trigger or permission scope name is compared by its published label (#802). One redacted label is still a distinct label, so it refuses @@ -2852,11 +3070,7 @@ def _uncompared_workflow_text( ("a reusable workflow secret name", any( entry["unresolved_reason"] == "redacted" for entry in mappings )), - # An agent setting and a checkout ref are compared text too (#823). - ("an agent launch setting", any( - setting.get("unresolved_reason") == "redacted" - for launch in grant.get("agent_launches", []) for setting in launch.get("settings", []) - )), + # A checkout ref names the code a job runs, as a step reference does (#823). ("a checkout ref", any( item.get("unresolved_reason") == "redacted" for item in grant.get("checkout_refs", []) )), diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index 0782299da..a7f5fe765 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -635,14 +635,25 @@ class HostWorkflowAgentSettingV7(BaseModel): values, ``apiKeyHelper`` and every secret-named value ````, a codex ``--config`` override under such a key publishes ````, and a URL publishes its scheme and host with ```` for its path - and query (#723). The rest is published through the workflow label - redaction (#802). A value it rewrites beyond that is credential-shaped: it - is published redacted with ``unresolved_reason: redacted`` and makes the - workflow a blocking limit, as a redacted step reference does (#767). A - value that is not a string (``not_a_string``), or one holding text that - starts like JSON and does not parse (``unparsed_json``), is - ``null`` and records a non-blocking coverage issue naming its + and query (#723). Each is withheld however it is attached to its flag: + ``--settings={…}`` and ``-c`` as well as a separate word. The + rest is published through the workflow label redaction (#802). A value it + rewrites beyond that is credential-shaped — a token, but also prose such + as "never print bearer tokens" — and is published redacted with + ``unresolved_reason: redacted``: it is compared as published, beside the + rules read from its declared text, and records a non-blocking coverage + issue naming its ``job/step``, because an edit inside what is redacted is + not reported. A value that is not a string (``not_a_string``), or one + holding text that starts like JSON and does not parse (``unparsed_json``), + is ``null`` and records a non-blocking coverage issue naming its ``job/step``: it is neither published nor compared. + + ``holds_expression`` is ``true`` when the declared text holds a ``${{ }}`` + expression, which GitHub substitutes before the action reads the input, + and is omitted otherwise. A documented widening rule is then read only + from the literal text the expression cannot reach, and a rule the launch + gains in the same job afterwards is not claimed, because the substituted + text may already have met it. """ model_config = ConfigDict(extra="forbid") @@ -650,14 +661,18 @@ class HostWorkflowAgentSettingV7(BaseModel): name: str value: str | None unresolved_reason: Literal["not_a_string", "redacted", "unparsed_json"] | None = None + holds_expression: bool = Field(default=False, exclude_if=lambda value: not value) class HostWorkflowAgentRuleV7(BaseModel): """One documented widening rule an agent launch meets, and the setting it was read from (#823). Decided when the workflow is read, from the declared text, before any of - it is withheld for publication, so redaction never hides a rule. Only a - literal value meets one: a value holding ``${{ }}`` meets none. + it is withheld for publication, so redaction never hides a rule. Only + literal text meets one: in a value holding ``${{ }}``, the words of + ``claude_args`` or ``codex-args`` before the first expression, less the + word it touches and a quoted run still open at it, and the entries of a + user gate that hold none; a mode input holding one meets none. ``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``, …) or the CLI flag's primary spelling. One rule compares as one whatever setting meets it, except ``open_gate``, which is one rule per gate input. @@ -728,7 +743,8 @@ class HostWorkflowCheckoutRefV7(BaseModel): checkout's default for the triggering event. A ref the label redaction rewrites is published redacted with ``unresolved_reason: redacted`` and makes the workflow a blocking limit, as a redacted step reference does - (#767). A value that is not a string, or ``with:`` that is not a mapping, + (#767): a ref names the code the job runs, as a step reference does. A + value that is not a string, or ``with:`` that is not a mapping, is ``null`` with ``unresolved_reason`` and records a non-blocking coverage issue. The ref is never resolved or fetched. """ diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 98684db06..19ce03c46 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -227,10 +227,12 @@ def paths(self) -> list[Path]: # control route, so it adds no claim; every route to the same object, # and every refusal it must keep, is held by # `tests/test_partial_host_comparison.py`. - # Its agent-launch cells and note (#823) - # restate no answer: direction comes from the engine's + # Its agent-launch cells and note (#823) restate no answer: direction + # comes from the engine's # `workflow_agent_widened_*` expansion signal, itself read off the - # `widening_rules` the engine published on each launch, and the note reads the + # `widening_rules` the engine published on each launch; the `why`'s + # moved, unread-before and expression sentences read the same + # `agent_rule_gains` the signal is computed from, and the note reads the # triggers, write scopes, secrets and checkout refs the engine already # published on the grant, so it adds no claim; # `tests/test_workflow_agent_launches.py` holds diff, verify, the PR diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 6f34aa5b0..8454d8de7 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -27,6 +27,7 @@ from agents_shipgate.core.host_grants import ( _claude_argument_input, _codex_argument_input, + _literal_argument_words, _uncompared_workflow_text, _workflow_grant, diff_host_grants, @@ -572,9 +573,14 @@ def test_renaming_or_moving_an_agent_step_within_its_job_is_quiet(): "accepts runs triggered by any user (allow-users: *)"), (_workflow({"run": "codex exec 'x'"}), _workflow({"run": "codex exec --sandbox danger-full-access 'x'"}), "runs without a sandbox"), + # Literal text an expression cannot reach still meets a rule (#823 review F4). + (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions ${{ inputs.extra }}")), + "skips permission checks"), + (_workflow(_agent()), _workflow(_agent(allowed_non_write_users="${{ vars.USERS }}, *")), + "accepts runs triggered by any user (allowed_non_write_users: *)"), ], ids=["skip-flag", "mode-flag", "gate", "bots", "codex-sandbox", "codex-unsafe", "codex-args", "codex-users", - "codex-cli"], + "codex-cli", "words-before-an-expression", "gate-entry-beside-an-expression"], ) def test_a_documented_rule_gained_is_a_widening(before, after, rule): changes = _changes(before, after) @@ -595,15 +601,22 @@ def test_a_documented_rule_gained_is_a_widening(before, after, rule): # narrowed (_workflow(_agent("--dangerously-skip-permissions")), _workflow(_agent('--allowedTools "Read"'))), (_workflow(_agent(allowed_non_write_users="*")), _workflow(_agent(allowed_non_write_users="octocat"))), - # an expression is text, never a rule - (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions ${{ inputs.extra }}"))), - (_workflow(_agent()), _workflow(_agent(allowed_non_write_users="${{ vars.USERS }}, *"))), + # text an expression can reach is never read for a rule + (_workflow(_agent()), _workflow(_agent("${{ inputs.extra }} --dangerously-skip-permissions"))), + (_workflow(_agent()), _workflow(_agent("--dangerously-skip-permissions${{ inputs.extra }}"))), + # a quote still open at the expression, which its substituted text may close + (_workflow(_agent()), + _workflow(_agent('--append-system-prompt "never pass --dangerously-skip-permissions ${{ inputs.p }}'))), + (_workflow(_agent()), _workflow(_agent(allowed_non_write_users="*${{ vars.USERS }}"))), + (_workflow({"uses": "openai/codex-action@v1"}), + _workflow({"uses": "openai/codex-action@v1", "with": {"sandbox": "${{ vars.SANDBOX }}"}})), # widened by a tool rule, which is #824's to rate (_workflow(_agent('--allowedTools "Read"')), _workflow(_agent('--allowedTools "Bash(*)"'))), (_workflow({"run": "claude -p --permission-mode default 'x'"}), _workflow({"run": "claude -p --permission-mode acceptEdits 'x'"})), ], - ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "expression", "gate-expression", "tool-rule", + ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "after-an-expression", "touching-an-expression", + "quoted-past-an-expression", "gate-entry-holding-an-expression", "mode-expression", "tool-rule", "accept-edits"], ) def test_any_other_edit_is_changed(before, after): @@ -635,6 +648,175 @@ def test_a_rule_gained_by_a_launch_read_on_both_sides_widens_beside_an_unread_on assert (row.direction, row.expands) == ("widened", True) +def _jobs(**steps): + return _workflow(jobs={job: {"runs-on": "ubuntu-latest", "steps": list(items)} for job, items in steps.items()}) + + +def _named(args, name="agent"): + return {"name": name, **_agent(args)} + + +BYPASS = "--dangerously-skip-permissions" + + +def test_renaming_a_job_that_launches_a_bypassing_agent_is_not_a_widening(): + """#823 review F3: a rule the launch already met in the job it left is moved, not gained.""" + + before, after = _jobs(review=[_named(BYPASS)]), _jobs(**{"code-review": [_named(BYPASS)]}) + + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert ( + "an agent launch that skips permission checks (bypassPermissions) moved between jobs " + "(review/agent → code-review/agent), which is not counted as a widening" + ) in row.why + assert "an agent launch now" not in row.why + + +@pytest.mark.parametrize( + ("before", "after"), + [ + # the agent step moved to another job, which launched no agent before + (_jobs(lint=[_named(BYPASS)], review=[{"run": "make"}]), + _jobs(lint=[{"run": "make"}], review=[_named(BYPASS)])), + # renamed and edited in the same change + (_jobs(review=[_named(BYPASS)]), _jobs(**{"code-review": [_named(f"{BYPASS} --max-turns 5")]})), + # the same launch now runs in the other job, and the other job's in this one + (_jobs(lint=[_named(BYPASS)], review=[_named("--allowedTools Read")]), + _jobs(lint=[_named("--allowedTools Read")], review=[_named(BYPASS)])), + # a gate the launch opened, moved with it + (_jobs(review=[{"name": "agent", **_agent(allowed_bots="*")}]), + _jobs(triage=[{"name": "agent", **_agent(allowed_bots="*")}])), + ], + ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved"], +) +def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert "moved between jobs" in row.why + + +@pytest.mark.parametrize( + ("before", "after"), + [ + # a second job now bypasses, beside the one that still does + (_jobs(lint=[_named(BYPASS)], review=[_named("--allowedTools Read")]), + _jobs(lint=[_named(BYPASS)], review=[_named(BYPASS)])), + # the job that met the rule still launches the agent, and a different launch meets it elsewhere + (_jobs(lint=[_named(BYPASS)], review=[_named("--allowedTools Read")]), + _jobs(lint=[_named("--allowedTools Read")], review=[_named(f"{BYPASS} --max-turns 5")])), + ], + ids=["second-job", "narrowed-there-widened-here"], +) +def test_a_rule_another_job_gains_while_no_launch_left_is_a_widening(before, after): + assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (review/agent)" in row.why + + +@pytest.mark.parametrize( + ("value", "words"), + [ + ("--dangerously-skip-permissions --model ${{ vars.M }}", ("--dangerously-skip-permissions", "--model")), + ("--model ${{ vars.M }} --dangerously-skip-permissions", ("--model",)), + # the word the expression touches + ("--dangerously-skip-permissions${{ vars.X }}", ()), + ("--permission-mode=${{ vars.MODE }}", ()), + # a quoted run open at the expression, balanced or not in the literal text + ('--append-system-prompt "--dangerously-skip-permissions ${{ vars.X }}"', ("--append-system-prompt",)), + ('--append-system-prompt "a --dangerously-skip-permissions ${{ vars.X }}', ("--append-system-prompt",)), + # a whole line before it, and a comment line the action drops whatever it holds + ("--dangerously-skip-permissions\n# model: ${{ vars.M }}", ("--dangerously-skip-permissions",)), + ("# don't\n--dangerously-skip-permissions ${{ vars.X }}", ("--dangerously-skip-permissions",)), + # an unquoted `#` ends the input, so what follows it reaches nothing + ("--dangerously-skip-permissions # ${{ vars.X }}", ("--dangerously-skip-permissions",)), + ], +) +def test_claude_args_rules_are_read_only_from_words_an_expression_cannot_reach(value, words): + """#823 review F4: GitHub substitutes the expression before the action splits the input.""" + + assert _literal_argument_words("claude", value) == words + + +@pytest.mark.parametrize( + ("value", "words"), + [ + ("--yolo ${{ vars.X }}", ("--yolo",)), + ("${{ vars.X }} --yolo", ()), + ('--yolo "a ${{ vars.X }}', ("--yolo",)), + ('["--yolo", "${{ vars.X }}"]', ("--yolo",)), + ('["${{ vars.X }}", "--yolo"]', ()), + # outside a JSON string, the substituted text decides whether the array parses + ('["--yolo", ${{ vars.X }}]', None), + ], +) +def test_codex_args_rules_are_read_only_from_words_an_expression_cannot_reach(value, words): + assert _literal_argument_words("codex", value) == words + + +def test_a_setting_holding_an_expression_is_marked_and_the_row_says_what_it_leaves_unread(): + """#823 review F4: the row no longer reads as though no documented rule was gained.""" + + before = _workflow(_agent("--model ${{ vars.CLAUDE_MODEL }}\n--allowedTools Read")) + after = _workflow(_agent("--model ${{ vars.CLAUDE_MODEL }}\n--dangerously-skip-permissions")) + + launch, = _launches(after) + setting, = launch["settings"] + assert setting["holds_expression"] is True + assert "widening_rules" not in launch + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert ( + "an agent launch setting holds a `${{ }}` expression (claude_args at review/steps[0]), which GitHub " + "substitutes before the action reads it; documented widening rules are read only from the literal " + "text the expression cannot reach, so this row does not say whether the text it reaches meets one" + ) in row.why + # A setting without one says nothing of the kind, and the key is omitted. + plain, = _launches(_workflow(_agent())) + assert "holds_expression" not in plain["settings"][0] + + +@pytest.mark.parametrize( + ("before", "after", "rule", "setting"), + [ + (_workflow(_agent(allowed_non_write_users="${{ vars.EXTRA_USERS }}")), + _workflow(_agent(allowed_non_write_users="${{ vars.EXTRA_USERS }}, *")), + "accepts runs triggered by any user (allowed_non_write_users: *)", "allowed_non_write_users"), + (_workflow(_agent("--allowedTools Read --model ${{ vars.CLAUDE_MODEL }}")), + _workflow(_agent("--dangerously-skip-permissions --model ${{ vars.CLAUDE_MODEL }}")), + "skips permission checks (bypassPermissions)", "claude_args"), + (_workflow(_agent("--model ${{ vars.M }} --dangerously-skip-permissions")), + _workflow(_agent("--model opus --dangerously-skip-permissions")), + "skips permission checks (bypassPermissions)", "claude_args"), + ], + ids=["gate", "claude-args", "expression-replaced"], +) +def test_a_rule_gained_where_the_setting_held_an_expression_before_is_named_and_not_claimed( + before, after, rule, setting +): + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert ( + f"an agent launch now {rule} (review/steps[0]), which is not counted as a widening: before, this " + f"job's {setting} held a `${{{{ }}}}` expression, whose substituted text this audit does not read" + ) in row.why + + +def test_replacing_an_expression_beside_a_rule_the_launch_already_met_gains_nothing(): + before = _workflow(_agent("--dangerously-skip-permissions --model ${{ vars.CLAUDE_MODEL }}")) + after = _workflow(_agent("--dangerously-skip-permissions --model opus")) + + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert "an agent launch now" not in row.why + + def test_a_new_workflow_that_bypasses_permissions_is_an_added_widening(): after = _grant(_workflow(_agent("--dangerously-skip-permissions"))) changes = diff_host_grants({"grants": []}, {"grants": [after]}) @@ -746,6 +928,37 @@ def test_a_json_value_publishes_only_what_the_host_readers_publish(): assert uncompared_agent_launch_texts(grant) == [] and _uncompared_workflow_text(grant) is None +def test_a_value_attached_to_its_flag_is_withheld_as_a_separate_word_is(): + """#823 review F1: `--settings=…`, `--mcp-config=…` and codex's `-c` published verbatim.""" + + grant = _grant(_workflow( + _agent(f"--allowedTools Read --settings='{SETTINGS_JSON}' --mcp-config='{MCP_JSON}'"), + {"uses": "openai/codex-action@v1", "with": { + "codex-args": '-cmcp_servers.db.env.REGION="canary-short" -c=mcp_servers.x.env.T=canary-eq --json', + }}, + )) + claude, codex = grant["agent_launches"] + + assert claude["settings"] == [{ + "name": "claude_args", + "value": f"--allowedTools Read '--settings={SETTINGS_PUBLISHED}' '--mcp-config={MCP_PUBLISHED}'", + "unresolved_reason": None, + }] + assert codex["settings"] == [{ + "name": "codex-args", + "value": "'-cmcp_servers.db.env.REGION=' '-c=mcp_servers.x.env.T=' --json", + "unresolved_reason": None, + }] + text = json.dumps(grant) + for canary in (*JSON_CANARIES, "canary-short", "canary-eq"): + assert canary not in text + # Each spelling compares as the host readers compare it: rotating an env value is quiet. + rotated = SETTINGS_JSON.replace("hunter2-canary", "rotated") + assert _rows( + _workflow(_agent(f"--settings='{SETTINGS_JSON}'")), _workflow(_agent(f"--settings='{rotated}'")) + ) == [] + + def test_a_withheld_json_value_compares_as_the_host_readers_compare_it(): def settings(env): return _workflow(_agent(settings=json.dumps({"env": env}))) @@ -782,6 +995,19 @@ def test_a_codex_config_override_withholds_env_header_and_secret_values(): assert "canary" not in json.dumps(grant) +def test_a_codex_config_table_that_does_not_parse_is_withheld_and_named(): + """#823 review (P3): string-argv keeps the quotes of `--config='k={…}'`, so it does not parse as TOML.""" + + codex_args = "--config='mcp_servers.db={command=\"x\", env={T=\"canary-quoted\"}}' --json" + grant = _grant(_workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}})) + launch, = grant["agent_launches"] + + assert launch["settings"] == [{"name": "codex-args", "value": None, "unresolved_reason": "unparsed_json"}] + assert "canary-quoted" not in json.dumps(grant) + limit, = uncompared_agent_launch_texts(grant) + assert "a codex `--config` table or array, and does not parse" in limit + + def test_text_that_starts_like_json_and_does_not_parse_is_withheld_and_named(): # shell-quote strips the double quotes of an unquoted JSON word, so the # action reads it as a path; its values cannot be told from its keys. @@ -795,7 +1021,7 @@ def test_text_that_starts_like_json_and_does_not_parse_is_withheld_and_named(): assert "canary-unquoted" not in json.dumps(grant) limit, = uncompared_agent_launch_texts(grant) assert limit.startswith("the claude_args value of the agent launch at review/steps[0] (anthropics/claude-code-action)") - assert "starts like JSON and does not parse" in limit + assert "starts like JSON, or a codex `--config` table or array, and does not parse" in limit assert _uncompared_workflow_text(grant) is None @@ -831,7 +1057,13 @@ def marketplace(url): assert _rows(marketplace("https://github.com/org/a.git"), marketplace("https://github.com/org/b.git")) == [] -def test_credential_shaped_text_is_published_redacted_and_blocks_the_workflow_as_a_step_reference_does(): +def test_credential_shaped_text_in_a_setting_is_published_redacted_and_named_while_a_ref_blocks(): + """#823 review F2: a redacted setting compares by its published text and rules, not blocking. + + A checkout ref names the code a job runs, as a step reference does, so a + redacted one still refuses (#767). + """ + grant = _grant(_workflow( _agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions", plugin_marketplaces="https://robot:PWCANARY@github.com/org/repo.git"), @@ -857,13 +1089,72 @@ def test_credential_shaped_text_is_published_redacted_and_blocks_the_workflow_as text = json.dumps(grant) for canary in ("ARGCANARY", "PWCANARY", "REFCANARY", "SECRETCANARY", "ghp_"): assert canary not in text - assert uncompared_agent_launch_texts(grant) == [] + assert uncompared_agent_launch_texts(grant) == [ + f"the {name} value of the agent launch at {where} contains credential-shaped text; it is published " + "redacted and compared as published, so an edit inside what is redacted that gains no documented " + "widening rule is not reported" + for name, where in ( + ("claude_args", "review/steps[0] (anthropics/claude-code-action)"), + ("plugin_marketplaces", "review/steps[0] (anthropics/claude-code-action)"), + ("--settings", "review/steps[2] (claude)"), + ) + ] assert _uncompared_workflow_text(grant) == ( - "an agent launch setting and a checkout ref contain credential-shaped text; " - "they are published redacted and cannot be compared" + "a checkout ref contains credential-shaped text; it is published redacted and cannot be compared" ) +#: Prose the #802 label redaction rewrites, as security-review prompts write it: +#: ``claude_args``, and a ``run:`` whose prompt is a variadic flag's value. +PROSE = [ + ('--append-system-prompt "Never print bearer tokens in review comments" --allowedTools Read', + "claude -p --allowedTools Read 'Never print bearer tokens in review comments'"), + ('--append-system-prompt "Flag Authorization: headers logged in plain text" --allowedTools Read', + "claude -p --allowedTools Read 'Flag Authorization: headers logged in plain text'"), + ('--allowedTools "Bash(curl -H Authorization:*)"', + "claude -p --allowedTools 'Bash(curl -H Authorization:*)' -- 'go'"), + ('--append-system-prompt "check the secret=... assignment" --allowedTools Read', + "claude -p --allowedTools Read 'check the secret=... assignment'"), +] + + +@pytest.mark.parametrize(("prose", "run"), PROSE, ids=["bearer", "authorization", "tool-rule", "assignment"]) +def test_prose_the_label_redaction_rewrites_is_a_named_limit_that_refuses_nothing(prose, run): + """#823 review F2: such prose used to make the whole workflow a blocking limit.""" + + for step in (_agent(prose), {"run": run}): + grant = _grant(_workflow(step)) + launch, = grant["agent_launches"] + assert [setting["unresolved_reason"] for setting in launch["settings"]] == ["redacted"] + assert _uncompared_workflow_text(grant) is None + limit, = uncompared_agent_launch_texts(grant) + assert "contains credential-shaped text" in limit + + # A permission change beside it keeps its row, and a rule gained beside it widens. + row, = _rows(_workflow(_agent(prose)), _workflow(_agent(prose), permissions={"pull-requests": "write"})) + assert row.direction == "widened" and "grants write permissions to workflow jobs" in row.why + row, = _rows(_workflow(_agent(prose)), _workflow(_agent(f"{prose} --dangerously-skip-permissions"))) + assert (row.direction, row.expands) == ("widened", True) + # Each cell shows the redacted text it is compared by, so the two sides differ. + assert "" in row.before and row.after.endswith("--dangerously-skip-permissions") + + +def test_an_expression_in_a_url_is_read_as_one_word_so_the_url_is_withheld_whole(): + """#823 review (P3): the expression's spaces used to split the URL, publishing its path.""" + + launch, = _launches(_workflow(_agent( + plugin_marketplaces="https://x-access-token:${{ secrets.MARKET_TOKEN }}@github.com/acme/market.git", + ))) + setting = next(item for item in launch["settings"] if item["name"] == "plugin_marketplaces") + assert setting == { + "name": "plugin_marketplaces", "value": "https://github.com/", + "unresolved_reason": "redacted", "holds_expression": True, + } + # An expression outside a URL is published as written. + ref, = _grant(_reproduction(ref=HEAD_SHA))["checkout_refs"] + assert ref["ref"] == HEAD_SHA + + def test_token_shaped_job_and_step_labels_are_redacted_in_every_entry(): job = "ghp_" + "B" * 36 grant = _grant(_workflow(jobs={job: {"steps": [ @@ -1156,7 +1447,15 @@ def test_no_canary_reaches_any_published_output(tmp_path): "command": "db-mcp", "env": {"DB_API_TOKEN": "canary-tok-123"}, "headers": {"X-API-Key": "canary-hdr-456", "Authorization": f"Bearer {canary}"}, }}}) - codex_args = "-c 'mcp_servers.db.env.TOKEN=\"canary-cfg-789\"' --full-auto" + codex_args = ( + "-c 'mcp_servers.db.env.TOKEN=\"canary-cfg-789\"' -cmcp_servers.db.env.REGION=canary-short-c " + "-c=mcp_servers.db.env.ZONE=canary-eq-c --full-auto" + ) + # The `=` spellings of #823 review F1, beside the separate-word ones. + equals = json.dumps({"env": {"DB_PASSWORD": "hunter2-eqcanary"}, "apiKeyHelper": "echo helper-eqcanary"}) + equals_mcp = json.dumps({"mcpServers": {"db": { + "command": "db-mcp", "env": {"DB_API_TOKEN": "tok-eqcanary"}, "headers": {"X-API-Key": "hdr-eqcanary"}, + }}}) base = _workflow(jobs={job: {"steps": [_agent()]}}) head = _workflow(jobs={job: {"steps": [ {"name": "Pull docker://ci:" + "p4ssCANARY" + "@gcr.io/x", "uses": "actions/checkout@v4"}, @@ -1165,6 +1464,7 @@ def test_no_canary_reaches_any_published_output(tmp_path): settings=SETTINGS_JSON, mcp_config=mcp, plugin_marketplaces="https://github.com/canary-org/canary-repo.git", ), + _agent(f"--allowedTools Read --settings='{equals}' --mcp-config='{equals_mcp}'"), {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read --mcp-config '{mcp}' 'go'"}, {"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}, ]}}) @@ -1183,21 +1483,19 @@ def test_no_canary_reaches_any_published_output(tmp_path): _assert_absent(joined, ( canary, "p4ssCANARY", job, "canary-org", "canary-repo", "canary-cfg-789", *JSON_CANARIES, + "hunter2-eqcanary", "helper-eqcanary", "tok-eqcanary", "hdr-eqcanary", "canary-short-c", "canary-eq-c", )) assert "runs claude -p with --allowedTools Read" in joined assert "mcp_servers.db.env.TOKEN=" in joined -def test_a_credential_shaped_setting_or_ref_refuses_a_changed_workflow_and_publishes_no_canary(tmp_path): +def test_a_credential_shaped_ref_refuses_a_changed_workflow_and_publishes_no_canary(tmp_path): base = _workflow(_agent()) - head = _workflow( - {"uses": "actions/checkout@v4", "with": {"ref": "token=REFCANARY"}}, - _agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions"), - ) + head = _workflow({"uses": "actions/checkout@v4", "with": {"ref": "token=REFCANARY"}}, _agent()) repo = _repo(tmp_path, {SOURCE: _yaml(base)}) _git(repo, "checkout", "-qb", "change") _write(repo, {SOURCE: _yaml(head)}) - _git(repo, "commit", "-qam", "credential-shaped values") + _git(repo, "commit", "-qam", "credential-shaped ref") payload = _diff(repo) assert payload["comparison_status"] == "incomparable" and payload["rows"] == [] @@ -1205,5 +1503,69 @@ def test_a_credential_shaped_setting_or_ref_refuses_a_changed_workflow_and_publi audit = json.loads(CliRunner().invoke(app, ["audit", "--host", "--workspace", str(repo), "--json"]).stdout) issue, = [item for item in audit["issues"] if item["host"] == "github"] assert (issue["kind"], issue["blocking"]) == ("unsupported", True) - assert "an agent launch setting and a checkout ref contain credential-shaped text" in issue["message"] - _assert_absent(_published_outputs(repo), ("ARGCANARY", "REFCANARY")) + assert "a checkout ref contains credential-shaped text" in issue["message"] + _assert_absent(_published_outputs(repo), ("REFCANARY",)) + + +def test_a_credential_shaped_setting_compares_on_every_route_and_publishes_no_canary(tmp_path): + base = _workflow(_agent()) + head = _workflow(_agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions")) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "credential-shaped setting") + + payload = _diff(repo) + assert payload["comparison_status"] == "comparable" + row, = payload["rows"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert "--append-system-prompt 'use token=' --dangerously-skip-permissions" in row["after"] + audit = json.loads(CliRunner().invoke(app, ["audit", "--host", "--workspace", str(repo), "--json"]).stdout) + issue, = [item for item in audit["issues"] if item["host"] == "github"] + assert (issue["kind"], issue["blocking"]) == ("unsupported", False) + github, = [item for item in audit["host_coverage"] if item["host"] == "github"] + assert github["status"] == "complete" + _assert_absent(_published_outputs(repo), ("ARGCANARY",)) + + +def _prose_repo(tmp_path: Path, head: dict, extra: dict[str, str] | None = None) -> Path: + base = _workflow(_agent(PROSE[0][0])) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head), **(extra or {})}) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "change") + return repo + + +def test_a_permission_change_beside_redacted_prose_keeps_its_row_on_every_route(tmp_path): + """#823 review F2 (a): the prose used to refuse the comparison and hide this row.""" + + repo = _prose_repo(tmp_path, _workflow(_agent(PROSE[0][0]), permissions={"contents": "read", "pull-requests": "write"})) + + payload = _diff(repo) + assert payload["comparison_status"] == "comparable" + row, = payload["rows"] + assert row["direction"] == "widened" and "grants write permissions to workflow jobs" in row["why"] + result = CliRunner().invoke(app, ["verify", "--workspace", str(repo), "--base", "main", "--head", "HEAD"]) + assert result.exit_code == 0, result.output + verifier = json.loads((repo / "agents-shipgate-reports/verifier.json").read_text()) + assert verifier["host_comparison"]["comparison_status"] == "comparable" + verified, = verifier["host_comparison"]["rows"] + assert verified["direction"] == "widened" + assert "Host capability comparison unavailable" not in (repo / "agents-shipgate-reports/pr-comment.md").read_text() + + +def test_an_unchanged_workflow_holding_redacted_prose_leaves_check_comparable(tmp_path): + """#823 review F2 (b): an unchanged workflow used to make every `check` incomparable (#721).""" + + mcp = json.dumps({"mcpServers": {"docs": {"command": "docs-mcp"}}}) + repo = _prose_repo(tmp_path, _workflow(_agent(PROSE[0][0])), extra={".mcp.json": mcp}) + + args = ["check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "agent-boundary-json"] + result = CliRunner().invoke(app, args) + assert result.exit_code == 0, result.output + boundary = json.loads(result.output) + assert boundary["comparison_status"] == "comparable", boundary + row, = boundary["rows"] + assert row["direction"] == "added" and ".mcp.json" in row["subject"] From 2e365537963e920b4c653a76247e4df18376af48 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Tue, 22 Sep 2026 23:02:39 -0700 Subject: [PATCH 05/14] Address review cycle 2 on agent launches in CI (#823) The third review of #850 at c7f67550 found one P1 and one P2 defect and four P3 notes. The branch is on origin/main 44b9e05d; no rebase was needed. A JSON setting publishes its shape, not its free text (C2-F1). _withheld_json published the whole _redact_secret_values tree. That tree is the host readers' digest input, not what they publish: it keeps every string outside env, headers and secret-named keys. So an mcp-remote --header "Authorization: Bearer ..." argument in --mcp-config, and a hook's curl command in settings, reached diff text, diff --json, audit --host --json, the PR comment and verifier.json. _json_shape now keeps key names, numbers, booleans and null, redacts what the host readers redact, and replaces each other string with , a 12-hex digest of redacted_config_sha256 for that string, so an edit to it is still a changed row. It keeps only the strings a host reader publishes: - a permissions.allow/ask/deny rule and a documented Claude Code setting's value (defaultMode, the switches, enabledMcpjsonServers entries); - an MCP server's command name and its URL's scheme and host, followed by the digest when they drop a command's arguments or a URL's query. A codex --config table or array is read under its key path, so mcp_servers.gh={command="gh", ...} keeps its command name. The canary sweep adds the mcp-remote header and the hook command in every spelling (action input, claude_args, CLI flag, codex -c table). Re-running the reviewer's two repositories through ./shipgate gives 0 canaries in every output. The documented bypasses written through listed inputs widen (C2-F2). - Claude Code settings written as JSON meet bypass_permissions when their defaultMode is bypassPermissions, read by claude_setting_values as the settings reader reads .claude/settings.json. This covers the action's settings input, a --settings value in claude_args, and the CLI's --settings flag. A path is not read, and a settings value holding an expression meets none. - openai/codex-action's permission-profile: :danger-full-access meets danger_full_access. ":danger-full-access" is Codex's reserved name for its built-in full-access profile (BUILT_IN_PERMISSION_PROFILE_DANGER_FULL_ACCESS). Mode inputs are now (input, value, rule) triples. settings and permission-profile join _RULE_SETTINGS, so a gain after an expression in either is named and not claimed, as for claude_args. P3 notes: - allowed_bots: "*" now reads "accepts runs triggered by any bot" through agent_rule_text. - Which of two gaining jobs a moved rule goes to no longer depends on declaration order: the same launch arriving is matched before a job that merely stopped launching the agent. The support page, the STABILITY migration note, the contract summary, the schema docstrings and the CHANGELOG entry now say what a structured value publishes instead of claiming it publishes what the host readers would, and list the two rules. host-grants 0.7 is extended in place; it is unreleased. --- CHANGELOG.md | 2 +- STABILITY.md | 14 +- docs/agent-contract-current.md | 15 +- docs/host-boundary-support.md | 81 ++++-- docs/host-grants-baseline-schema.v0.7.json | 4 +- docs/host-grants-inventory-schema.v0.7.json | 4 +- llms-full.txt | 15 +- .../core/capability_diff_rows.py | 13 +- src/agents_shipgate/core/host_grants.py | 258 ++++++++++++++---- src/agents_shipgate/schemas/host_grants.py | 50 ++-- tests/test_workflow_agent_launches.py | 164 ++++++++++- 11 files changed, 499 insertions(+), 121 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index cb26d2274..c6261132c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks, a bypassed or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A setting publishes what the host readers would: a JSON object its key names, with `env` and `headers` values and `apiKeyHelper` withheld however it is attached to its flag, and a URL its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A JSON object in a setting, however it is attached to its flag, publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index c1211aedc..4d879309f 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -32,9 +32,13 @@ launch meets are read from its declared text and published as `widening_rules`, and `claude_args` and `codex-args` are split as each action splits them, and only from literal text a `${{ }}` expression cannot reach (a setting holding one says so with `holds_expression`); a rule a launch already -met in a job it left is moved, not gained. A setting publishes what the host -readers would: a JSON object its key names, however it is attached to its flag, -a URL its scheme and host. A compound command, an expansion or an expression in +met in a job it left is moved, not gained. A JSON object in a setting, however +it is attached to its flag, publishes its shape and none of its free text: key +names, with each string a `` digest except those a host reader +publishes (a permission rule, a documented setting's value, an MCP server's +command name and URL host), so an MCP server's arguments and a hook's command +are compared but never published; a URL publishes its scheme and host. A +compound command, an expansion or an expression in `run:` is `unresolved`, a named non-blocking limit that leaves coverage complete; a setting holding credential-shaped text, prose included, is published redacted, compared as published and named the same way, while a @@ -353,8 +357,8 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i - **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. -- **What is withheld.** A setting publishes what the host readers would publish for the same text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — publishes its key names, with `env` and `headers` values, `apiKeyHelper` and every secret-named value ``, as canonical JSON; a codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. A URL publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions`, one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox, `safety-strategy: unsafe`, or a `*` entry in `allowed_bots`, `allowed_non_write_users` or `allow-users`. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox` or `safety-strategy` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **What is withheld.** A structured value publishes its shape and none of its free text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex `--config` table or array publish, as canonical JSON, their key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. A codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. Other argument text — a prompt, a flag's value, a codex `--config` override's scalar value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to such a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access`, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. - **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index f89d3882a..e80b77e79 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -58,11 +58,16 @@ with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, `widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, `null` for the default. Values are compared as text and never executed; `claude_args` and `codex-args` are split -as each action splits them, and a setting publishes what the host readers -would, a JSON object by its key names and a URL by its scheme and host. Only a -documented rule a job's launches gain — bypassed permission checks, a bypassed -or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate -opened to `*` — raises `workflow_agent_widened_` and makes the +as each action splits them. A JSON object in a setting publishes its shape and +none of its free text — key names, with each string a `` digest +except those a host reader publishes (a permission rule, a documented +setting's value, an MCP server's command name and URL host) — so an MCP +server's arguments and a hook's command are compared but never published; a +URL publishes its scheme and host. Only a documented rule a job's launches +gain — bypassed permission checks (a flag, or JSON settings whose +`defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` +sandbox (`permission-profile: :danger-full-access` included), +`safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`. A rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left, or where the job's launch before was unread or held an expression the rule is read from, is named diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 658ec1173..9df589871 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -137,9 +137,9 @@ things are listed on the workflow grant, each naming its `job/step` (the step's | Action | Inputs compared as text | Documented widening | |---|---|---| - | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions`; `allowed_bots` or `allowed_non_write_users` gains a `*` entry | - | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` as above | - | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access`; `allow-users` gains a `*` entry | + | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions`, `--permission-mode bypassPermissions` or a JSON `--settings` value whose `defaultMode` is `bypassPermissions`; `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | + | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` and `settings` as above | + | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access`; `allow-users` gains a `*` entry | No shell reads `claude_args` or `codex-args`: each action splits its own input, and a rule is met only by the words the action passes on. The @@ -164,8 +164,10 @@ things are listed on the workflow grant, each naming its `job/step` (the step's `--allow-dangerously-skip-permissions`, `--allowedTools`/`--allowed-tools`, `--disallowedTools`/`--disallowed-tools`, `--add-dir`, `--mcp-config`, `--settings` and `--permission-prompt-tool`; gaining - `--dangerously-skip-permissions` or `--permission-mode bypassPermissions` - (one rule, so moving between the two spellings is not a widening) widens. + `--dangerously-skip-permissions`, `--permission-mode bypassPermissions` or a + JSON `--settings` value whose `defaultMode` is `bypassPermissions` (one + rule, so moving between the spellings is not a widening) widens; a + `--settings` path names a file this audit does not read. For `codex exec`: `--sandbox`/`-s`, `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, `--approve-for-me`/`--not-so-yolo`, `--dangerously-bypass-hook-trust`, @@ -202,8 +204,9 @@ an input before the action reads it, and the substituted text may be anything `claude_args` or `codex-args` before the first expression, less the word it touches and any quoted run still open at it (in a JSON-array `codex-args`, the elements before the one holding it), and the entries of a user gate that hold -no expression, so `"${{ vars.USERS }}, *"` opens the gate. A `sandbox` or -`safety-strategy` value holding one meets none. Such a setting is published +no expression, so `"${{ vars.USERS }}, *"` opens the gate. A `sandbox`, +`permission-profile`, `safety-strategy` or `settings` value holding one meets +none. Such a setting is published with `holds_expression: true`, and a row that changes it says the text the expression reaches is not read for a rule, rather than that none was gained. @@ -216,8 +219,10 @@ names the rule and step. Three gains are named in the `why` and not claimed: launch may already have met it, as a job whose permissions were not explicit may already have held a write scope; - where the job's launch of that agent held a `${{ }}` expression before in an - input the rule is read from (`claude_args` for bypassed permission checks, - the gate for a `*` entry), because the substituted text may already have met + input the rule is read from (`claude_args` or `settings` for bypassed + permission checks, `sandbox`, `permission-profile` or `codex-args` for a + full-access sandbox, the gate for a `*` entry), because the substituted text + may already have met it — so replacing `--model ${{ vars.M }}` with `--model opus` beside `--dangerously-skip-permissions` is not a widening; - where the rule moved between jobs: another job met it before and the launch @@ -243,23 +248,47 @@ code (`github.event.pull_request.head.sha`, `.head.ref` or `.merge_commit_sha`, direction, and `if:` conditions and the default checkout of a `pull_request` event are not read into it. A removed workflow gets no note. -A setting publishes what the host readers would publish for the same text. A -JSON object — a `settings` or `mcp_config` value, a `--settings` or -`--mcp-config` value written as its own word or as `--settings={…}`, or any -word of `claude_args` or `codex-args` — publishes -as `.claude/settings.json` and `.mcp.json` do: its key names, with `env` and -`headers` values, `apiKeyHelper` and every other secret-named value read as -``, in canonical JSON, so rotating an `env` value or reordering keys -compares as unchanged and adding a key is a change. A codex `--config` -override — `-c key=value`, `--config=key=value`, `-ckey=value` or -`-c=key=value` — under `env`, `headers` or a secret-named key publishes -`` for its value, and a table or array value by the same JSON rule. A -URL publishes its scheme, host and port, with `` for any path and -no query, as an MCP server's URL does (#723), and the rest of the setting is -compared, so a change only to a URL's path or query — which repository a -`plugin_marketplaces` URL names, for one — is not reported; a zero-row result -says redacted values are not compared. A `${{ }}` expression is one word while -this is decided, so one inside a URL's userinfo is withheld with it. +A structured value publishes its shape and none of its free text, so a setting +never publishes what `.claude/settings.json` and `.mcp.json` would withhold +(#823 review). A JSON object — a `settings` or `mcp_config` value, a +`--settings` or `--mcp-config` value written as its own word or as +`--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex +`--config` table or array publish, in canonical JSON: + +- their key names, numbers, booleans and `null`, so reordering keys compares + as unchanged and adding a key is a change; +- `` for `env` and `headers` values (and codex's `http_headers` and + `env_http_headers`), `apiKeyHelper`, every other secret-named value and the + word after a secret-named argument such as `--token`, as the host readers + redact them, so rotating one compares as unchanged; +- each other string as ``, a short digest of what the host + readers digest for it: the sanitized text, and a URL's query. Editing it is + a `changed` row, and none of its text is published. So an MCP server's + `args` — an `mcp-remote --header "Authorization: Bearer …"` included — and a + hook's command, matcher and type publish only digests; +- except the strings a host reader publishes: a `permissions.allow`, `ask` or + `deny` rule, and the value of a documented Claude Code setting + (`defaultMode`, the switches in [the ratings table](#claude-code-setting-ratings), + `enabledMcpjsonServers` entries), as the settings reader publishes them; and, + under `mcpServers` (or `mcp_servers`), a server's command name and its URL's + scheme and host, as the MCP reader publishes them. Each is followed by the + digest when it drops something the digest reads: `"command":"npx "` + for `npx -y some-server`, a URL's digest for its query. A URL's path is + neither published nor compared, as an MCP server's is not (#723). + +A codex `--config` override — `-c key=value`, `--config=key=value`, +`-ckey=value` or `-c=key=value` — under `env`, `headers` or a secret-named key +publishes `` for its value; a table or array value is read under its +key path, so `mcp_servers.gh={command="gh", …}` is a server whose command name +is kept. Other argument text — a prompt, a flag's value, a codex `--config` +override's scalar value such as `model="o3"` — is published as written through +the label redaction below, except that a URL in it publishes its scheme, host +and port, with `` for any path and no query, as an MCP server's +URL does (#723); the rest of the setting is compared, so a change only to such +a URL's path or query — which repository a `plugin_marketplaces` URL names, for +one — is not reported, and a zero-row result says redacted values are not +compared. A `${{ }}` expression is one word while this is decided, so one +inside a URL's userinfo is withheld with it. Any other text the #802 label redaction rewrites is credential-shaped: a token shape, a credential assignment such as `token=…`, a bearer or header value, a diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 2deac0c4f..7eade822f 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1365,7 +1365,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode input holding one meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode or ``settings`` input holding one meets\nnone. Claude Code settings written as JSON \u2014 the ``settings`` input, or a\n``--settings`` value in ``claude_args`` or on the CLI \u2014 meet\n``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``,\nread as the settings reader reads it; a path to a settings file is not\nread. ``setting`` is the input (``claude_args``, ``allowed_bots``,\n``sandbox``, ``permission-profile``, \u2026) or the CLI flag's primary\nspelling. One rule compares as one whatever setting meets it, except\n``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1392,7 +1392,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). Each is withheld however it is attached to its flag:\n``--settings={\u2026}`` and ``-c`` as well as a separate word. The\nrest is published through the workflow label redaction (#802). A value it\nrewrites beyond that is credential-shaped \u2014 a token, but also prose such\nas \"never print bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. A\nstructured value \u2014 a JSON object (a ``settings`` or ``mcp_config`` value,\na ``--settings`` or ``--mcp-config`` value, any argument word), or a codex\n``--config`` table or array \u2014 publishes its shape and none of its free\ntext: key names, numbers, booleans and ``null``, with each string\nreplaced by ````, a short digest of what the host readers\ndigest for it, so an edit to it is still a change. ``env`` and\n``headers`` values, ``apiKeyHelper`` and every secret-named value are\n````, as the host readers redact them. The strings a host reader\npublishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a\ndocumented Claude Code setting's value such as ``defaultMode``, and an\nMCP server's command name and its URL's scheme and host, each followed by\nthe digest when it drops something the digest reads (a command's\narguments, a URL's query). So an MCP server's arguments and a hook's\ncommand publish nothing, as `.mcp.json` and `.claude/settings.json` do\nnot (#823 review). A codex ``--config`` override under ``env``,\n``headers`` or a secret-named key publishes ```` for its value,\nand a URL elsewhere publishes its scheme and host with\n```` for its path and query (#723). Each is withheld\nhowever it is attached to its flag: ``--settings={\u2026}`` and\n``-c`` as well as a separate word. Other argument text \u2014 a\nprompt, a flag's value, a codex ``--config`` override's scalar value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", "properties": { "holds_expression": { "default": false, diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index eec78d398..6eac7688a 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1423,7 +1423,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode input holding one meets none.\n``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``,\n\u2026) or the CLI flag's primary spelling. One rule compares as one whatever\nsetting meets it, except ``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode or ``settings`` input holding one meets\nnone. Claude Code settings written as JSON \u2014 the ``settings`` input, or a\n``--settings`` value in ``claude_args`` or on the CLI \u2014 meet\n``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``,\nread as the settings reader reads it; a path to a settings file is not\nread. ``setting`` is the input (``claude_args``, ``allowed_bots``,\n``sandbox``, ``permission-profile``, \u2026) or the CLI flag's primary\nspelling. One rule compares as one whatever setting meets it, except\n``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1450,7 +1450,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. What the\nhost readers withhold stays withheld: a JSON object (a ``settings`` or\n``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any\nargument word) publishes its key names with ``env`` and ``headers``\nvalues, ``apiKeyHelper`` and every secret-named value ````, a\ncodex ``--config`` override under such a key publishes ````, and\na URL publishes its scheme and host with ```` for its path\nand query (#723). Each is withheld however it is attached to its flag:\n``--settings={\u2026}`` and ``-c`` as well as a separate word. The\nrest is published through the workflow label redaction (#802). A value it\nrewrites beyond that is credential-shaped \u2014 a token, but also prose such\nas \"never print bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. A\nstructured value \u2014 a JSON object (a ``settings`` or ``mcp_config`` value,\na ``--settings`` or ``--mcp-config`` value, any argument word), or a codex\n``--config`` table or array \u2014 publishes its shape and none of its free\ntext: key names, numbers, booleans and ``null``, with each string\nreplaced by ````, a short digest of what the host readers\ndigest for it, so an edit to it is still a change. ``env`` and\n``headers`` values, ``apiKeyHelper`` and every secret-named value are\n````, as the host readers redact them. The strings a host reader\npublishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a\ndocumented Claude Code setting's value such as ``defaultMode``, and an\nMCP server's command name and its URL's scheme and host, each followed by\nthe digest when it drops something the digest reads (a command's\narguments, a URL's query). So an MCP server's arguments and a hook's\ncommand publish nothing, as `.mcp.json` and `.claude/settings.json` do\nnot (#823 review). A codex ``--config`` override under ``env``,\n``headers`` or a secret-named key publishes ```` for its value,\nand a URL elsewhere publishes its scheme and host with\n```` for its path and query (#723). Each is withheld\nhowever it is attached to its flag: ``--settings={\u2026}`` and\n``-c`` as well as a separate word. Other argument text \u2014 a\nprompt, a flag's value, a codex ``--config`` override's scalar value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", "properties": { "holds_expression": { "default": false, diff --git a/llms-full.txt b/llms-full.txt index c89be0865..bf6692d19 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1632,11 +1632,16 @@ with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, `widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is each `actions/checkout` step's `with.ref`, `null` for the default. Values are compared as text and never executed; `claude_args` and `codex-args` are split -as each action splits them, and a setting publishes what the host readers -would, a JSON object by its key names and a URL by its scheme and host. Only a -documented rule a job's launches gain — bypassed permission checks, a bypassed -or `danger-full-access` sandbox, `safety-strategy: unsafe`, or a user gate -opened to `*` — raises `workflow_agent_widened_` and makes the +as each action splits them. A JSON object in a setting publishes its shape and +none of its free text — key names, with each string a `` digest +except those a host reader publishes (a permission rule, a documented +setting's value, an MCP server's command name and URL host) — so an MCP +server's arguments and a hook's command are compared but never published; a +URL publishes its scheme and host. Only a documented rule a job's launches +gain — bypassed permission checks (a flag, or JSON settings whose +`defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` +sandbox (`permission-profile: :danger-full-access` included), +`safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`. A rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left, or where the job's launch before was unread or held an expression the rule is read from, is named diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index a94d5345b..bb02719c3 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -28,10 +28,10 @@ from agents_shipgate.core.host_grants import ( AGENT_RULE_INPUTS, - AGENT_WIDENING_RULES, UNTRUSTED_INPUT_TRIGGERS, agent_launch_key, agent_rule_gains, + agent_rule_text, checkout_ref_key, hook_loading_basis, host_grant_expansion_signals, @@ -337,34 +337,31 @@ def _agent_launch_reasons( def where(item: dict[str, Any]) -> str: return f"{item['job']}/{item['step']}" - def meets(rule: str, detail: str) -> str: - return AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") - gains = agent_rule_gains(before, after) reasons: list[str] = [] # Launches a sentence about a rule already names. named: set[str] = set() for _job, rule, detail, entry in gains.claimed: named.add(where(entry)) - reasons.append(f"an agent launch now {meets(rule, detail)} ({where(entry)})") + reasons.append(f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)})") for _job, rule, detail, entry in gains.unread_before: named.add(where(entry)) reasons.append( - f"an agent launch now {meets(rule, detail)} ({where(entry)}), which is not counted as a " + f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)}), which is not counted as a " "widening: before, this job launched the agent in a form this audit does not read, which " "may already have done the same" ) for (_job, rule, detail, entry), setting in gains.expression_before: named.add(where(entry)) reasons.append( - f"an agent launch now {meets(rule, detail)} ({where(entry)}), which is not counted as a " + f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)}), which is not counted as a " f"widening: before, this job's {setting} held a " + "`${{ }}`" + " expression, whose " "substituted text this audit does not read and which may already have done the same" ) for (_job, rule, detail, entry), source in gains.moved: named.update({where(entry), where(source)}) reasons.append( - f"an agent launch that {meets(rule, detail)} moved between jobs ({where(source)} → " + f"an agent launch that {agent_rule_text(rule, detail)} moved between jobs ({where(source)} → " f"{where(entry)}), which is not counted as a widening: the launch already met that " "rule in the job it left, and it now runs with the receiving job's token permissions" ) diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 72e100bc6..b4b1c617f 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -54,6 +54,7 @@ ) from agents_shipgate.core.host_settings import ( CLAUDE_LIST_SETTINGS, + CLAUDE_SCALAR_SETTINGS, claude_setting_values, rate_claude_setting, ) @@ -1620,17 +1621,21 @@ class _AgentAction: """What one documented agent action's inputs mean to this reader (#823). ``inputs`` are compared as text. ``gates`` are comma-separated user lists - where a ``*`` entry opens the gate to every user. ``args`` names the input + where a ``*`` entry opens the gate to every user (to every bot, for + ``allowed_bots``). ``args`` names the input that carries agent CLI arguments, read with the family's flag table for - the documented widening rules. ``modes`` are inputs whose value is itself - a documented widening. + the documented widening rules. ``modes`` are ``(input, value, rule)``: an + input whose value is itself a documented widening. ``settings`` names the + input that holds Claude Code settings, JSON or a path; written as JSON, its + ``defaultMode`` is read as the settings reader reads it (#823 review). """ family: Literal["claude", "codex"] inputs: tuple[str, ...] gates: tuple[str, ...] = () args: str | None = None - modes: tuple[tuple[str, str], ...] = () + modes: tuple[tuple[str, str, str], ...] = () + settings: str | None = None _CLAUDE_BASE_ACTION = _AgentAction( @@ -1640,6 +1645,7 @@ class _AgentAction: "plugin_marketplaces", "plugins", "settings", ), args="claude_args", + settings="settings", ) #: The documented agent actions, by ``owner/repo`` (matched case-insensitively, @@ -1658,6 +1664,7 @@ class _AgentAction: ), gates=("allowed_bots", "allowed_non_write_users"), args="claude_args", + settings="settings", ), "anthropics/claude-code-base-action": _CLAUDE_BASE_ACTION, "anthropics/claude-code-action/base-action": _CLAUDE_BASE_ACTION, @@ -1669,7 +1676,14 @@ class _AgentAction: ), gates=("allow-users",), args="codex-args", - modes=(("sandbox", "danger-full-access"), ("safety-strategy", "unsafe")), + # `:danger-full-access` is Codex's reserved name for its built-in + # full-access permission profile, the profile form of the + # `danger-full-access` sandbox (#823 review). + modes=( + ("sandbox", "danger-full-access", "danger_full_access"), + ("permission-profile", ":danger-full-access", "danger_full_access"), + ("safety-strategy", "unsafe", "unsafe_safety_strategy"), + ), ), } @@ -1712,7 +1726,11 @@ class _AgentAction: AGENT_RULE_INPUTS: frozenset[str] = frozenset( name for spec in _AGENT_ACTIONS.values() - for name in (*spec.gates, *(mode for mode, _value in spec.modes), *((spec.args,) if spec.args else ())) + for name in ( + *spec.gates, + *(mode for mode, _value, _rule in spec.modes), + *(item for item in (spec.args, spec.settings) if item), + ) ) #: The widening each documented rule names, as a row's ``why`` says it. @@ -1724,6 +1742,18 @@ class _AgentAction: "open_gate": "accepts runs triggered by any user", } +#: A user gate whose ``*`` entry admits any bot rather than any user. +_BOT_GATES = frozenset({"allowed_bots"}) + + +def agent_rule_text(rule: str, detail: str) -> str: + """How a row's ``why`` names one documented rule: ``detail`` is the gate an ``open_gate`` opened.""" + + if rule == "open_gate" and detail in _BOT_GATES: + return f"accepts runs triggered by any bot ({detail}: *)" + return AGENT_WIDENING_RULES[rule] + (f" ({detail}: *)" if detail else "") + + #: A literal checkout ref that names pull request code, not the base branch's: #: the documented pull request and workflow-run head expressions, and #: ``refs/pull//head`` or ``/merge``. @@ -2153,20 +2183,114 @@ def _literal_argument_words(family: str, text: str) -> tuple[str, ...] | None: #: codex host reader never publishes, as ``.mcp.json`` ``env``/``headers`` do not. _CONFIG_WITHHELD_KEYS = frozenset({"env", "headers", "http_headers", "env_http_headers"}) +#: Keys whose value maps MCP server names to servers, anywhere in a structured +#: value: Claude's ``mcpServers`` and codex's ``mcp_servers``. VS Code's +#: ``servers`` counts only at the top, where `.vscode/mcp.json` puts it. +_SERVER_MAP_KEYS = frozenset({"mcpServers", "mcp_servers"}) +_TOP_LEVEL_SERVER_MAP_KEYS = _SERVER_MAP_KEYS | {"servers"} +#: Where, from the top of a Claude Code settings value, the settings reader +#: publishes a string as it is: a ``permissions.allow``/``ask``/``deny`` rule, +#: and the value of a documented setting, under ``permissions`` or at the top. +_PUBLISHED_STRING_PATHS: frozenset[tuple[str, ...]] = frozenset({ + *(("permissions", name) for name in ("allow", "ask", "deny", *CLAUDE_SCALAR_SETTINGS)), + *((name,) for name in (*CLAUDE_SCALAR_SETTINGS, *CLAUDE_LIST_SETTINGS)), +}) + + +def _withheld_string(value: str, label: str | None = None) -> str: + """One string of a structured value as it is published (#823 review C2-F1). + + ````: a short digest of what the host readers digest for it — + the sanitized text, and a URL's query as an MCP server's is (#723) — so + editing it is still a change while none of its text is published. Where a + host reader publishes a label for the string (``label``: an MCP server's + command name or URL, a permission rule, a documented setting's value), the + label is published, followed by the digest only when the label drops + something the digest reads, such as a command's arguments or a URL's query. + """ + + withheld = f"" + if label is None: + return withheld + if label == _sanitize_sensitive_string(value) and not _url_capability_parts(value): + return label + return f"{label} {withheld}" + + +def _server_command(command: str) -> str: + """An MCP server's ``command`` as the MCP reader names it: its first word's file name.""" + + first = command.strip().split(maxsplit=1)[0] if command.strip() else "" + return _withheld_string(command, _sanitize_sensitive_string(Path(first).name or first) or None) + -def _withheld_json(value: Any) -> str | None: - """A JSON value as the host readers publish one: key names, secret-bearing values replaced. +def _json_shape(value: Any, path: tuple[str, ...] = ()) -> Any: + """A structured value's shape: what a host reader would publish of it, and no free text. - ``env`` and ``headers`` keep their keys with every value ````, - ``apiKeyHelper`` and every other secret-named key's value is ````, - and strings go through the host sanitizer (URLs, bearer and header - values), exactly as `.claude/settings.json` and `.mcp.json` are read. - Canonical, so reformatting or reordering keys changes nothing. + Key names, numbers, booleans and ``null`` are kept. Secret-bearing values + are ```` exactly as the host readers redact them before any + digest: ``env`` and ``headers`` (and codex's ``http_headers`` and + ``env_http_headers``) keep their keys, a secret-named key's value + (``apiKeyHelper``) and the word after a secret-named argument are + replaced, and ``policyHelper`` is excluded. Every other string is + :func:`_withheld_string`, keeping only what a host reader publishes: a + ``permissions.allow``/``ask``/``deny`` rule and a documented Claude Code + setting's value (``defaultMode``, ``enabledMcpjsonServers`` entries) at the + top of the value, as the settings reader publishes them; and an MCP + server's command name and URL scheme and host, as the MCP reader publishes + them. So an MCP server's arguments and a hook's command publish nothing + of their text (#823 review C2-F1). ``path`` is the keys above ``value``. + """ + + parent = path[-1] if path else None + if isinstance(value, dict): + in_server = len(path) >= 2 and ( + path[-2] in _SERVER_MAP_KEYS or (len(path) == 2 and path[0] in _TOP_LEVEL_SERVER_MAP_KEYS) + ) + shaped: dict[str, Any] = {} + for key, inner in value.items(): + key_text = str(key) + if key_text == "policyHelper": + shaped[key_text] = "" + elif _is_secret_key(key_text) or parent in _CREDENTIAL_CONTAINER_KEYS: + shaped[key_text] = "" + elif key_text in _CONFIG_WITHHELD_KEYS and isinstance(inner, dict): + shaped[key_text] = {str(name): "" for name in inner} + elif in_server and key_text == "command" and isinstance(inner, str): + shaped[key_text] = _server_command(inner) + elif in_server and key_text in {"url", "serverUrl"} and isinstance(inner, str): + shaped[key_text] = _withheld_string(inner, _sanitize_url(inner)) + else: + shaped[key_text] = _json_shape(inner, (*path, key_text)) + return shaped + if isinstance(value, list): + items: list[Any] = [] + redact_next = False + for item in value: + if redact_next: + items.append("") + redact_next = False + continue + if isinstance(item, str) and not _SECRET_ARG_RE.fullmatch(item): + redact_next = item.lower().lstrip("-").replace("-", "_") in _SECRET_KEY_MARKERS + items.append(_json_shape(item, path)) + return items + if isinstance(value, str): + published = path in _PUBLISHED_STRING_PATHS + return _withheld_string(value, _sanitize_sensitive_string(value) if published else None) + return value + + +def _withheld_json(value: Any, path: tuple[str, ...] = ()) -> str | None: + """A structured value as it may be published: its :func:`_json_shape`, as canonical JSON. + + Canonical, so reformatting or reordering keys changes nothing. ``path`` is + the keys above ``value``, for a codex ``--config`` override's value. """ try: return json.dumps( - _redact_secret_values(value), sort_keys=True, separators=(",", ":"), + _json_shape(value, path), sort_keys=True, separators=(",", ":"), ensure_ascii=False, default=str, ) except (RecursionError, TypeError, ValueError): @@ -2194,10 +2318,12 @@ def _withheld_config(text: str) -> str | None: A key path through ``env``, ``headers`` or a secret-named key publishes ```` for its value; a table or array value, parsed as TOML as - codex parses it, publishes by :func:`_withheld_json`. A value that starts - like a table or array and does not parse — string-argv keeps the quotes of - ``--config='k={…}'`` — is ``None``: what it holds cannot be told apart from - its keys. Anything else is kept. + codex parses it, publishes its shape by :func:`_withheld_json`, read under + its key path, so ``mcp_servers.gh={command="gh", …}`` is a server whose + command name is kept. A value that starts like a table or array and does + not parse — string-argv keeps the quotes of ``--config='k={…}'`` — is + ``None``: what it holds cannot be told apart from its keys. Anything else, + a scalar value such as ``model="o3"``, is argument text and is kept. """ key, equals, value = text.partition("=") @@ -2211,7 +2337,7 @@ def _withheld_config(text: str) -> str | None: except (tomllib.TOMLDecodeError, RecursionError): return None if value.strip().strip("\"'").startswith(("{", "[")) else text if isinstance(loaded, (dict, list)): - shown = _withheld_json(loaded) + shown = _withheld_json(loaded, tuple(segments)) return None if shown is None else f"{key}={shown}" return text @@ -2257,9 +2383,9 @@ def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> list[ def _withheld_arguments(family: str, value: str) -> str | None: """An argument input as it may be published: the text the action parses, JSON words withheld. - A word the host readers would not publish is replaced, quoted, by what they - would; every other character stays as declared. A JSON array - ``codex-args`` publishes as its array of withheld words. + A JSON word, or a codex ``--config`` override, is replaced, quoted, by its + shape (:func:`_withheld_json`); every other character stays as declared. + A JSON array ``codex-args`` publishes as its array of withheld words. """ parsed = _argument_input(family, value) @@ -2609,25 +2735,48 @@ def checkout_ref_key(entry: dict[str, Any]) -> tuple[str, str, str]: ) +def _settings_bypass_permissions(text: str) -> bool: + """Whether Claude Code settings written as JSON set ``defaultMode: bypassPermissions`` (#823 review). + + Read as the settings reader reads `.claude/settings.json` + (:func:`claude_setting_values`): under ``permissions``, else at the top. + A path, or text that does not parse, meets nothing: no file is read. + """ + + if not text.lstrip().startswith("{"): + return False + try: + loaded = json.loads(text) + except (ValueError, RecursionError): + return False + return any( + item.setting == "defaultMode" and item.value == "bypassPermissions" + for item in claude_setting_values(loaded) + ) + + def _claude_action_rules(words: tuple[str, ...]) -> set[str]: """The widening rules the words a Claude action passes on meet. Read as ``parse-sdk-options.ts`` reads them: a word starting with ``--`` is always a flag and never another flag's value, so ``--dangerously-skip-permissions`` counts wherever it stands, and - ``--permission-mode`` takes the next word unless that starts with ``--``. + ``--permission-mode`` and ``--settings`` take the next word unless that + starts with ``--``. A ``--settings`` value written as JSON meets the rule + its ``defaultMode`` sets. """ rules: set[str] = set() for index, word in enumerate(words): name, equals, attached = word.partition("=") + following = words[index + 1] if index + 1 < len(words) else "" + value = attached if equals else ("" if following.startswith("--") else following) if name == "--dangerously-skip-permissions": rules.add("bypass_permissions") - elif name == "--permission-mode": - following = words[index + 1] if index + 1 < len(words) else "" - mode = attached if equals else ("" if following.startswith("--") else following) - if mode == "bypassPermissions": - rules.add("bypass_permissions") + elif name == "--permission-mode" and value == "bypassPermissions": + rules.add("bypass_permissions") + elif name == "--settings" and _settings_bypass_permissions(value): + rules.add("bypass_permissions") return rules @@ -2642,6 +2791,7 @@ def _flag_rules( if family == "claude" and ( name == "--dangerously-skip-permissions" or (name == "--permission-mode" and value == "bypassPermissions") + or (name == "--settings" and value is not None and _settings_bypass_permissions(value)) ): rules.add(("bypass_permissions", name)) if family == "codex" and name == "--dangerously-bypass-approvals-and-sandbox": @@ -2659,8 +2809,8 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu expression into the input before the action reads it, so a rule is read only where the substituted text cannot reach — an argument input's words before the first expression (:func:`_literal_argument_words`), a user - gate's entries that hold none — and a mode input holding one meets none. - Each rule names the input it was read from. + gate's entries that hold none — and a mode or settings input holding one + meets none. Each rule names the input it was read from. """ rules: set[tuple[str, str]] = set() @@ -2673,9 +2823,11 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu entries = _EXPRESSION_SPAN_RE.sub(_EXPRESSION_MARK, text).split(",") if "*" in {entry.strip() for entry in entries}: rules.add(("open_gate", name)) - for mode, widening in spec.modes: + for mode, widening, rule in spec.modes: if name == mode and text == widening: - rules.add(("danger_full_access" if mode == "sandbox" else "unsafe_safety_strategy", name)) + rules.add((rule, name)) + if name == spec.settings and not holds_expression(text) and _settings_bypass_permissions(text): + rules.add(("bypass_permissions", name)) if name == spec.args: words = _literal_argument_words(spec.family, text) if words is None: @@ -2727,9 +2879,9 @@ def agent_family(agent: str) -> str: #: The agent action inputs each documented rule is read from; a user gate's #: rule is read from the gate it names. _RULE_SETTINGS: dict[str, frozenset[str]] = { - "bypass_permissions": frozenset({"claude_args"}), + "bypass_permissions": frozenset({"claude_args", "settings"}), "bypass_approvals_and_sandbox": frozenset({"codex-args"}), - "danger_full_access": frozenset({"sandbox", "codex-args"}), + "danger_full_access": frozenset({"sandbox", "permission-profile", "codex-args"}), "unsafe_safety_strategy": frozenset({"safety-strategy"}), } @@ -2801,27 +2953,41 @@ def launches(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[st old, new = met(before), met(after) launched_before, launched_after = launches(before), launches(after) lost = [key for key in old if key not in new] + gained = [key for key in new if key not in old] - def left(key: _RuleKey, arriving: list[dict[str, Any]]) -> bool: - # The launch that met the rule in the losing job left it: the job no - # longer launches that agent, or the same launch now runs elsewhere. - if (key[0], key[1]) not in launched_after: - return True + def same_launch(key: _RuleKey, arriving: list[dict[str, Any]]) -> bool: + # The launch that met the rule in the losing job now runs here. return any( agent_launch_key(entry)[1:] == agent_launch_key(other)[1:] for entry in old[key] for other in arriving ) + def job_left(key: _RuleKey, _arriving: list[dict[str, Any]]) -> bool: + # The losing job no longer launches that agent at all. + return (key[0], key[1]) not in launched_after + + # A rule moved when the launch that met it left the losing job. The same + # launch arriving is matched first, so which of two gaining jobs a rule + # moved to does not depend on the order the jobs are declared in (#823 review). + sources: dict[_RuleKey, _RuleKey] = {} + for left in (same_launch, job_left): + for key in gained: + if key in sources: + continue + source = next( + (other for other in lost if other[1:] == key[1:] and left(other, new[key])), None + ) + if source is not None: + lost.remove(source) + sources[key] = source + gains = AgentRuleGains(claimed=[], unread_before=[], expression_before=[], moved=[]) - for key, entries in new.items(): - if key in old: - continue + for key in gained: + entries = new[key] job, family, rule, detail = key widening: AgentWidening = (job, rule, detail, entries[0]) - source = next((other for other in lost if other[1:] == key[1:] and left(other, entries)), None) - if source is not None: - lost.remove(source) - gains.moved.append((widening, old[source][0])) + if key in sources: + gains.moved.append((widening, old[sources[key]][0])) continue before_launches = launched_before.get((job, family), []) if before_launches and not any(entry.get("form") == "read" for entry in before_launches): diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index a7f5fe765..e870dad0b 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -628,18 +628,31 @@ class HostWorkflowAgentSettingV7(BaseModel): flag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too). ``value`` is the declared text, stripped, as it may be published; a flag that takes no value has ``null``. ``claude_args`` is the text the Claude - actions parse, without the full-line ``#`` comments they drop. What the - host readers withhold stays withheld: a JSON object (a ``settings`` or - ``mcp_config`` value, a ``--settings`` or ``--mcp-config`` value, any - argument word) publishes its key names with ``env`` and ``headers`` - values, ``apiKeyHelper`` and every secret-named value ````, a - codex ``--config`` override under such a key publishes ````, and - a URL publishes its scheme and host with ```` for its path - and query (#723). Each is withheld however it is attached to its flag: - ``--settings={…}`` and ``-c`` as well as a separate word. The - rest is published through the workflow label redaction (#802). A value it - rewrites beyond that is credential-shaped — a token, but also prose such - as "never print bearer tokens" — and is published redacted with + actions parse, without the full-line ``#`` comments they drop. A + structured value — a JSON object (a ``settings`` or ``mcp_config`` value, + a ``--settings`` or ``--mcp-config`` value, any argument word), or a codex + ``--config`` table or array — publishes its shape and none of its free + text: key names, numbers, booleans and ``null``, with each string + replaced by ````, a short digest of what the host readers + digest for it, so an edit to it is still a change. ``env`` and + ``headers`` values, ``apiKeyHelper`` and every secret-named value are + ````, as the host readers redact them. The strings a host reader + publishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a + documented Claude Code setting's value such as ``defaultMode``, and an + MCP server's command name and its URL's scheme and host, each followed by + the digest when it drops something the digest reads (a command's + arguments, a URL's query). So an MCP server's arguments and a hook's + command publish nothing, as `.mcp.json` and `.claude/settings.json` do + not (#823 review). A codex ``--config`` override under ``env``, + ``headers`` or a secret-named key publishes ```` for its value, + and a URL elsewhere publishes its scheme and host with + ```` for its path and query (#723). Each is withheld + however it is attached to its flag: ``--settings={…}`` and + ``-c`` as well as a separate word. Other argument text — a + prompt, a flag's value, a codex ``--config`` override's scalar value — is + published through the workflow label redaction (#802). A value it + rewrites is credential-shaped — a token, but also prose such as "never + print bearer tokens" — and is published redacted with ``unresolved_reason: redacted``: it is compared as published, beside the rules read from its declared text, and records a non-blocking coverage issue naming its ``job/step``, because an edit inside what is redacted is @@ -672,10 +685,15 @@ class HostWorkflowAgentRuleV7(BaseModel): literal text meets one: in a value holding ``${{ }}``, the words of ``claude_args`` or ``codex-args`` before the first expression, less the word it touches and a quoted run still open at it, and the entries of a - user gate that hold none; a mode input holding one meets none. - ``setting`` is the input (``claude_args``, ``allowed_bots``, ``sandbox``, - …) or the CLI flag's primary spelling. One rule compares as one whatever - setting meets it, except ``open_gate``, which is one rule per gate input. + user gate that hold none; a mode or ``settings`` input holding one meets + none. Claude Code settings written as JSON — the ``settings`` input, or a + ``--settings`` value in ``claude_args`` or on the CLI — meet + ``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``, + read as the settings reader reads it; a path to a settings file is not + read. ``setting`` is the input (``claude_args``, ``allowed_bots``, + ``sandbox``, ``permission-profile``, …) or the CLI flag's primary + spelling. One rule compares as one whatever setting meets it, except + ``open_gate``, which is one rule per gate input. """ model_config = ConfigDict(extra="forbid") diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 8454d8de7..a28f584eb 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -557,8 +557,9 @@ def test_renaming_or_moving_an_agent_step_within_its_job_is_quiet(): "skips permission checks"), (_workflow(_agent(allowed_non_write_users="octocat")), _workflow(_agent(allowed_non_write_users="octocat, *")), "accepts runs triggered by any user (allowed_non_write_users: *)"), - (_workflow(_agent()), _workflow(_agent(allowed_bots="*")), - "accepts runs triggered by any user (allowed_bots: *)"), + # The bot gate admits any bot, not any user (#823 review cycle 2). + (_workflow(_agent()), _workflow(_agent(allowed_bots="dependabot,*")), + "accepts runs triggered by any bot (allowed_bots: *)"), (_workflow({"uses": "openai/codex-action@v1", "with": {"sandbox": "read-only"}}), _workflow({"uses": "openai/codex-action@v1", "with": {"sandbox": "danger-full-access"}}), "runs without a sandbox (danger-full-access)"), @@ -578,9 +579,25 @@ def test_renaming_or_moving_an_agent_step_within_its_job_is_quiet(): "skips permission checks"), (_workflow(_agent()), _workflow(_agent(allowed_non_write_users="${{ vars.USERS }}, *")), "accepts runs triggered by any user (allowed_non_write_users: *)"), + # #823 review C2-F2: the documented bypasses written through inputs the reader lists. + (_workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "default"}}))), + _workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "bypassPermissions"}}))), + "skips permission checks (bypassPermissions)"), + (_workflow(_agent()), _workflow(_agent(settings=json.dumps({"defaultMode": "bypassPermissions"}))), + "skips permission checks (bypassPermissions)"), + (_workflow(_agent()), + _workflow(_agent("--settings '{\"permissions\":{\"defaultMode\":\"bypassPermissions\"}}'")), + "skips permission checks (bypassPermissions)"), + (_workflow({"run": "claude -p 'x'"}), + _workflow({"run": "claude -p --settings '{\"permissions\":{\"defaultMode\":\"bypassPermissions\"}}' 'x'"}), + "skips permission checks (bypassPermissions)"), + (_workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":workspace"}}), + _workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":danger-full-access"}}), + "runs without a sandbox (danger-full-access)"), ], ids=["skip-flag", "mode-flag", "gate", "bots", "codex-sandbox", "codex-unsafe", "codex-args", "codex-users", - "codex-cli", "words-before-an-expression", "gate-entry-beside-an-expression"], + "codex-cli", "words-before-an-expression", "gate-entry-beside-an-expression", "settings-default-mode", + "settings-top-level-default-mode", "args-settings", "cli-settings", "codex-permission-profile"], ) def test_a_documented_rule_gained_is_a_widening(before, after, rule): changes = _changes(before, after) @@ -614,10 +631,22 @@ def test_a_documented_rule_gained_is_a_widening(before, after, rule): (_workflow(_agent('--allowedTools "Read"')), _workflow(_agent('--allowedTools "Bash(*)"'))), (_workflow({"run": "claude -p --permission-mode default 'x'"}), _workflow({"run": "claude -p --permission-mode acceptEdits 'x'"})), + # one rule, written as a flag and as the settings it passes + (_workflow(_agent("--dangerously-skip-permissions")), + _workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "bypassPermissions"}})))), + # settings holding an expression, or naming a file, meet no rule + (_workflow(_agent()), + _workflow(_agent(settings='{"permissions":{"defaultMode":"${{ vars.MODE }}"}}'))), + (_workflow(_agent()), _workflow(_agent(settings=".github/claude-settings.json"))), + (_workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "default"}}))), + _workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "acceptEdits"}})))), + (_workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":read-only"}}), + _workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":workspace"}})), ], ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "after-an-expression", "touching-an-expression", "quoted-past-an-expression", "gate-entry-holding-an-expression", "mode-expression", "tool-rule", - "accept-edits"], + "accept-edits", "flag-to-settings", "settings-expression", "settings-path", "settings-accept-edits", + "codex-workspace-profile"], ) def test_any_other_edit_is_changed(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] @@ -717,6 +746,21 @@ def test_a_rule_another_job_gains_while_no_launch_left_is_a_widening(before, aft assert "an agent launch now skips permission checks (bypassPermissions) (review/agent)" in row.why +@pytest.mark.parametrize("renamed_first", [True, False], ids=["renamed-declared-first", "renamed-declared-last"]) +def test_a_renamed_job_takes_the_move_whatever_order_the_jobs_are_declared_in(renamed_first): + """#823 review cycle 2 (P3): the rule moved to the job running the same launch, not the first one declared.""" + + renamed = ("code-review", [_named(BYPASS)]) + added = ("triage", [_named(f"{BYPASS} --max-turns 5")]) + before = _jobs(review=[_named(BYPASS)]) + after = _jobs(**dict([renamed, added] if renamed_first else [added, renamed])) + + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert "moved between jobs (review/agent → code-review/agent)" in row.why + assert "an agent launch now skips permission checks (bypassPermissions) (triage/agent)" in row.why + + @pytest.mark.parametrize( ("value", "words"), [ @@ -792,8 +836,12 @@ def test_a_setting_holding_an_expression_is_marked_and_the_row_says_what_it_leav (_workflow(_agent("--model ${{ vars.M }} --dangerously-skip-permissions")), _workflow(_agent("--model opus --dangerously-skip-permissions")), "skips permission checks (bypassPermissions)", "claude_args"), + # the settings input is read for the rule only when it holds no expression (#823 review C2-F2) + (_workflow(_agent(settings='{"permissions":{"defaultMode":"${{ vars.MODE }}"}}')), + _workflow(_agent(settings='{"permissions":{"defaultMode":"bypassPermissions"}}')), + "skips permission checks (bypassPermissions)", "settings"), ], - ids=["gate", "claude-args", "expression-replaced"], + ids=["gate", "claude-args", "expression-replaced", "settings"], ) def test_a_rule_gained_where_the_setting_held_an_expression_before_is_named_and_not_claimed( before, after, rule, setting @@ -904,6 +952,105 @@ def test_a_removed_workflow_gets_no_note(): ) JSON_CANARIES = ("hunter2-canary", "canary-9f8e7d", "canary-helper-value", "canary-tok-123", "canary-hdr-456") +#: #823 review C2-F1: text the host readers never publish, in no secret-named +#: key: an `mcp-remote` bearer header among a server's `args`, and a hook's command. +REMOTE_MCP_JSON = json.dumps({"mcpServers": {"remote": {"command": "npx", "args": [ + "mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY0123456789abcdef", +]}}}, separators=(",", ":")) +HOOK_JSON = json.dumps({"hooks": {"Stop": [{"hooks": [{ + "type": "command", "command": 'curl -H "X-Auth-Token: hookCANARY77" https://hooks.example.com/notify', +}]}]}}, separators=(",", ":")) +REMOTE_CODEX_CONFIG = ( + 'mcp_servers.remote={command="npx", args=["mcp-remote", "https://mcp.example.com/sse", ' + '"--header", "Authorization: Bearer tokCANARY-codex-0123456789"]}' +) +SHAPE_CANARIES = ( + "tokCANARY0123456789abcdef", "tokCANARY-codex-0123456789", "hookCANARY77", "hooks.example.com", + "mcp.example.com", "mcp-remote", "X-Auth-Token", +) + + +def _digest(text): + """What one withheld string publishes: a digest of what the host readers digest for it.""" + + from agents_shipgate.core.host_grants import redacted_config_sha256 + + return f"" + + +def test_a_json_value_publishes_its_shape_and_none_of_its_free_text(): + """#823 review C2-F1: every string a host reader does not publish is withheld, and still compared. + + The same server in `.mcp.json` publishes `remote (command name npx)`, and + the same hook in `.claude/settings.json` publishes `Stop`; neither + publishes an argument or a command. + """ + + grant = _grant(_workflow( + _agent(f"--allowedTools Read --mcp-config '{REMOTE_MCP_JSON}'", settings=HOOK_JSON), + {"run": f"codex exec -c '{REMOTE_CODEX_CONFIG}' 'go'"}, + )) + action, cli = grant["agent_launches"] + args = ["mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY0123456789abcdef"] + server = json.dumps( + {"mcpServers": {"remote": {"args": [_digest(arg) for arg in args], "command": "npx"}}}, + separators=(",", ":"), + ) + hook = ( + '{"hooks":{"Stop":[{"hooks":[{"command":"' + + _digest('curl -H "X-Auth-Token: hookCANARY77" https://hooks.example.com/notify') + + '","type":"' + _digest("command") + '"}]}]}}' + ) + assert {item["name"]: item["value"] for item in action["settings"]} == { + "claude_args": f"--allowedTools Read --mcp-config '{server}'", + "settings": hook, + } + codex_args = ["mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY-codex-0123456789"] + assert cli["settings"] == [{ + "name": "--config", + "value": 'mcp_servers.remote={"args":[' + ",".join(f'"{_digest(arg)}"' for arg in codex_args) + + '],"command":"npx"}', + "unresolved_reason": None, + }] + text = json.dumps(grant) + for canary in SHAPE_CANARIES: + assert canary not in text + # Nothing was redacted, so no limit is named: nothing of the text is published to redact. + assert uncompared_agent_launch_texts(grant) == [] + + # A withheld string is still compared: a new argument or command is a row. + edited = HOOK_JSON.replace("curl -H", "wget --header") + row, = _rows(_workflow(_agent(settings=HOOK_JSON)), _workflow(_agent(settings=edited))) + assert (row.direction, row.expands) == ("changed", False) + assert row.before != row.after and "wget" not in row.after + # A documented setting's value, a permission rule and an enabled server are + # published as the settings reader publishes them; other strings are not. + launch, = _launches(_workflow(_agent(settings=json.dumps({ + "permissions": {"defaultMode": "acceptEdits", "allow": ["Bash(npm test)"], "additionalDirectories": ["../x"]}, + "enabledMcpjsonServers": ["github"], "model": "claude-opus", + })))) + setting = next(item for item in launch["settings"] if item["name"] == "settings") + assert setting["value"] == ( + '{"enabledMcpjsonServers":["github"],"model":"' + _digest("claude-opus") + '",' + '"permissions":{"additionalDirectories":["' + _digest("../x") + '"],"allow":["Bash(npm test)"],' + '"defaultMode":"acceptEdits"}}' + ) + + +def test_an_mcp_server_url_publishes_its_scheme_and_host_and_compares_its_query_as_the_mcp_reader_does(): + def config(url): + return _workflow(_agent(mcp_config=json.dumps({"mcpServers": {"db": {"url": url}}}))) + + launch, = _launches(config("https://mcp.example.com/v1/sse")) + assert { + "name": "mcp_config", "value": '{"mcpServers":{"db":{"url":"https://mcp.example.com/"}}}', + "unresolved_reason": None, + } in launch["settings"] + # A query decides which tools a server exposes, so it is compared (#723); a path is not. + row, = _rows(config("https://mcp.example.com/sse?read_only=true"), config("https://mcp.example.com/sse")) + assert "read_only" not in row.before + row.after + assert _rows(config("https://mcp.example.com/a"), config("https://mcp.example.com/b")) == [] + def test_a_json_value_publishes_only_what_the_host_readers_publish(): grant = _grant(_workflow( @@ -1467,6 +1614,11 @@ def test_no_canary_reaches_any_published_output(tmp_path): _agent(f"--allowedTools Read --settings='{equals}' --mcp-config='{equals_mcp}'"), {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read --mcp-config '{mcp}' 'go'"}, {"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}, + # #823 review C2-F1: an MCP server's arguments and a hook's command, + # which the host readers never publish, in every spelling. + _agent(f"--allowedTools Read\n--mcp-config '{REMOTE_MCP_JSON}'", settings=HOOK_JSON), + {"run": f"claude -p --settings '{HOOK_JSON}' --mcp-config '{REMOTE_MCP_JSON}' 'go'"}, + {"run": f"codex exec -c '{REMOTE_CODEX_CONFIG}' 'go'"}, ]}}) repo = _repo(tmp_path, {SOURCE: _yaml(base)}) _git(repo, "checkout", "-qb", "change") @@ -1484,9 +1636,11 @@ def test_no_canary_reaches_any_published_output(tmp_path): _assert_absent(joined, ( canary, "p4ssCANARY", job, "canary-org", "canary-repo", "canary-cfg-789", *JSON_CANARIES, "hunter2-eqcanary", "helper-eqcanary", "tok-eqcanary", "hdr-eqcanary", "canary-short-c", "canary-eq-c", + *SHAPE_CANARIES, )) assert "runs claude -p with --allowedTools Read" in joined assert "mcp_servers.db.env.TOKEN=" in joined + assert '"command":"npx"' in joined and '"Stop":[{"hooks":[{"command":" Date: Wed, 23 Sep 2026 00:10:47 -0700 Subject: [PATCH 06/14] Address review cycle 3 on agent launches in CI (#823) Rebased onto origin/main 01777037 (#852, #821), which had already moved the unreleased runtime contract 40 -> 41, verifier 0.20 -> 0.21 and capability diff 0.3 -> 0.4. This change now extends contract 41 in place instead of minting it: one comment in schemas/contract.py names #821 and #823, and the texts that said verifier 0.20 and capability diff 0.3 were unchanged now say capability_diff registry row keeps #852's two added roots and both paragraphs; STABILITY keeps the #821, #823 and #827 notes, with #821's "host-grants stays 0.6" and #827's version sentence corrected for a tree that also carries host-grants 0.7; CHANGELOG keeps both Unreleased entries. llms-full.txt is rebuilt. The pilot ledger's source-tree column was re-measured on the rebased tree beside e3c6cb0c: identical cells except contract 40/41 and inventory schema 0.6/0.7, identical diff text, rows, check JSON and inventory apart from its schema version, and diff --json / verifier.json differing only in #821's schema versions and coverage members. A codex exec step that selects the full-access sandbox through --config was a changed row while --sandbox danger-full-access widened, and -sdanger-full-access was not read at all. The flag reader now reads a short flag's attached value as clap does (-s, -s=, -c, -c=), and a --config override that sets sandbox_mode to danger-full-access, or default_permissions to :danger-full-access (what codex-action's permission-profile input passes the CLI), meets danger_full_access, its value read as the CLI's parse_overrides reads it. As the CLI resolves them, the last override of a key counts, default_permissions outranks sandbox_mode, and a --sandbox flag outranks both, so an override beside --sandbox meets none. In codex-args such an override meets none, because the action appends its own --sandbox or default_permissions selection after codex-args. The support page and the STABILITY Direction bullet list the spellings. Withholding a quoted URL inside a JSON-array codex-args element no longer drops the element's escaped closing quote, so the published value stays the JSON array the support page describes. --- STABILITY.md | 31 +++--- docs/agent-contract-current.md | 12 ++- docs/design-partner-pilot-results.md | 39 ++++---- docs/host-boundary-support.md | 20 +++- llms-full.txt | 12 ++- src/agents_shipgate/core/host_grants.py | 97 ++++++++++++++++-- tests/test_workflow_agent_launches.py | 126 ++++++++++++++++++++++++ 7 files changed, 286 insertions(+), 51 deletions(-) diff --git a/STABILITY.md b/STABILITY.md index 4d879309f..f1432563d 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -18,9 +18,10 @@ workspace too. `minimum_control_contract_version` stays `21`. See [the migration note](#unread-changed-inputs-821). -Unreleased, runtime contract v41 reads how a coding agent is launched inside a -workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so -host-grants inventory, baseline and drift schemas move to `0.7`: a workflow +Also in unreleased runtime contract v41, extended in place: the host inventory +reads how a coding agent is launched inside a workflow job (#823). +Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and +drift schemas move to `0.7`: a workflow grant adds `agent_launches[]` — a documented agent action's permission inputs, or the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, compared as text and never executed — and @@ -44,9 +45,9 @@ complete; a setting holding credential-shaped text, prose included, is published redacted, compared as published and named the same way, while a checkout ref holding it refuses as a redacted step reference does. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one -without a workflow stays comparable. Verifier `0.20`, capability diff `0.3` and -`minimum_control_contract_version` `21` are unchanged. See -[the migration note](#workflow-agent-launches-contract-v41-823). +without a workflow stays comparable. It moves neither #821's verifier `0.21` +nor its capability diff `0.4`, and `minimum_control_contract_version` stays +`21`. See [the migration note](#workflow-agent-launches-contract-v41-823). Also unreleased, and moving no version of its own: a Claude Code setting that disables prompts or approves project MCP servers carries one rating on every @@ -334,7 +335,7 @@ is added. ## Migration Note: Unreleased — workflow agent launches (host-grants `0.7`, contract v41, #823) -Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` and runtime contract `41` rather than extending them in place. The `0.6` schema files stay published and unchanged. A workflow grant adds two members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: +Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` rather than extending it in place, and extends in place the unreleased runtime contract `41` that #821 minted ([its note](#unread-changed-inputs-821)). The `0.6` schema files stay published and unchanged. A workflow grant adds two members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: ```json { @@ -358,7 +359,7 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i - **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. - **What is withheld.** A structured value publishes its shape and none of its free text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex `--config` table or array publish, as canonical JSON, their key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. A codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. Other argument text — a prompt, a flag's value, a codex `--config` override's scalar value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to such a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access`, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. - **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. @@ -367,7 +368,7 @@ Contract v40 and host-grants `0.6` shipped in 1.1.0, so this mints host-grants i **Compatibility.** - **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. - **A `0.4`–`0.6` baseline with no workflow grant** stays comparable for drift. `audit --host --save-baseline` refuses to overwrite any baseline older than `0.7`, with or without a workflow grant, and exits `2` with `unsupported_baseline_schema`; move it aside and re-save. -- **Git-backed `diff`, `check` and manifest-free `verify`** read both refs with the current reader and need no migration. Their rows keep their shape, and verifier `0.20` and capability diff `0.3` do not move. What changes is values: a workflow row can now be `widened` for an agent launch, its `before`/`after` cells list changed launches and checkout refs, and the `why` of every workflow row whose workflow runs an agent gains the note, so a consumer that matches `why` text exactly sees new text. No check id is added or removed, and `check` decides as before. +- **Git-backed `diff`, `check` and manifest-free `verify`** read both refs with the current reader and need no migration. Their rows keep their shape, and this change moves neither verifier `0.21` nor capability diff `0.4`, which are #821's. What changes is values: a workflow row can now be `widened` for an agent launch, its `before`/`after` cells list changed launches and checkout refs, and the `why` of every workflow row whose workflow runs an agent gains the note, so a consumer that matches `why` text exactly sees new text. No check id is added or removed, and `check` decides as before. - **Validators pinned to the `0.6` schemas** reject a `0.7` inventory, baseline or drift payload. - **`minimum_control_contract_version`** stays `21`. @@ -419,7 +420,7 @@ command, verdict, reader, row or control state is added. **One route moves, on `verify` and `verify --preview` alike.** `verify` without a `shipgate.yaml` returned to the setup route (`Shipgate config not found`, exit `2`) whenever neither side of the comparison held a host artifact, and that route says nothing about the change. A comparison that read no artifact but names a changed input this entry does not read, or counts one or more changed candidate inputs as not examined (`unread_candidates_not_examined` above `0`, the one place that change is mentioned), is now published instead, on the existing manifest-free host route: advisory, exit `0`, `control.state` `agent_action_required` with the `audit --host` next action that route already names. `verify --preview` runs the same comparison and moves the same way: where its next action was `initialize` (`init --write`) with `host_comparison: null`, it is now `discover` (`audit --host`) with the comparison published and the host route's headline; `control.state` stays `agent_action_required` and the exit stays `0`. That includes an agent-related workspace, such as one whose change also adds a tool: a published host comparison takes the preview route whenever one exists, exactly as it already did when the change edits a host file this entry reads, such as the root `.claude/settings.json`. A comparison that reads no artifact, names nothing and counts nothing as not examined still takes the setup route on `verify` and `initialize` on `verify --preview`, as before; so does one whose changed files could not be listed (`unread_candidates: not_examined`), which says nothing about whether a candidate changed. -**What does not change.** `comparison_status`, `incomparable_reasons`, `rows` and every row value, `review`, `unchanged_limits`, every other coverage item, the inventory digests, saved host-grants baselines and drift payloads (host-grants stays `0.6`), `audit --host`, `check`'s decision, rows and text, the control envelope's `capability_rows`, and every control state, permission and next action on a comparison that reads a host artifact. The host-config and cold-start benchmark replays reproduce their run-of-record scores. `minimum_control_contract_version` stays `21`. +**What does not change.** `comparison_status`, `incomparable_reasons`, `rows` and every row value, `review`, `unchanged_limits`, every other coverage item, the inventory digests, saved host-grants baselines and drift payloads (this change moves no host-grants schema; the unreleased host-grants `0.7` is #823's, [its note](#workflow-agent-launches-contract-v41-823)), `audit --host`, `check`'s decision, rows and text, the control envelope's `capability_rows`, and every control state, permission and next action on a comparison that reads a host artifact. The host-config and cold-start benchmark replays reproduce their run-of-record scores. `minimum_control_contract_version` stays `21`. **Compatibility.** `coverage` and its items are closed objects, so a reader validating against the published [`docs/verifier-schema.v0.20.json`](docs/verifier-schema.v0.20.json) rejects a `0.21` artifact's new members; that schema stays frozen. The current reader reads a `0.20` artifact as `0.21` with `unread_candidates: null`, which is what that build knew, and refuses one that claims a `changed_not_read` item, a `candidate`, `read_sources_only: false` or either `unread_candidates` member. A `diff --json` consumer sees `capability_diff_schema_version: "0.4"`. A consumer switching on `coverage.items[].status` should treat an unknown status as a change it must read, not as no change. @@ -428,10 +429,12 @@ command, verdict, reader, row or control state is added. ## Migration Note: Unreleased — one rating per Claude Code setting (#827) This change moves no version of its own: no schema, member, check id or -`minimum_control_contract_version` moves, and host-grants stays `0.6`, as -shipped in 1.1.0. The capability diff `0.4`, verifier `0.21` and runtime -contract `41` of the unreleased tree are #821's -([migration note](#unread-changed-inputs-821)), not this change's. What moves +`minimum_control_contract_version` moves. Of the unreleased tree's versions, +host-grants `0.7` is #823's +([migration note](#workflow-agent-launches-contract-v41-823)), capability +diff `0.4` and verifier `0.21` are #821's +([migration note](#unread-changed-inputs-821)), and runtime contract `41` +carries both; none is this change's. What moves is the value of existing fields for the Claude Code settings the host inventory publishes as `permission_mode` grants. One table, `core/host_settings.py`, now rates each value, and the grant's `access` and `risk`, a row's `severity`, diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index e80b77e79..857f54c50 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -44,8 +44,8 @@ directory, still refuses its comparison. A `0.20` verifier claiming a partial comparison or a `scope` is refused. See [the migration note](../STABILITY.md#partial-host-comparison-808). -Runtime contract v41, unreleased, reads how a coding agent is launched inside a -workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so +Runtime contract v41, extended in place, also reads how a coding agent is +launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. An agent launch is a step whose `uses:` is a documented agent action @@ -66,7 +66,9 @@ server's arguments and a hook's command are compared but never published; a URL publishes its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` -sandbox (`permission-profile: :danger-full-access` included), +sandbox (`permission-profile: :danger-full-access` included, and in a +`codex exec` step without `--sandbox` a `--config` override of `sandbox_mode` +or `default_permissions` that selects it), `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`. A rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left, or where the job's @@ -79,8 +81,8 @@ redacted, compared as published and a named non-blocking limit, and a checkout ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays -comparable. Verifier `0.20`, capability diff `0.3` and -`minimum_control_contract_version` `21` are unchanged. See +comparable. It moves neither #821's verifier `0.21` nor its capability diff +`0.4`, and `minimum_control_contract_version` stays `21`. See [the migration note](../STABILITY.md#workflow-agent-launches-contract-v41-823). Previous runtime contract v40 reads the action reference each workflow step declares diff --git a/docs/design-partner-pilot-results.md b/docs/design-partner-pilot-results.md index d91ec036e..faaa52733 100644 --- a/docs/design-partner-pilot-results.md +++ b/docs/design-partner-pilot-results.md @@ -93,22 +93,27 @@ the `disposition` field, and it publishes no `review` or `coverage` block; `1.1.0`'s `review.summary` counts 6 changes from 6 rows, 4 widening. On this fixture `1.1.0` changes what a run says about the rows, not which rows it finds. -#821 then moved this tree's runtime contract to 41, and the source-tree column -was rerun on 2026-09-22 through `./shipgate` on the fixture rebuilt from the -description below, beside the `v1.1.0` release commit (`e3c6cb0c`, runtime -contract 40) run the same way. The two returned identical cells except the -runtime contract, 40 against 41: host-grant inventory schema 0.6, `check` -blocking with four violations and visible coverage, the host-only `init` -handoff with no manifest or workflow written, manifest-free `verify` exiting 0 -with the same six advisory rows, drift naming all four expansion signals, and -`diff` against the fixture base exiting 0 with `comparison_status: comparable` -and the same six rows, four of them widening. The `diff` text is identical -apart from the fixture's commit ids. `diff --json` and `verifier.json` differ -only in their schema versions (capability diff 0.3 against 0.4, verifier 0.20 -against 0.21) and in the members #821 adds to the coverage block: each item's -`candidate` is `null`, and `unread_candidates` is `examined` with none left -unexamined. This fixture changes only the two files the entry reads, so #821 -names nothing on it and `read_sources_only` stays `true`. +#821 then moved this tree's runtime contract to 41, and #823, extending it in +place, the host-grant inventory schema to 0.7. With both, the source-tree +column was rerun on 2026-09-23 through this tree's engine on the fixture +rebuilt from the description below, beside the `v1.1.0` release commit +(`e3c6cb0c`, runtime contract 40) run the same way. The two returned identical +cells except the runtime contract, 40 against 41, and the host-grant inventory +schema, 0.6 against 0.7: `check` blocking with four violations and visible +coverage, its `agent-boundary-json` identical apart from the workspace path, +the host-only `init` handoff with no manifest or workflow written, +manifest-free `verify` exiting 0 with the same six advisory rows, drift naming +all four expansion signals, and `diff` against the fixture base exiting 0 with +`comparison_status: comparable` and the same six rows, four of them widening. +The `diff` text is identical apart from the fixture's commit ids, and the host +inventory apart from its schema version. `diff --json` and `verifier.json` +differ only in their schema versions (capability diff 0.3 against 0.4, +verifier 0.20 against 0.21) and in the members #821 adds to the coverage +block: each item's `candidate` is `null`, and `unread_candidates` is +`examined` with none left unexamined. This fixture changes only the two files +the entry reads and has no workflow, so #821 names nothing on it and +`read_sources_only` stays `true`, and #823, which reads agent launches and +checkout refs in workflows, adds nothing to its grants. An older release, `v0.15.0`, measured on 2026-09-05, did not. It reported runtime contract 10 and inventory schema 0.1; `check` returned `warn` / `none` @@ -128,7 +133,7 @@ expansion signals. It shipped as a qualified release. | `diff` against the fixture base (Git-backed Route H) | exit 0, `comparable`, 6 rows, 4 widening | not measured | exit 0, `comparable`, the same 6 rows | | Qualification | **none** — advisory channel, no qualification claim | **none** — no adjudicated corpus, nothing signed | not a distributed build | -Every 2026-09-22 rerun reproduced four boundary violations (`block` / +Every rerun on 2026-09-22 and 2026-09-23 reproduced four boundary violations (`block` / `critical`) and visible coverage. `init --write --ci` writes no manifest or workflow and routes the host-only fixture to the read-only audit route. Manifest-free `verify` now succeeds as an advisory comparison: six rows name three added permissions, two removed narrower rules, diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 9df589871..325e2848d 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -139,7 +139,7 @@ things are listed on the workflow grant, each naming its `job/step` (the step's |---|---|---| | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions`, `--permission-mode bypassPermissions` or a JSON `--settings` value whose `defaultMode` is `bypassPermissions`; `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` and `settings` as above | - | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access`; `allow-users` gains a `*` entry | + | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access` (`-s`, attached or not); `allow-users` gains a `*` entry. A sandbox `--config` override in `codex-args` meets none: after `codex-args` the action appends its own `--sandbox`, or its own `default_permissions` override for a `permission-profile`, which takes precedence | No shell reads `claude_args` or `codex-args`: each action splits its own input, and a rule is met only by the words the action passes on. The @@ -171,9 +171,21 @@ things are listed on the workflow grant, each naming its `job/step` (the step's For `codex exec`: `--sandbox`/`-s`, `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, `--approve-for-me`/`--not-so-yolo`, `--dangerously-bypass-hook-trust`, - `--add-dir`, `--config`/`-c` and `--profile`/`-p`; gaining - `--dangerously-bypass-approvals-and-sandbox` or `--sandbox danger-full-access` - widens. A flag the CLI reads as variadic (`--allowedTools`, `--add-dir`, …) + `--add-dir`, `--config`/`-c` and `--profile`/`-p`, a short flag's value + read attached as clap reads it (`-sdanger-full-access`, `-s=…`, + `-c`, `-c=`); gaining + `--dangerously-bypass-approvals-and-sandbox` or the full-access sandbox + widens. The full-access sandbox is `--sandbox danger-full-access` or, when + the command passes no `--sandbox`, which takes precedence, a `--config` + override in any of its four spellings that sets `sandbox_mode` to + `danger-full-access` (the setting `--sandbox` sets) or `default_permissions` + to `:danger-full-access` (the built-in full-access profile, which the + action's `permission-profile` input passes the CLI the same way), its value + read as TOML and otherwise as text with its quotes trimmed, as the CLI + reads it. The last override of a key counts, and a `default_permissions` + override outranks a `sandbox_mode` one. A key under another table, such as + `profiles..sandbox_mode`, and a `--profile`, which names a + configuration this audit does not read, meet none. A flag the CLI reads as variadic (`--allowedTools`, `--add-dir`, …) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after it is compared as one of its values; the row shows it. A `run:` holding more than one command (a newline, `&&`, `;`, diff --git a/llms-full.txt b/llms-full.txt index bf6692d19..c3cbd29cd 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1618,8 +1618,8 @@ directory, still refuses its comparison. A `0.20` verifier claiming a partial comparison or a `scope` is refused. See [the migration note](../STABILITY.md#partial-host-comparison-808). -Runtime contract v41, unreleased, reads how a coding agent is launched inside a -workflow job (#823). Contract v40 and host-grants `0.6` shipped in 1.1.0, so +Runtime contract v41, extended in place, also reads how a coding agent is +launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. An agent launch is a step whose `uses:` is a documented agent action @@ -1640,7 +1640,9 @@ server's arguments and a hook's command are compared but never published; a URL publishes its scheme and host. Only a documented rule a job's launches gain — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` -sandbox (`permission-profile: :danger-full-access` included), +sandbox (`permission-profile: :danger-full-access` included, and in a +`codex exec` step without `--sandbox` a `--config` override of `sandbox_mode` +or `default_permissions` that selects it), `safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the row `widened`. A rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left, or where the job's @@ -1653,8 +1655,8 @@ redacted, compared as published and a named non-blocking limit, and a checkout ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays -comparable. Verifier `0.20`, capability diff `0.3` and -`minimum_control_contract_version` `21` are unchanged. See +comparable. It moves neither #821's verifier `0.21` nor its capability diff +`0.4`, and `minimum_control_contract_version` stays `21`. See [the migration note](../STABILITY.md#workflow-agent-launches-contract-v41-823). Previous runtime contract v40 reads the action reference each workflow step declares diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index b4b1c617f..89a284285 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -1891,14 +1891,22 @@ def _read_flags( A flag is read by name wherever it is a word of its own; every other word — the prompt, ``--model`` and any undocumented flag — is not compared. A - flag that takes no value, or is given none, has ``None``. + flag that takes no value, or is given none, has ``None``. A long flag's + value may be attached as ``--name=value``, and a short one's as + ``-s`` or ``-s=``, which clap reads as ``-s `` + (#823 review cycle 3). """ flags: list[tuple[str, int | None, list[str] | None]] = [] index = 0 while index < len(words): word = words[index] - name, equals, attached = word.partition("=") if word.startswith("--") else (word, "", "") + if word.startswith("--"): + name, equals, attached = word.partition("=") + elif len(word) > 2 and table.get(word[:2], ("", 0))[1] == 1: + name, equals, attached = word[:2], "=", word[2:].removeprefix("=") + else: + name, equals, attached = word, "", "" spec = table.get(name) index += 1 if spec is None: @@ -2451,8 +2459,16 @@ def _published_value(text: str) -> tuple[str, bool]: if len(expressions) > len(_EXPRESSION_SLOTS) or _EXPRESSION_SLOT_RE.search(text): # No free character to stand for each expression: read the text as written. expressions = [] + def url_withheld(match: re.Match[str]) -> str: + # A backslash the URL ends at is kept, so the `\"` a JSON-array + # `codex-args` element escapes a quote with stays an escape + # (#823 review cycle 3). + url = match.group(0) + bare = url.rstrip("\\") + return _url_withheld(bare) + url[len(bare):] + def urls_withheld(value: str) -> str: - return _URL_RE.sub(lambda match: _url_withheld(match.group(0)), value) + return _URL_RE.sub(url_withheld, value) slots = iter(_EXPRESSION_SLOTS) marked = _EXPRESSION_SPAN_RE.sub(lambda _match: chr(next(slots)), text) if expressions else text @@ -2780,12 +2796,77 @@ def _claude_action_rules(words: tuple[str, ...]) -> set[str]: return rules +def _codex_override(text: str) -> tuple[str, Any] | None: + """A codex ``--config`` override's key and value, as the CLI's ``parse_overrides`` reads them. + + The key is the text before the first ``=``, trimmed; the value is parsed + as TOML, and text that does not parse is a string with its surrounding + quotes trimmed, so ``sandbox_mode=danger-full-access`` and + ``sandbox_mode="danger-full-access"`` set one value. + """ + + key, equals, value = text.partition("=") + if not equals or not key.strip(): + return None + raw = value.strip() + try: + loaded = tomllib.loads(f"value = {raw}").get("value") + except (tomllib.TOMLDecodeError, RecursionError): + loaded = raw.strip("\"'") + return key.strip(), loaded + + +def _codex_config_full_access(flags: list[tuple[str, int | None, list[str] | None]]) -> bool: + """Whether ``codex exec``'s ``--config`` overrides select its full-access sandbox (#823 review cycle 3). + + ``sandbox_mode = "danger-full-access"`` is the setting ``--sandbox`` sets, + and ``default_permissions = ":danger-full-access"`` selects the built-in + full-access profile. As the CLI resolves them, the last override of a key + counts, a ``default_permissions`` override selects the permission + profile over a ``sandbox_mode`` one, and a ``--sandbox`` flag takes + precedence over both, so this reads none while one is passed. + """ + + if any(name == "--sandbox" for name, _arity, _values in flags): + return False + profile: Any = None + mode: Any = None + for name, _arity, values in flags: + if name != "--config": + continue + for value in values or []: + override = _codex_override(value) + if override is None: + continue + key, loaded = override + if key == "default_permissions": + profile = loaded + elif key == "sandbox_mode": + mode = loaded + if profile is not None: + return profile == ":danger-full-access" + return mode == "danger-full-access" + + def _flag_rules( - family: str, flags: list[tuple[str, int | None, list[str] | None]] + family: str, + flags: list[tuple[str, int | None, list[str] | None]], + *, + config_selects_sandbox: bool = True, ) -> set[tuple[str, str]]: - """The widening rules a CLI's documented flags meet, each with the flag that met it.""" + """The widening rules a CLI's documented flags meet, each with the flag that met it. + + ``config_selects_sandbox`` is false for the words `openai/codex-action` + passes on from ``codex-args``: after them the action appends its own + ``--sandbox ``, or for a ``permission-profile`` its own + ``default_permissions`` override (``runCodexExec.ts``), and either takes + precedence over a sandbox ``--config`` override written before it, so + one there selects nothing. + """ rules: set[tuple[str, str]] = set() + if family == "codex" and config_selects_sandbox and _codex_config_full_access(flags): + rules.add(("danger_full_access", "--config")) for name, _arity, values in flags: value = values[0] if values else None if family == "claude" and ( @@ -2835,7 +2916,11 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu if spec.family == "claude": found = _claude_action_rules(words) else: - found = {rule for rule, _flag in _flag_rules("codex", _read_flags(words, _CODEX_FLAGS))} + found = { + rule for rule, _flag in _flag_rules( + "codex", _read_flags(words, _CODEX_FLAGS), config_selects_sandbox=False + ) + } rules.update((rule, name) for rule in found) return rules diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index a28f584eb..83390ab73 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -654,6 +654,118 @@ def test_any_other_edit_is_changed(before, after): assert (row.direction, row.expands) == ("changed", False) +# --- codex exec's full-access sandbox, however the CLI reads it (#823 review cycle 3) --- + +CODEX_WORKSPACE = _workflow({"run": "codex exec -s workspace-write 'review'"}) + + +@pytest.mark.parametrize( + "run", + [ + "codex exec -s danger-full-access 'review'", + "codex exec --sandbox=danger-full-access 'review'", + # clap reads a short option's attached value, with or without `=`. + "codex exec -sdanger-full-access 'review'", + "codex exec -s=danger-full-access 'review'", + # `sandbox_mode` is the setting --sandbox sets, in each way -c is written. + "codex exec -c sandbox_mode=\"danger-full-access\" 'review'", + "codex exec -c 'sandbox_mode=\"danger-full-access\"' 'review'", + "codex exec --config=sandbox_mode=danger-full-access 'review'", + "codex exec -csandbox_mode=danger-full-access 'review'", + "codex exec -c=sandbox_mode=danger-full-access 'review'", + "codex exec -c ' sandbox_mode = \"danger-full-access\" ' 'review'", + # The built-in full-access profile, which is what the action's + # `permission-profile: :danger-full-access` passes the CLI. + "codex exec -c default_permissions=\":danger-full-access\" 'review'", + "codex exec --config 'default_permissions=\":danger-full-access\"' 'review'", + # The last override of a key counts; a profile override outranks a sandbox one. + "codex exec -c sandbox_mode=read-only -c sandbox_mode=danger-full-access 'review'", + "codex exec -c sandbox_mode=read-only -c default_permissions=:danger-full-access 'review'", + ], + ids=["short", "long-attached", "short-attached", "short-equals", "config", "config-quoted", "config-attached", + "c-attached", "c-equals", "config-spaced", "profile", "profile-quoted", "last-override", + "profile-over-sandbox-mode"], +) +def test_codex_exec_full_access_widens_in_every_spelling_the_cli_reads(run): + after = _workflow({"run": run}) + + assert host_grant_expansion_signals(_changes(CODEX_WORKSPACE, after)) == [ + f"workflow_agent_widened_changed: {SOURCE}" + ] + row, = _rows(CODEX_WORKSPACE, after) + assert (row.direction, row.expands) == ("widened", True) + assert "an agent launch now runs without a sandbox (danger-full-access) (review/steps[0])" in row.why + assert "no permission flags" not in row.after + + +@pytest.mark.parametrize( + "run", + [ + # --sandbox takes precedence over a --config override. + "codex exec -s workspace-write -c sandbox_mode=danger-full-access 'review'", + "codex exec -sworkspace-write -c default_permissions=:danger-full-access 'review'", + # The last override counts, and a profile override outranks sandbox_mode. + "codex exec -c sandbox_mode=danger-full-access -c sandbox_mode=read-only 'review'", + "codex exec -c sandbox_mode=danger-full-access -c default_permissions=:workspace 'review'", + # A key under another table, or another value, is not the setting. + "codex exec -c profiles.ci.sandbox_mode=danger-full-access 'review'", + "codex exec -c default_permissions=danger-full-access 'review'", + "codex exec -sread-only 'review'", + ], + ids=["sandbox-flag-wins", "attached-sandbox-flag-wins", "last-override", "profile-over-sandbox-mode", + "profile-scoped-key", "custom-profile-name", "read-only"], +) +def test_a_codex_exec_sandbox_the_cli_does_not_select_is_changed(run): + after = _workflow({"run": run}) + + assert host_grant_expansion_signals(_changes(CODEX_WORKSPACE, after)) == [] + row, = _rows(CODEX_WORKSPACE, after) + assert (row.direction, row.expands) == ("changed", False) + + +def test_attached_short_values_publish_under_the_primary_spelling(): + launch, = _launches(_workflow({"run": "codex exec -sdanger-full-access -c=model=o3 -pci 'review'"})) + + assert launch["settings"] == [ + {"name": "--config", "value": "model=o3", "unresolved_reason": None}, + {"name": "--profile", "value": "ci", "unresolved_reason": None}, + {"name": "--sandbox", "value": "danger-full-access", "unresolved_reason": None}, + ] + assert launch["widening_rules"] == [{"rule": "danger_full_access", "setting": "--sandbox"}] + # One setting, two spellings: respelling it is quiet. + assert _rows( + _workflow({"run": "codex exec -s danger-full-access 'review'"}), + _workflow({"run": "codex exec -sdanger-full-access 'review'"}), + ) == [] + # One rule, two spellings: moving between the flag and the override is not a widening. + before = _workflow({"run": "codex exec -s danger-full-access 'review'"}) + after = _workflow({"run": "codex exec -c sandbox_mode=danger-full-access 'review'"}) + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + + +@pytest.mark.parametrize( + ("codex_args", "direction"), + [ + # The action appends its own --sandbox, or its own default_permissions + # override for a permission-profile, after codex-args, and either + # takes precedence over a sandbox --config override written before it. + ("-c sandbox_mode=danger-full-access", "changed"), + ('--config=default_permissions=":danger-full-access"', "changed"), + # A --sandbox word is read as the flag it is, attached or not, as before. + ("-sdanger-full-access", "widened"), + ], + ids=["sandbox-mode-override", "profile-override", "attached-short-sandbox"], +) +def test_codex_args_sandbox_overrides_are_read_as_the_action_passes_them(codex_args, direction): + before = _workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": "--json"}}) + after = _workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}) + + row, = _rows(before, after) + assert row.direction == direction + + def test_a_rule_gained_where_the_job_launched_the_agent_only_unread_before_is_not_claimed(): before = _workflow({"run": "npm ci && claude -p --dangerously-skip-permissions 'review'"}) after = _workflow({"run": "claude -p --dangerously-skip-permissions 'review'"}) @@ -1189,6 +1301,20 @@ def test_a_url_path_is_withheld_while_the_rest_of_the_setting_and_a_rule_beside_ assert _uncompared_workflow_text(_grant(after)) is None +def test_a_quoted_url_in_a_json_array_codex_args_element_keeps_the_array_valid(): + """#823 review cycle 3: withholding the URL keeps the element's escaped closing quote.""" + + codex_args = json.dumps(["-c", 'mcp_servers.x.url="https://h2.example.com/p?token=canary-q"', "--json"]) + launch, = _launches(_workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}})) + setting, = launch["settings"] + + assert json.loads(setting["value"]) == [ + "-c", 'mcp_servers.x.url="https://h2.example.com/"', "--json", + ] + assert setting["unresolved_reason"] is None + assert "canary-q" not in json.dumps(launch) + + def test_a_marketplace_url_compares_by_scheme_and_host_as_an_mcp_server_url_does(): def marketplace(url): return _workflow(_agent(plugin_marketplaces=url)) From 81d69303d3d595773fbdde6b8499d4fbdf2f4eb4 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 04:26:51 -0700 Subject: [PATCH 07/14] Address review cycle 2 on agent launches in CI (#823) An agent CLI inside a double-quoted $(...) or a backtick substitution, or after a shell reserved word, got no launch, no row and no coverage issue: gh pr comment --body "$(claude -p --dangerously-skip-permissions ...)" read as "No static host-grant changes detected", while the support page said such a run: is listed as unresolved once for each agent CLI it starts at the head of a command. The word splitter reads a double-quoted substitution as one word and a backtick as no operator, and the head reader took then, do, { and ! for the command. _command_substitutions now lists the text of each $(...) and backtick substitution outside single quotes, double-quoted ones included, and an agent CLI heading a command inside one is an unresolved launch (shell_expansion for a single top-level command, compound_command otherwise), publishing none of its text. An escaped \$, a single-quoted '$(...)', a comment and $((...)) arithmetic are not substitutions, so echo "claude -p ..." stays unlisted. The scanner keeps an explicit stack and reads each nested substitution as `_` in the one around it, so no nesting depth recurses or re-splits text. When the run's quoting does not balance, each line's substitutions are read as well. A command's head is now read after shell reserved words (!, time [-p], {, if, then, elif, else, while, until, do, function NAME), and a command after one is not a simple command, so its launch is compound_command and never read. The compound_command and shell_expansion limit phrases, the launch schema description (host-grants 0.7 schemas regenerated), the support page and the STABILITY bullet say so. A read launch that became a form this audit does not recognise (npx, a path, codex options before exec) was worded "a step no longer launches an agent", though the step still starts one. The removed case now reads "a step no longer declares an agent launch this audit reads", and a row whose workflow still exists adds that the step may still start an agent in a way this audit does not read, naming those forms. codex with an option before exec is listed under Known unread surfaces and in STABILITY's "What is not read". Tests cover each shape the review named, the negative controls, the diff and audit --host route of the reproduction, the reworded removal for npx, a path and codex root options, and 20000 nested substitutions. --- CHANGELOG.md | 2 +- STABILITY.md | 4 +- docs/host-boundary-support.md | 30 ++- docs/host-grants-baseline-schema.v0.7.json | 2 +- docs/host-grants-inventory-schema.v0.7.json | 2 +- .../core/capability_diff_rows.py | 11 +- src/agents_shipgate/core/host_grants.py | 195 ++++++++++++++++-- src/agents_shipgate/schemas/host_grants.py | 7 +- tests/test_workflow_agent_launches.py | 138 ++++++++++++- 9 files changed, 352 insertions(+), 39 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c6261132c..d1024a1e2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained. A compound command, an expansion or an expression is a named non-blocking limit and publishes none of its text. A JSON object in a setting, however it is attached to its flag, publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained. A compound command, an expansion or an expression — an agent CLI inside a quoted `$(…)` or a backtick substitution, or after a reserved word such as `then`, included — is a named non-blocking limit and publishes none of its text; a launch that becomes one this audit does not recognise (`npx`, a path, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a setting, however it is attached to its flag, publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index f1432563d..d018c1c43 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -361,9 +361,9 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin - **What is withheld.** A structured value publishes its shape and none of its free text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex `--config` table or array publish, as canonical JSON, their key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. A codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. Other argument text — a prompt, a flag's value, a codex `--config` override's scalar value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to such a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). - **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. -- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command, a shell expansion or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. +- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command or a shell reserved word such as `then` or `!`, a shell expansion — an agent CLI heading a command inside a `$(…)` or backtick substitution, double-quoted or not, included — or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. -- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, a path), and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. +- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, `bash -c`, a path), `codex` with an option before `exec`, and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. **Compatibility.** - **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 325e2848d..21eb8d5e3 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -68,9 +68,14 @@ the changed inputs the candidate rules at the end of this section name (#821): (#823): an action outside its table, even one that takes `claude_args`; a composite action (#701); a script the step runs (`run: ./scripts/review.sh`); an agent CLI reached through another command (`npx @anthropic-ai/claude-code`, - `timeout 600 claude`, `sudo`, a path such as `./node_modules/.bin/claude`); - and a step's `env:`, `shell:` and `if:`. Editing one gives no row and names - no limit. + `timeout 600 claude`, `sudo`, `bash -c`, a path such as + `./node_modules/.bin/claude`); `codex` with an option before `exec` + (`codex -c sandbox_mode=danger-full-access exec`, `codex --yolo exec`), which + can change how `exec` runs; and a step's `env:`, `shell:` and `if:`. + Editing one gives no row and names no limit. A launch this audit read that + becomes one of these is a row saying the step no longer declares an agent + launch this audit reads, and that it may still start one this way; it never + says the step no longer starts an agent. Review changes to those files and fields as you would a change to the workflow, hook or server entry that holds them. @@ -189,12 +194,19 @@ things are listed on the workflow grant, each naming its `job/step` (the step's takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after it is compared as one of its values; the row shows it. A `run:` holding more than one command (a newline, `&&`, `;`, - `|`, a redirection or a here-doc) or quoting that does not balance, a shell - expansion (`$VAR`, `$(…)`, a backtick) or a `${{ }}` expression is listed as - `unresolved` with that reason, once for each agent CLI it starts at the head - of a command (of a line, when its quoting does not balance), and none of its - text is published. A command that launches no headless agent — `claude mcp add`, - `codex login`, an `echo` that mentions either — is not listed. + `|`, a redirection or a here-doc), a shell reserved word before a command + (`if … then`, `for … do`, `{ …; }`, `!`, `time`) or quoting that does not + balance, a shell expansion (`$VAR`, `$(…)`, a backtick) or a `${{ }}` + expression is listed as `unresolved` with that reason, once for each agent + CLI it starts at the head of a command (of a line, when its quoting does not + balance), and none of its text is published. A command's head is read after + any reserved words, and inside each `$(…)` or backtick substitution outside + single quotes, so `gh pr comment --body "$(claude -p …)"` and + ``REVIEW=`claude -p …` `` are listed (`shell_expansion`), and so is + `if …; then claude -p …; fi` (`compound_command`). A command that launches no + headless agent — `claude mcp add`, `codex login`, an `echo` that mentions + either, a quoted `"claude -p …"` or a single-quoted `'$(claude -p …)'` — is + not listed. - **Each `actions/checkout` step's `with.ref`**, or the default when it declares none or an empty one. Adding `ref: ${{ github.event.pull_request.head.sha }}` is a `changed` row naming the diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 7eade822f..16ac7e7c6 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1284,7 +1284,7 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command, a shell reserved word or\nquoting that does not balance, a shell expansion (an agent CLI inside a\ncommand substitution included), a ``${{ }}`` expression, or ``with:``\nthat is not a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index 6eac7688a..3efd9fef7 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1342,7 +1342,7 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command or quoting that does not\nbalance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is\nnot a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command, a shell reserved word or\nquoting that does not balance, a shell expansion (an agent CLI inside a\ncommand substitution included), a ``${{ }}`` expression, or ``with:``\nthat is not a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index bb02719c3..09ef1917a 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -383,7 +383,9 @@ def where(item: dict[str, Any]) -> str: for verb, wording in ( ("changed", "an agent launch's declared settings changed"), ("added", "a step now launches an agent"), - ("removed", "a step no longer launches an agent"), + # What was established is that the step declares no launch this + # audit reads, not that it starts no agent (#823 review). + ("removed", "a step no longer declares an agent launch this audit reads"), ) if verb in groups ] @@ -392,6 +394,13 @@ def where(item: dict[str, Any]) -> str: f"{_joined_words(phrases)}; agent launch settings are compared as declared text, " "and a change that gains no documented widening rule is not counted as a widening" ) + if "removed" in groups and after is not None: + reasons.append( + "a step that no longer declares one may still start an agent in a way this audit " + "does not read, such as an action outside its table, `npx`, a script, a path such as " + "`./node_modules/.bin/claude` or `codex` options before `exec`, so this row does not " + "say that it no longer starts one" + ) # A rule is read only from literal text a `${{ }}` expression cannot # reach, so the row says where that leaves text unread (#823 review). expressions = list(dict.fromkeys( diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 89a284285..6e0c26f63 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -1852,16 +1852,165 @@ def _has_shell_comment(text: str) -> bool: return False +@dataclass +class _SubstitutionLevel: + """One open level of :func:`_command_substitutions`: the top of the text, or one substitution.""" + + #: The character that ends it: `)` for `$(`, a backtick for a backtick, none at the top. + closer: str + #: Where its text starts. + start: int + #: ``$((…))`` is arithmetic, not a command. + arithmetic: bool = False + quoted: bool = False + #: Parentheses opened inside a `$(…)` and not yet closed. + depth: int = 0 + #: Its text so far, each nested substitution in it replaced by `_`. + parts: list[str] = field(default_factory=list) + #: Where its text not yet in ``parts`` starts. + cursor: int = 0 + + +def _command_substitutions(text: str) -> list[str]: + """The text of each ``$(…)`` and backtick command substitution the shell would run in ``text``. + + Found outside single quotes, inside double quotes too: the word splitter + reads a double-quoted ``"$(claude -p …)"`` as one word and a backtick as + no operator, so the command inside either never heads a command it splits + (#823 review). A nested substitution is listed on its own and reads as + ``_`` in the text around it, so each character is listed once and no + nesting depth costs more than the text's length. An escaped ``\\$`` or + backtick, a comment and the ``$((…))`` of arithmetic are not + substitutions, though one inside arithmetic still is, and one left open + runs to the end of the text. Only used to name an agent CLI as + unresolved: nothing found here is read or published. + """ + + bodies: list[str] = [] + levels = [_SubstitutionLevel(closer="", start=0)] + + def open_level(opener: int, closer: str, start: int, arithmetic: bool = False) -> None: + parent = levels[-1] + parent.parts += [text[parent.cursor:opener], "_"] + levels.append(_SubstitutionLevel(closer=closer, start=start, arithmetic=arithmetic, cursor=start)) + + def close_level(end: int) -> None: + level = levels.pop() + if not level.arithmetic: + bodies.append("".join([*level.parts, text[level.cursor:end]])) + levels[-1].cursor = min(end + 1, len(text)) + + index = 0 + while index < len(text): + level = levels[-1] + char = text[index] + if char == "\\": + index += 2 + continue + if char == "`": + if level.closer == "`": + close_level(index) + else: + open_level(index, "`", index + 1) + index += 1 + continue + if char == "$" and text.startswith("(", index + 1): + open_level(index, ")", index + 2, arithmetic=text.startswith("((", index + 1)) + index += 2 + continue + if level.quoted: + level.quoted = char != '"' + elif char == '"': + level.quoted = True + elif char == "'": + quote_end = text.find("'", index + 1) + index = len(text) if quote_end < 0 else quote_end + 1 + continue + elif char == "#" and ( + index == level.start or text[index - 1].isspace() or text[index - 1] in _SHELL_OPERATOR_CHARS + ): + # A comment runs to the end of its line, or inside backticks to + # the backtick that closes them. + ends = [text.find("\n", index), text.find("`", index) if level.closer == "`" else -1] + index = min((end for end in ends if end >= 0), default=len(text)) + continue + elif level.closer == ")" and char == "(": + level.depth += 1 + elif level.closer == ")" and char == ")": + if level.depth: + level.depth -= 1 + else: + close_level(index) + index += 1 + while len(levels) > 1: + close_level(len(text)) + return bodies + + +def _commands(words: list[str]) -> list[list[str]]: + """``words`` grouped into the commands their operator tokens separate, empty ones included.""" + + commands: list[list[str]] = [[]] + for word in words: + if _is_operator(word): + commands.append([]) + else: + commands[-1].append(word) + return commands + + +def _substituted_agents(run: str) -> set[str]: + """The agent CLIs ``run`` starts at the head of a command inside a command substitution.""" + + agents: set[str] = set() + for body in _command_substitutions(run): + words = _shell_words(body) + if words is None: + agents.update(agent for line in body.splitlines() if (agent := _line_agent(line))) + continue + agents.update(launch[0] for command in _commands(words) if (launch := _agent_command(command))) + return agents + + +#: Shell reserved words that can come before a command's name: a pipeline's +#: ``!`` and ``time`` (with its ``-p``), and the words that open a list inside +#: a compound command, as in ``if …; then claude -p …; fi`` (#823 review). A +#: command after one is not a simple command, so an agent CLI there is named +#: as unresolved and never read. +_RESERVED_PREFIXES = frozenset({"!", "time", "{", "if", "then", "elif", "else", "while", "until", "do"}) + + +def _reserved_prefix_length(words: list[str]) -> int: + """How many leading words of one command are shell reserved words, a ``function NAME`` included. + + Reserved words are read only before a command's assignments and name, so + ``FOO=1 then`` runs a command named ``then``. + """ + + index = 0 + while index < len(words): + if words[index] in _RESERVED_PREFIXES: + index += 2 if words[index] == "time" and words[index + 1:index + 2] == ["-p"] else 1 + elif words[index] == "function": + # `function NAME { …; }`: the name is not a command. + index += 2 + else: + break + return min(index, len(words)) + + def _agent_command(words: list[str]) -> tuple[str, list[str]] | None: - """The agent CLI one simple command launches headless, and its arguments. + """The agent CLI one command launches headless, and its arguments. ``claude`` with ``-p``/``--print`` anywhere in its arguments, or ``codex`` with ``exec`` (or its alias ``e``) as its first argument, after any leading - ``NAME=value`` assignments. Any other command — ``claude mcp add``, + shell reserved words (``then``, ``do``, ``{``, ``!``, ``time``, a + ``function NAME``) and then any ``NAME=value`` assignments, the order the + shell reads them in. Any other command — ``claude mcp add``, ``codex login``, an ``echo`` that mentions either — launches none. """ - index = 0 + index = _reserved_prefix_length(words) while index < len(words) and _ASSIGNMENT_RE.match(words[index]): index += 1 if index >= len(words): @@ -2624,8 +2773,9 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: the agent CLI (after literal ``NAME=value`` assignments), holds no shell expansion and no ``${{ }}`` expression: then its documented permission flags are its settings. Otherwise an agent CLI found at the start of any - simple command in it is one ``unresolved`` entry with the reason, and none - of its text is published. + command in it — after any reserved words, and inside a ``$(…)`` or + backtick substitution outside single quotes — is one ``unresolved`` entry + with the reason, and none of its text is published. """ run = run.strip() @@ -2633,41 +2783,45 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: base = {"job": job, "step": step_label} if words is None: # Quoting that does not balance cannot be split into commands, so no - # word of it is read; a line that starts an agent CLI is still named. + # word of it is read; a line that starts an agent CLI, or holds a + # substitution that does, is still named. agents = sorted({ agent for line in run.splitlines() - if (agent := _line_agent(line)) is not None + for agent in ({_line_agent(line)} | _substituted_agents(line)) + if agent is not None }) return [ {**base, "agent": agent, "form": "unresolved", "unresolved_reason": "compound_command", "settings": []} for agent in agents ] - commands: list[list[str]] = [[]] - for word in words: - if _is_operator(word): - commands.append([]) - else: - commands[-1].append(word) + commands = _commands(words) launches = [launch for command in commands if (launch := _agent_command(command))] - if not launches: + # An agent CLI inside a `$(…)` or backtick substitution heads no command + # the splitter sees when the substitution is double-quoted or a backtick + # (#823 review), and is named as unresolved. + substituted = _substituted_agents(run) + if not launches and not substituted: return [] # A comment is an unquoted `#` at the start of a word, read off the raw # text: the splitter keeps `#` literal, so a quoted "#123 review" is one word. + # A command after a reserved word (`then`, `!`, `time`) is not a simple one. + simple = [command for command in commands if command] single = ( - len([command for command in commands if command]) == 1 + len(simple) == 1 and not any(_is_operator(word) for word in words) and not _has_shell_comment(run) + and not _reserved_prefix_length(simple[0]) ) if "${{" in run: reason: str | None = "expression" elif not single: reason = "compound_command" - elif _has_shell_expansion(run): + elif substituted or _has_shell_expansion(run): reason = "shell_expansion" else: reason = None if reason is not None: - agents = sorted({agent for agent, _arguments in launches}) + agents = sorted({agent for agent, _arguments in launches} | substituted) return [ {**base, "agent": agent, "form": "unresolved", "unresolved_reason": reason, "settings": []} for agent in agents @@ -3096,8 +3250,11 @@ def gained_agent_widenings(before: dict[str, Any] | None, after: dict[str, Any] #: How an unresolved agent launch or checkout reads in the limit that names it. _UNRESOLVED_AGENT_PHRASES = { - "compound_command": "part of a `run:` that holds more than one command, or quoting this audit cannot split", - "shell_expansion": "a command with a shell expansion this static audit does not evaluate", + "compound_command": ( + "part of a `run:` that holds more than one command or a shell reserved word, " + "or quoting this audit cannot split" + ), + "shell_expansion": "a command holding, or inside, a shell expansion this static audit does not evaluate", "expression": "a `run:` holding a `${{ }}` expression, which GitHub substitutes before the shell reads it", "inputs_not_a_mapping": "a step whose `with:` is not a mapping", } diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index e870dad0b..2b75205e9 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -719,9 +719,10 @@ class HostWorkflowAgentLaunchV7(BaseModel): flags the step declares in ``settings``, and the documented widening rules they meet in ``widening_rules``, omitted when none. ``form: unresolved`` names why the step's settings were not - read — a ``run:`` holding more than one command or quoting that does not - balance, a shell expansion, a ``${{ }}`` expression, or ``with:`` that is - not a mapping — with no + read — a ``run:`` holding more than one command, a shell reserved word or + quoting that does not balance, a shell expansion (an agent CLI inside a + command substitution included), a ``${{ }}`` expression, or ``with:`` + that is not a mapping — with no settings, and records a non-blocking coverage issue. ``job_secrets`` names the secrets the step's job references (``${{ secrets.NAME }}``) and the workflow-level ``env`` passes: context for the row that names this step, diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 83390ab73..87513cfc8 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -27,6 +27,7 @@ from agents_shipgate.core.host_grants import ( _claude_argument_input, _codex_argument_input, + _command_substitutions, _literal_argument_words, _uncompared_workflow_text, _workflow_grant, @@ -236,8 +237,17 @@ def test_literal_assignments_before_the_command_are_skipped_and_never_published( 'echo "claude -p --dangerously-skip-permissions"', "echo claude -p done", "npm test", + # a substitution the shell never runs, or one that runs another command + 'echo "\\$(claude -p --dangerously-skip-permissions)"', + "echo '$(claude -p --dangerously-skip-permissions)'", + "echo '`claude -p --dangerously-skip-permissions`'", + 'echo "$(date) claude -p --dangerously-skip-permissions"', + # a reserved word is read only before a command's assignments + "CI=true then claude -p 'go'", ], - ids=["claude-mcp", "claude-version", "codex-login", "echo-quoted", "echo-bare", "unrelated"], + ids=["claude-mcp", "claude-version", "codex-login", "echo-quoted", "echo-bare", "unrelated", + "escaped-substitution", "single-quoted-substitution", "single-quoted-backtick", + "substitution-of-another-command", "reserved-word-after-assignment"], ) def test_a_command_that_launches_no_headless_agent_is_not_listed(run): assert _launches(_workflow({"run": run})) == [] @@ -255,9 +265,25 @@ def test_a_command_that_launches_no_headless_agent_is_not_listed(run): ('claude -p "Fix ${{ github.event.issue.title }}"', "expression"), ("cat < prompt.md\nIt's broken\nEOF\nclaude -p --dangerously-skip-permissions 'go'", "compound_command"), ("claude -p --dangerously-skip-permissions \"go", "compound_command"), + # an agent CLI inside a double-quoted or backtick substitution (#823 review) + ('gh pr comment "$PR" --body "$(claude -p --dangerously-skip-permissions \'go\')"', "shell_expansion"), + ('REVIEW="$(claude -p --dangerously-skip-permissions \'go\')"', "shell_expansion"), + ("REVIEW=`claude -p --dangerously-skip-permissions 'go'`", "shell_expansion"), + ('echo "$(echo "$(claude -p --dangerously-skip-permissions \'go\')")"', "compound_command"), + # quoting that does not balance is read a line at a time, substitutions included + ("cat < prompt.md\nIt's broken\nEOF\nREVIEW=\"$(claude -p --dangerously-skip-permissions 'go')\"", + "compound_command"), + # an agent CLI after a shell reserved word + ("if true; then claude -p --dangerously-skip-permissions 'go'; fi", "compound_command"), + ("for f in a b; do claude -p --dangerously-skip-permissions 'go'; done", "compound_command"), + ("{ claude -p --dangerously-skip-permissions 'go'; }", "compound_command"), + ("! claude -p --dangerously-skip-permissions 'go'", "compound_command"), + ("time -p claude -p --dangerously-skip-permissions 'go'", "compound_command"), ], ids=["and", "lines", "pipe", "heredoc", "variable", "substitution", "expression", "unbalanced-heredoc", - "unbalanced"], + "unbalanced", "quoted-substitution-argument", "quoted-substitution-assignment", "backtick", + "nested-substitution", "unbalanced-substitution", "if-then", "for-do", "brace-group", "negated", + "timed"], ) def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(run, reason): launch, = _launches(_workflow({"run": run})) @@ -270,6 +296,17 @@ def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(r assert "not reported" in limit +def test_deeply_nested_substitutions_are_read_in_one_pass(): + """#823 review: a substitution is found without recursion, and each character is split once.""" + + depth = 20000 + run = 'REVIEW="' + "$(" * depth + "claude -p --dangerously-skip-permissions 'go'" + ")" * depth + '"' + launch, = _launches(_workflow({"run": run})) + assert (launch["agent"], launch["form"], launch["unresolved_reason"]) == ("claude", "unresolved", "shell_expansion") + # Each nested substitution reads as `_` in the one around it. + assert _command_substitutions('echo "$(echo "$(codex exec x)")"') == ["codex exec x", 'echo "_"'] + + def test_a_quoted_word_that_starts_with_a_hash_is_not_a_comment(): launch, = _launches(_workflow({"run": 'claude -p --allowedTools Read "#123 review"'})) assert launch["form"] == "read" @@ -532,6 +569,75 @@ def test_adding_an_unresolved_launch_is_a_row_that_claims_no_effect(): assert "does not say what that agent may do" in row.why +SUBSTITUTED = 'gh pr comment "$PR" --body "$(claude -p {flags} \'Review this change\')"' + + +@pytest.mark.parametrize( + "run", + [ + SUBSTITUTED.format(flags="--dangerously-skip-permissions"), + "if true; then claude -p --dangerously-skip-permissions 'Review this change'; fi", + ], + ids=["quoted-substitution", "if-then"], +) +def test_an_agent_cli_the_splitter_does_not_head_is_a_row_and_a_named_limit(run): + """#823 review: a launch inside a quoted `$(…)`, or after `then`, is named, never missed.""" + + row, = _rows(_reproduction(), _reproduction(run=run)) + assert (row.direction, row.expands) == ("changed", False) + assert "a step now launches an agent (review/steps[2])" in row.why + assert "a step launches an agent in a form this audit does not read (review/steps[2])" in row.why + assert "dangerously" not in row.after + + limit, = uncompared_agent_launch_texts(_grant(_reproduction(run=run))) + assert limit.startswith("the agent launch at review/steps[2] (claude) is ") + + +def test_an_edit_inside_a_quoted_substitution_is_quiet_and_named_as_a_limit(): + before = _workflow({"run": SUBSTITUTED.format(flags="--allowedTools Read")}) + after = _workflow({"run": SUBSTITUTED.format(flags="--dangerously-skip-permissions")}) + + assert _rows(before, after) == [] + limit, = uncompared_agent_launch_texts(_grant(after)) + assert "a command holding, or inside, a shell expansion this static audit does not evaluate" in limit + + +@pytest.mark.parametrize( + ("before", "after"), + [ + ("claude -p --allowedTools Read 'Review'", + "npx @anthropic-ai/claude-code -p --dangerously-skip-permissions 'Review'"), + ("claude -p --allowedTools Read 'Review'", + "./node_modules/.bin/claude -p --dangerously-skip-permissions 'Review'"), + ("codex exec -s workspace-write 'review'", "codex -c sandbox_mode=danger-full-access exec 'review'"), + ], + ids=["npx", "path", "codex-root-options"], +) +def test_a_read_launch_that_becomes_a_form_this_audit_does_not_read_is_not_called_gone(before, after): + """#823 review: what was established is that no launch this audit reads is declared.""" + + row, = _rows(_workflow({"run": before}), _workflow({"run": after})) + assert (row.direction, row.expands) == ("changed", False) + assert "a step no longer declares an agent launch this audit reads (review/steps[0])" in row.why + assert ( + "a step that no longer declares one may still start an agent in a way this audit does not read, " + "such as an action outside its table, `npx`, a script, a path such as `./node_modules/.bin/claude` " + "or `codex` options before `exec`, so this row does not say that it no longer starts one" + ) in row.why + assert "no longer launches an agent" not in row.why + + +def test_a_read_launch_that_moves_into_a_quoted_substitution_is_changed_and_named_unread(): + row, = _rows( + _workflow({"run": "claude -p --allowedTools Read 'Review this change'"}), + _workflow({"run": SUBSTITUTED.format(flags="--dangerously-skip-permissions")}), + ) + assert (row.direction, row.expands) == ("changed", False) + assert "an agent launch's declared settings changed (review/steps[0])" in row.why + assert "a step launches an agent in a form this audit does not read (review/steps[0])" in row.why + assert "no longer" not in row.why + + def test_an_action_ref_bump_is_a_step_reference_change_only(): row, = _rows(_workflow(_agent()), _workflow({**_agent(), "uses": "anthropics/claude-code-action@v2"})) @@ -1603,6 +1709,34 @@ def test_an_unresolved_launch_is_a_non_blocking_limit_that_leaves_coverage_compl assert "review/steps[0] (claude)" in issue["message"] +def test_the_quoted_substitution_of_the_review_is_a_row_in_diff_and_a_coverage_issue(tmp_path): + """#823 review: `--body "$(claude -p …)"` gave no row, no launch and no coverage issue.""" + + from agents_shipgate.cli.host_audit import host_audit_inventory + + repo = _repo(tmp_path, {SOURCE: _yaml(_reproduction())}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(_reproduction(run=SUBSTITUTED.format(flags="--dangerously-skip-permissions")))}) + _git(repo, "commit", "-qam", "comment with a review") + + row, = _diff(repo)["rows"] + assert (row["direction"], row["expands"]) == ("changed", False) + assert "a step launches an agent in a form this audit does not read (review/steps[2])" in row["why"] + text = CliRunner().invoke(app, ["diff", "--workspace", str(repo), "--base", "main"]) + assert text.exit_code == 0, text.output + assert "No static host-grant changes detected" not in text.output + assert "review/steps[2]" in text.output and "dangerously" not in text.output + + inventory = host_audit_inventory(repo) + workflow, = [grant for grant in inventory["grants"] if grant.get("kind") == "workflow"] + assert [(item["step"], item["form"], item["unresolved_reason"]) for item in workflow["agent_launches"]] == [ + ("steps[1]", "read", None), ("steps[2]", "unresolved", "shell_expansion"), + ] + issue, = [item for item in inventory["issues"] if item["host"] == "github"] + assert (issue["kind"], issue["blocking"]) == ("unsupported", False) + assert "review/steps[2] (claude)" in issue["message"] + + # --- the same row on every route --------------------------------------------------- From 846763fd286567b9e91db503d0720e49849d6d60 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 05:35:24 -0700 Subject: [PATCH 08/14] Address review cycle 3 on agent launches in CI (#823) A step that drafts a PR comment in a quoted here-doc, such as cat > comment.md <<'EOF' with "Reproduce locally with `claude -p ...`" in its body, was listed as an unresolved agent launch: a "high changed" row saying a step now launches an agent, and a GitHub coverage issue. The shell passes a here-doc's body to its command as input and, with a quoted delimiter, expands nothing in it, so the step starts no agent. The phantom also hid a real gain: a --dangerously-skip-permissions step added beside it was "not counted as a widening" because the job "launched the agent in a form this audit does not read" before. The substitution scanner read backticks and $( across the whole run:, and the word splitter read a body line starting with claude -p as a command head. _here_documents now removes each here-doc's body and closing line before the run: is split or scanned. <<, <<- and a quoted, escaped or partly quoted delimiter are read outside quotes, comments and $((...)) arithmetic, inside a $(...) or backtick substitution too; <<< is a here-string; bodies start after the opening line and follow in order when one line opens several. A body is never a command. Only an unquoted <` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, while a job that remains has not been left, since it may still run its launch in a form this audit does not read. A compound command, an expansion or an expression — an agent CLI inside a quoted `$(…)` or a backtick substitution, or after a reserved word such as `then`, included — is a named non-blocking limit and publishes none of its text, while a here-doc's body is input and never a command, so a Markdown `` `claude -p …` `` in a comment drafted with `cat <<'EOF'` is no launch and only an unquoted `<` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index d018c1c43..f692b49c2 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -359,9 +359,9 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin - **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. - **What is withheld.** A structured value publishes its shape and none of its free text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex `--config` table or array publish, as canonical JSON, their key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. A codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. Other argument text — a prompt, a flag's value, a codex `--config` override's scalar value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to such a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists or no longer launches that agent, or the same launch now runs elsewhere), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. -- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command or a shell reserved word such as `then` or `!`, a shell expansion — an agent CLI heading a command inside a `$(…)` or backtick substitution, double-quoted or not, included — or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), and a ref that is not a string, has `value`/`ref: null`. A setting holding credential-shaped text is published redacted (`redacted`). Each records a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`, and adding, removing or re-forming such an entry, or its gaining a rule, is still a row. Only an edit inside it that gains no rule is not reported. +- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command or a shell reserved word such as `then` or `!`, a shell expansion — an agent CLI heading a command inside a `$(…)` or backtick substitution, double-quoted or not, included — or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A here-doc's body is input to its command, never a command: a `claude -p` line in it, or a backticked `` `claude -p …` `` in a body whose delimiter is quoted (`<<'EOF'`, `<<"EOF"`, `<<\EOF`), is no launch, while an agent CLI heading a `$(…)` or backtick substitution in an unquoted `< str: f"a step launches an agent in a form this audit does not read ({', '.join(unread)}); " "its settings are not compared, so this row does not say what that agent may do" ) - checkouts = list(dict.fromkeys( - f"{item['job']}/{item['step']}" for item in (*gone_checkouts, *new_checkouts) - )) - if checkouts: + # A checkout step on one side only, such as one in an added job, is + # worded as added or removed, not as a changed ref (#823 review cycle 3). + old_checkouts = {where(item) for item in gone_checkouts} + new_checkouts_at = {where(item) for item in new_checkouts} + checkout_groups: dict[str, list[str]] = {} + for item in (*new_checkouts, *gone_checkouts): + label = where(item) + verb = ( + "changed" if label in old_checkouts and label in new_checkouts_at + else ("added" if label in new_checkouts_at else "removed") + ) + checkout_groups.setdefault(verb, []) + if label not in checkout_groups[verb]: + checkout_groups[verb].append(label) + checkout_phrases = [ + f"{wording} ({', '.join(checkout_groups[verb])})" + for verb, wording in ( + ("changed", "a checkout's declared ref changed"), + ("added", "a step now declares a checkout"), + ("removed", "a step no longer declares a checkout"), + ) + if verb in checkout_groups + ] + if checkout_phrases: reasons.append( - f"a checkout's declared ref changed ({', '.join(checkouts)}); a ref names which " - "commit's code the job runs and adds no scope" + f"{_joined_words(checkout_phrases)}; a ref names which commit's code the job runs " + "and adds no scope" ) return reasons diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 6e0c26f63..93fb029e8 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -8,6 +8,7 @@ from __future__ import annotations +import bisect import errno import hashlib import json @@ -1871,7 +1872,7 @@ class _SubstitutionLevel: cursor: int = 0 -def _command_substitutions(text: str) -> list[str]: +def _command_substitutions(text: str, *, here_doc_body: bool = False) -> list[str]: """The text of each ``$(…)`` and backtick command substitution the shell would run in ``text``. Found outside single quotes, inside double quotes too: the word splitter @@ -1884,6 +1885,12 @@ def _command_substitutions(text: str) -> list[str]: substitutions, though one inside arithmetic still is, and one left open runs to the end of the text. Only used to name an agent CLI as unresolved: nothing found here is read or published. + + ``here_doc_body`` reads ``text`` as the body of a here-document whose + delimiter is unquoted (:func:`_here_documents`): the shell runs its + substitutions, but quotes and ``#`` in it are ordinary characters, so + ``'$(claude -p …)'`` there is a substitution. Inside a substitution the + usual quoting applies again. """ bodies: list[str] = [] @@ -1918,7 +1925,10 @@ def close_level(end: int) -> None: open_level(index, ")", index + 2, arithmetic=text.startswith("((", index + 1)) index += 2 continue - if level.quoted: + if here_doc_body and len(levels) == 1: + # Outside a substitution, a here-document body quotes nothing. + pass + elif level.quoted: level.quoted = char != '"' elif char == '"': level.quoted = True @@ -1947,6 +1957,176 @@ def close_level(end: int) -> None: return bodies +def _here_doc_delimiter(text: str, index: int) -> tuple[str, bool, int]: + """The here-document delimiter word at ``index``: its text after quote removal, whether any of it is quoted, and where it ends. + + An unclosed quote gives an empty delimiter, which is not read as one. + """ + + word: list[str] = [] + quoted = False + while index < len(text) and not text[index].isspace() and text[index] not in _SHELL_OPERATOR_CHARS: + char = text[index] + if char == "\\": + quoted = True + word.append(text[index + 1:index + 2]) + index += 2 + elif char in {"'", '"'}: + end = text.find(char, index + 1) + if end < 0: + return "", quoted, len(text) + quoted = True + word.append(text[index + 1:end]) + index = end + 1 + else: + word.append(char) + index += 1 + return "".join(word), quoted, index + + +@dataclass +class _LineStarts: + """Where each line of a text starts, by its content and by its content after leading tabs.""" + + exact: dict[str, list[int]] + tab_stripped: dict[str, list[int]] + + @classmethod + def of(cls, text: str) -> _LineStarts: + exact: dict[str, list[int]] = {} + tab_stripped: dict[str, list[int]] = {} + start = 0 + for line in text.split("\n"): + exact.setdefault(line, []).append(start) + tab_stripped.setdefault(line.lstrip("\t"), []).append(start) + start += len(line) + 1 + return cls(exact=exact, tab_stripped=tab_stripped) + + +def _here_doc_bodies_end( + text: str, lines: _LineStarts, start: int, pending: list[tuple[str, bool, bool]], +) -> tuple[int, list[str]] | None: + """Where the bodies of ``pending`` here-documents, read in order from ``start``, end, and the bodies the shell expands. + + Each body runs to the first line that is exactly its delimiter (after + leading tabs, for ``<<-``). ``None`` when one has no such line. The line + is looked up rather than searched for, so a text of many here-documents + with no closing line costs no more than its length times a logarithm. + """ + + position = start + expanded: list[str] = [] + for delimiter, quoted, strip_tabs in pending: + starts = (lines.tab_stripped if strip_tabs else lines.exact).get(delimiter, []) + found = bisect.bisect_left(starts, position) + if found == len(starts): + return None + closing = starts[found] + if not quoted: + expanded.append(text[position:closing]) + line_end = text.find("\n", closing) + position = len(text) if line_end < 0 else line_end + 1 + return position, expanded + + +def _here_documents(text: str) -> tuple[str, list[str]]: + """``text`` without the body and closing line of each here-document, and the bodies the shell expands. + + The shell never runs a here-document's lines as commands: it passes them + to the command as input. When any part of the delimiter is quoted + (``<<'EOF'``, ``<<"EOF"``, ``<<\\EOF``) the body is passed as written, so + ``claude -p`` in Markdown backticks there starts nothing (#823 review); + when it is unquoted (``<= 0), default=len(text)) + continue + elif char == "<" and text.startswith("<<", index) and not level.arithmetic: + if text.startswith("<<<", index): + index += 3 + continue + index += 2 + strip_tabs = text.startswith("-", index) + if strip_tabs: + index += 1 + while index < len(text) and text[index] in " \t": + index += 1 + delimiter, quoted, index = _here_doc_delimiter(text, index) + if delimiter: + pending.append((delimiter, quoted, strip_tabs)) + continue + elif level.closer == ")" and char == "(": + level.depth += 1 + elif level.closer == ")" and char == ")": + if level.depth: + level.depth -= 1 + else: + levels.pop() + index += 1 + kept.append(text[cursor:]) + return "".join(kept), expanded + + def _commands(words: list[str]) -> list[list[str]]: """``words`` grouped into the commands their operator tokens separate, empty ones included.""" @@ -1959,11 +2139,11 @@ def _commands(words: list[str]) -> list[list[str]]: return commands -def _substituted_agents(run: str) -> set[str]: +def _substituted_agents(run: str, *, here_doc_body: bool = False) -> set[str]: """The agent CLIs ``run`` starts at the head of a command inside a command substitution.""" agents: set[str] = set() - for body in _command_substitutions(run): + for body in _command_substitutions(run, here_doc_body=here_doc_body): words = _shell_words(body) if words is None: agents.update(agent for line in body.splitlines() if (agent := _line_agent(line))) @@ -2775,21 +2955,28 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: flags are its settings. Otherwise an agent CLI found at the start of any command in it — after any reserved words, and inside a ``$(…)`` or backtick substitution outside single quotes — is one ``unresolved`` entry - with the reason, and none of its text is published. + with the reason, and none of its text is published. A here-document's + body is never a command, and only an unquoted delimiter's body has + substitutions the shell runs (:func:`_here_documents`). """ run = run.strip() - words = _shell_words(run) + # The commands the shell runs, without the here-document bodies it passes + # them as input; a step with one is never a simple command, since `<<` + # stays in `script`, so what is published is always read from `run`. + script, expanded = _here_documents(run) + here_doc_agents = {agent for body in expanded for agent in _substituted_agents(body, here_doc_body=True)} + words = _shell_words(script) base = {"job": job, "step": step_label} if words is None: # Quoting that does not balance cannot be split into commands, so no # word of it is read; a line that starts an agent CLI, or holds a # substitution that does, is still named. agents = sorted({ - agent for line in run.splitlines() + agent for line in script.splitlines() for agent in ({_line_agent(line)} | _substituted_agents(line)) if agent is not None - }) + } | here_doc_agents) return [ {**base, "agent": agent, "form": "unresolved", "unresolved_reason": "compound_command", "settings": []} for agent in agents @@ -2799,7 +2986,7 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: # An agent CLI inside a `$(…)` or backtick substitution heads no command # the splitter sees when the substitution is double-quoted or a backtick # (#823 review), and is named as unresolved. - substituted = _substituted_agents(run) + substituted = _substituted_agents(script) | here_doc_agents if not launches and not substituted: return [] # A comment is an unquoted `#` at the start of a word, read off the raw @@ -2809,14 +2996,14 @@ def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: single = ( len(simple) == 1 and not any(_is_operator(word) for word in words) - and not _has_shell_comment(run) + and not _has_shell_comment(script) and not _reserved_prefix_length(simple[0]) ) if "${{" in run: reason: str | None = "expression" elif not single: reason = "compound_command" - elif substituted or _has_shell_expansion(run): + elif substituted or _has_shell_expansion(script): reason = "shell_expansion" else: reason = None @@ -3155,10 +3342,15 @@ class AgentRuleGains: a ``${{ }}`` expression in an input the rule is read from, and GitHub's substituted text may already have met it. Each names that input. - ``moved``: the rule left another job whose launch that met it left that - job — the job no longer exists or no longer launches that agent, or the - same launch now runs here — as when a job is renamed or an agent step - moves to another job (#823 review). Each names the launch it left, the - way a step reference moved between jobs adds no scope (#771). + job — the job no longer exists, or the same launch now runs here while + each launch of that agent this job had still runs here or in that job — + as when a job is renamed, an agent step moves to another job or two + jobs swap launches (#823 review). A job that remains may still run its + launch in a form this reader does not read, so a launch that only stops + being read has not left it, and one edited in place into the launch + that job had gains the rule (#823 review cycle 3). Each names the + launch it left, the way a step reference moved between jobs adds no + scope (#771). """ claimed: list[AgentWidening] @@ -3194,16 +3386,33 @@ def launches(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[st lost = [key for key in old if key not in new] gained = [key for key in new if key not in old] - def same_launch(key: _RuleKey, arriving: list[dict[str, Any]]) -> bool: - # The launch that met the rule in the losing job now runs here. - return any( - agent_launch_key(entry)[1:] == agent_launch_key(other)[1:] - for entry in old[key] for other in arriving - ) - - def job_left(key: _RuleKey, _arriving: list[dict[str, Any]]) -> bool: - # The losing job no longer launches that agent at all. - return (key[0], key[1]) not in launched_after + def compared(entries: list[dict[str, Any]]) -> set[tuple[Any, ...]]: + # A launch as it is compared, less its job. + return {agent_launch_key(entry)[1:] for entry in entries} + + def same_launch(source: _RuleKey, target: _RuleKey) -> bool: + # The launch that met the rule in the losing job now runs here, and + # none of this job's own launches of that agent changed in place: + # each still runs here or now runs in the losing job, as when two + # jobs swap. A job whose launch was edited into the one the losing + # job had, while the losing job still exists, gained it (#823 + # review cycle 3). + if not compared(old[source]) & compared(new[target]): + return False + family = target[1] + remaining = compared([ + entry for job in (target[0], source[0]) for entry in launched_after.get((job, family), []) + ]) + return compared(launched_before.get((target[0], family), [])) <= remaining + + jobs_after = {str(context["job"]) for context in (after or {}).get("permission_contexts", [])} + + def job_left(source: _RuleKey, _target: _RuleKey) -> bool: + # The losing job no longer exists, as when it is renamed. A job that + # remains may still run its launch in a form this reader does not + # read, such as `npx`, so its launch is not taken to have left + # (#823 review cycle 3). + return source[0] not in jobs_after # A rule moved when the launch that met it left the losing job. The same # launch arriving is matched first, so which of two gaining jobs a rule @@ -3214,7 +3423,7 @@ def job_left(key: _RuleKey, _arriving: list[dict[str, Any]]) -> bool: if key in sources: continue source = next( - (other for other in lost if other[1:] == key[1:] and left(other, new[key])), None + (other for other in lost if other[1:] == key[1:] and left(other, key)), None ) if source is not None: lost.remove(source) diff --git a/src/agents_shipgate/schemas/contract.py b/src/agents_shipgate/schemas/contract.py index 081668c13..8238c8cc1 100644 --- a/src/agents_shipgate/schemas/contract.py +++ b/src/agents_shipgate/schemas/contract.py @@ -214,7 +214,7 @@ # ``host_comparison.coverage`` and ``shipgate diff --json`` (capability diff # 0.3) the same block. It is evidence beside the rows and moves no state, # permission, route or row. A 0.19 verifier reads with coverage not recorded. -# v41, unreleased, carries two changes. It names the changed inputs a host +# v41, unreleased, carries three changes. It names the changed inputs a host # comparison does not read (#821): verifier 0.21 and ``shipgate diff --json`` # (capability diff 0.4) add a ``changed_not_read`` coverage item with its # ``candidate`` rule, a ``read_sources_only`` that is ``false`` while one is @@ -238,7 +238,7 @@ # (``baseline_workflow_agent_launches_unavailable``); one without a workflow # stays comparable. #823 moves neither verifier 0.21 nor capability diff 0.4: # its rows keep their shape. ``MINIMUM_CONTROL_CONTRACT_VERSION`` stays at 21. -# v41 also keeps what a comparison established outside a plugin directory it +# And v41 keeps what a comparison established outside a plugin directory it # could not compare (#808), extended in place because v41 is unreleased: where # every blocking limit is a plugin-reference limit that its plugin directory # bounds, and no compared source depends on that directory, verifier 0.21 and diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 87513cfc8..7e1fbfab1 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -138,6 +138,23 @@ def test_a_head_ref_checkout_is_one_changed_row_naming_the_default_and_the_new_r ) in row.why +def test_a_checkout_step_on_one_side_only_is_worded_as_added_or_removed(): + """#823 review cycle 3 (P3): a default checkout in an added job is not a changed ref.""" + + checkout = {"uses": "actions/checkout@v4"} + before = _jobs(test=[checkout, {"run": "make test"}]) + after = _jobs(test=[checkout, {"run": "make test"}], lint=[checkout, {"run": "make lint"}]) + + row, = _rows(before, after) + assert "lint/steps[0]: checkout of the default ref" in row.after + assert "a step now declares a checkout (lint/steps[0]); a ref names which commit's code" in row.why + assert "declared ref changed" not in row.why + + removed, = _rows(after, before) + assert "a step no longer declares a checkout (lint/steps[0])" in removed.why + assert "declared ref changed" not in removed.why + + def test_an_untrusted_trigger_with_a_write_scope_names_the_agent_step_it_now_reaches(): row, = _rows(_reproduction(), _reproduction(trigger="issue_comment", pr="write")) @@ -270,9 +287,12 @@ def test_a_command_that_launches_no_headless_agent_is_not_listed(run): ('REVIEW="$(claude -p --dangerously-skip-permissions \'go\')"', "shell_expansion"), ("REVIEW=`claude -p --dangerously-skip-permissions 'go'`", "shell_expansion"), ('echo "$(echo "$(claude -p --dangerously-skip-permissions \'go\')")"', "compound_command"), - # quoting that does not balance is read a line at a time, substitutions included + # a here-document body is not read, so an apostrophe in it no longer unbalances the rest ("cat < prompt.md\nIt's broken\nEOF\nREVIEW=\"$(claude -p --dangerously-skip-permissions 'go')\"", "compound_command"), + # quoting that does not balance is read a line at a time, substitutions included + ("echo 'broken\nclaude -p --dangerously-skip-permissions go", "compound_command"), + ("echo 'broken\nREVIEW=\"$(claude -p --dangerously-skip-permissions go)\"", "compound_command"), # an agent CLI after a shell reserved word ("if true; then claude -p --dangerously-skip-permissions 'go'; fi", "compound_command"), ("for f in a b; do claude -p --dangerously-skip-permissions 'go'; done", "compound_command"), @@ -280,10 +300,10 @@ def test_a_command_that_launches_no_headless_agent_is_not_listed(run): ("! claude -p --dangerously-skip-permissions 'go'", "compound_command"), ("time -p claude -p --dangerously-skip-permissions 'go'", "compound_command"), ], - ids=["and", "lines", "pipe", "heredoc", "variable", "substitution", "expression", "unbalanced-heredoc", + ids=["and", "lines", "pipe", "heredoc", "variable", "substitution", "expression", "after-heredoc", "unbalanced", "quoted-substitution-argument", "quoted-substitution-assignment", "backtick", - "nested-substitution", "unbalanced-substitution", "if-then", "for-do", "brace-group", "negated", - "timed"], + "nested-substitution", "substitution-after-heredoc", "unbalanced-lines", "unbalanced-substitution", + "if-then", "for-do", "brace-group", "negated", "timed"], ) def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(run, reason): launch, = _launches(_workflow({"run": run})) @@ -602,6 +622,149 @@ def test_an_edit_inside_a_quoted_substitution_is_quiet_and_named_as_a_limit(): assert "a command holding, or inside, a shell expansion this static audit does not evaluate" in limit +# --- here-documents (#823 review cycle 3, C3-F1) ------------------------------------- + +#: The step of the review: a PR comment drafted in a quoted here-document. +HERE_DOC_COMMENT = "cat > comment.md <<'EOF'\nReproduce locally with `claude -p \"review this change\"`.\nEOF\n" +COMMENT_PERMISSIONS = {"contents": "read", "pull-requests": "write"} + + +@pytest.mark.parametrize( + ("opener", "closer"), + [("<<'EOF'", "EOF"), ('<<"EOF"', "EOF"), ("<<\\EOF", "EOF"), ("<<-'EOF'", "\t\tEOF"), ("<< 'EOF'", "EOF"), + ("< comment.md {opener}\n{body}\n{closer}\n"} + assert _launches(_workflow(step, permissions=COMMENT_PERMISSIONS)) == [] + assert _rows( + _workflow({"run": "echo done"}, permissions=COMMENT_PERMISSIONS), + _workflow({"run": "echo done"}, step, permissions=COMMENT_PERMISSIONS), + ) == [] + + +def test_beside_a_quoted_here_doc_a_new_bypass_step_is_a_widening(): + """The here-document is no launch the job had before, so the gain is claimed.""" + + here_doc = {"run": HERE_DOC_COMMENT} + before = _workflow(here_doc, permissions=COMMENT_PERMISSIONS) + after = _workflow( + here_doc, {"run": 'claude -p --dangerously-skip-permissions "Review"'}, permissions=COMMENT_PERMISSIONS, + ) + + assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[1])" in row.why + assert "a form this audit does not read" not in row.why + assert "review/steps[0]" not in row.why + row.before + row.after + + +@pytest.mark.parametrize( + ("body", "agent"), + [ + ("claude -p --dangerously-skip-permissions 'review'", None), + ("Summary: $(claude -p 'review')", "claude"), + ("Summary: `codex exec 'review'`", "codex"), + # quotes and `#` are ordinary characters in the body + ("'$(claude -p review)'", "claude"), + ("# $(claude -p review)", "claude"), + ("\\$(claude -p review)", None), + ], + ids=["line-head", "substitution", "backtick", "single-quoted", "hash", "escaped"], +) +def test_an_unquoted_here_doc_runs_its_substitutions_and_none_of_its_lines(body, agent): + launches = _launches(_workflow({"run": f"cat > comment.md < prompt.md <<'EOF'\nreview\nEOF\nclaude -p --dangerously-skip-permissions 'go'", + # two on one line, their bodies in order + "cat <<'A' < Date: Wed, 23 Sep 2026 06:28:53 -0700 Subject: [PATCH 09/14] Name an unchanged instruction limit reached through an in-tree link instead of refusing (#822) An instruction file whose limit this entry cannot resolve, such as a SKILL.md whose metadata holds a non-string value, is named as an unchanged limit when a change leaves it alone (#721). Reached through an in-tree link the reader reads through (#700) - `.claude/skills -> ../.agents/skills`, a per-skill link or a file link - the same untouched file refused the whole comparison: diff, verify and the manifest-free PR comment printed `base_inventory_incomplete; head_inventory_incomplete` and no row, hiding a removed deny rule beside it. `unchanged_limits` asked blob_path_unchanged whether the source, the path the link is read under, was one regular file in Git, and a path through a link never is. blob_path_unchanged now resolves the path on each side the way the reader reaches it: from Git tree entries for the base and a commit head, and for a working-tree head without following any link (each component's own entry, each link's own text, the file's unfiltered hash). It holds only when both resolutions are equal - every link at the same path with the same text, every other component a directory - and the file they land on is the same regular-file blob at the same in-tree path. A link is followed only under the rules the reader and the base archive already use (#700, #711): a relative text that lands inside the tree after normalization, directories above where it lands, at most eight links, and links at one component of the path. A path no link reaches still takes the one `ls-tree` it took before, and blob object IDs are still compared, so no filter or textconv can make two byte sequences equal. Everything that asks the proof moves together: unchanged_limits in `diff --json` and verifier.json, the text and PR comment, the unchanged limits a partial comparison may carry (#808) and the shared plugin-reference limits check leaves out (#714). check's boundary result still cannot name a limit, so it now refuses these comparisons with unchanged_limits_not_representable, as it does for a limit at its own path; its rows, decision and violations do not move. No schema, member, reason code or check id is added. The metadata value is not coerced. tests/test_linked_unchanged_limits.py holds every layout (directory link, per-skill link, file link, a chain of file links) to the direct result on diff, verify, the PR comment and check; keeps the negative controls refused (skill added or edited behind the link, link retargeted to an identical copy, link text rewritten to land on the same file, link replaced by a directory or the reverse, a later hop retargeted, working-tree-only retargets and edits); and holds the proof to exactly the links the reader reads through, across dangling, looping, absolute, escaping, over-long, intermediate and nested links. The #812 coverage case that pinned the refusal is retired, and the static-only allowlist follows its two call sites down the file. --- CHANGELOG.md | 37 ++ STABILITY.md | 86 ++++ docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 7 +- src/agents_shipgate/cli/verify/git.py | 231 +++++++++-- src/agents_shipgate/core/host_comparison.py | 11 +- tests/test_adapter_static_only.py | 10 +- tests/test_distribution_surface_parity.py | 7 +- tests/test_host_comparison_coverage.py | 12 +- tests/test_linked_unchanged_limits.py | 427 ++++++++++++++++++++ 10 files changed, 782 insertions(+), 48 deletions(-) create mode 100644 tests/test_linked_unchanged_limits.py diff --git a/CHANGELOG.md b/CHANGELOG.md index d98d51771..82fc687e1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -134,6 +134,43 @@ - No schema, contract, member, error kind, refusal code or exit code moves. See the `STABILITY.md` migration note. (#807) +- An unchanged instruction file reached through an in-tree link no longer + refuses the whole host comparison. A `SKILL.md` whose `metadata` holds a + non-string value, which this entry cannot resolve, is named as an unchanged + limit when the change leaves it alone (#721). Read through + `.claude/skills -> ../.agents/skills`, a per-skill link or a file link, the + same untouched file made `diff`, `verify` and the manifest-free PR comment + print `Cannot compare against HEAD~1: base_inventory_incomplete; + head_inventory_incomplete` and no row, hiding a removed `deny` rule beside it. + - **The cause.** The unchanged proof asked whether the path the link is read + under, `.claude/skills/review/SKILL.md`, was one regular file in Git, and a + path through a link never is. + - **Now.** The proof follows the link as the reader does (#700), from Git + tree entries, and holds only when both are unchanged: the link, a link at + the same path with the same text at each link on the way, and the file it + lands on, the same blob at the same in-tree path. A working-tree head is + read without following any link, each link by its own text and the file by + its unfiltered hash. The limit is then named and the rest compared, exactly + as when the skill sits at its own path: the issue's reproduction shows + `⚠ low removed claude-code .claude/settings.json`, + `deny: Bash(curl *) → gone` for the directory link and the per-skill link + alike. + - **Still refused.** A skill added or edited behind the link; the link + retargeted, even to an identical copy, or its text rewritten to land on the + same file; a link replaced by a directory holding the same bytes, or the + reverse; and every link the reader does not read through (dangling, + looping, absolute, escaping, past eight links, or inside a linked + directory), which stays `unreadable` and is never an unchanged limit. The + metadata value is not coerced: `internal: true` stays an `unsupported` + structure (#811). + - **`check`.** Its boundary result still cannot name a limit, so it now + refuses such a comparison with `unchanged_limits_not_representable`, the + reason it gives for a limit at its own path, instead of + `base_inventory_incomplete` / `head_inventory_incomplete`. Rows stay empty + and its decision and violations do not move. + - No schema, contract, member, reason code or check id is added, and + host-grants stays `0.6`. See the `STABILITY.md` migration note. (#822) + ## 1.1.0 - 2026-09-22 A legibility and presentation-correctness release on the advisory channel. diff --git a/STABILITY.md b/STABILITY.md index d49c517b0..c757f7a8c 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -594,6 +594,92 @@ Outside a repository a preview still declares no snapshot and still reads. `verification_plan` through the pointer reads `verification-plan.json` or `verify-run.json` directly; neither was ever current evidence for a preview. + + +## Migration Note: Unreleased — an unchanged limit reached through an in-tree link is named, not a refusal (#822) + +No schema, member, reason code, check id or `minimum_control_contract_version` +moves, and host-grants stays `0.6`. This change adds no version of its own: +capability diff, verifier and the runtime contract are what the rest of the +unreleased tree carries. What moves is which comparisons name an unchanged +limit instead of refusing, and so which are `comparable`. + +Since `1.0.0`, a per-source `unsupported` or `parse_failed` limit that both +sides share on an untouched file is named in `unchanged_limits` and the rest +is compared ([#721](#unchanged-comparison-limits-contract-v37-721)). The host +reader reads an in-tree link at a boundary path through to its target +([#700](#link-read-through-at-boundary-paths-contract-v39-700)), +but the unchanged proof accepted only a file at its own path, so a limit on a +file read through a link refused the whole comparison. With a skill whose +`metadata.internal` is `true`, which this entry's bounded profile does not +accept, and a change that only drops `deny: Bash(curl *)` from +`.claude/settings.json`: + +| Where the unchanged skill lives | `1.1.0` `diff`, `verify` and PR comment | now | +| --- | --- | --- | +| `.claude/skills/review/SKILL.md` | `comparable`; the limit named; the removed `deny` row | unchanged | +| `.agents/skills/review/SKILL.md`, read through `.claude/skills -> ../.agents/skills` | `incomparable`, `base_inventory_incomplete; head_inventory_incomplete`, no row | `comparable`; the limit named on `.claude/skills/review/SKILL.md`, and on `.agents/skills/review/SKILL.md` for each host that reads it there; the removed `deny` row | +| `skills/review/SKILL.md`, read through `.claude/skills/review -> ../../skills/review` | the same refusal | `comparable`; the limit named; the removed `deny` row | +| a file link at `.claude/skills/review/SKILL.md`, or a chain of file links | the same refusal | the same | + +- **The proof.** A limit is still named only when it is present with the same + kind, host and source on both sides and its source is unchanged. For a + source no link reaches, unchanged means what it did: the same regular-file + blob at that path, by Git object ID between commits and by unfiltered hash + against a working tree. For a source the reader reaches through an in-tree + link, two things must hold on both sides: + - **the link:** at each link the resolution follows, a link entry at the + same path with the same text, and a directory at every other component; + - **what it lands on:** the same regular-file blob at the same in-tree path. + + The base and a commit head are read from their Git tree entries, never from + the archived tree the reader read. A working-tree head is read without + following any link: each link by its own text, each directory by its own + entry, and the file by its unfiltered hash against the base's blob. No + `textconv` or filter runs. +- **Only links the reader reads through.** A link is followed only as the + reader and the base archive already follow one (#700, #711): a relative text + that lands inside the repository after normalization, only directories above + where it lands, a chain that ends at a directory or regular file within eight + links, and links at one component of the path only, since a directory the + reader reads through holds no link. A link it does not read through is an + `unreadable` limit, and `unreadable` is never named as unchanged, so a + dangling, looping, absolute or escaping link, a chain past eight links, a + link above where another lands, and a link inside a linked directory all + still refuse, unchanged or not. +- **Still refused.** Any change to the link or to what it lands on: a skill + added or edited behind the link, a link retargeted, even to an identical + copy, or rewritten to land on the same file (`../../skills/review` to + `../../skills/./review`), a link replaced by a directory holding the same + bytes or the reverse, and any later link of a chain retargeted. +- **Not coerced.** The metadata value is not reinterpreted: `internal: true` + is still an unresolved structure, `unsupported`, published with the same + `detail` as at a direct path. `internal: "true"` was and is read, with no + limit. +- **Who reads the proof.** One function answers it for every consumer, so they + move together: `unchanged_limits` in `diff --json` and `verifier.json`; the + `Not compared: unchanged in this change and not read` list of `diff`, + `verify` text and the PR comment; the unchanged limits a `partial` + comparison may carry beside its withheld directories + ([#808](#partial-host-comparison-808)); and the shared plugin-reference + limits `check` leaves out of its comparison (#714). `check`'s boundary result + still cannot name a limit, so where the other routes now compare past one it + refuses with `unchanged_limits_not_representable`, the reason it already + gives for a limit at its own path, instead of `base_inventory_incomplete` / + `head_inventory_incomplete`. Its rows stay empty, and its decision, + violations and control do not move. +- **Not changed.** The inventory, its issues and `resolved_through`, saved + host-grants baselines, drift, `audit --host`, every row and every control + state and route. The coverage block asks its own identity question and asks + none for a file read through a link, so a zero-row file read that way is + still `unchanged_not_proven`. The host-config and cold-start benchmark + replays reproduce their run-of-record scores. + +**Compatibility.** No field changes shape. A consumer that switches on +`comparison_status` reads `comparable` where it read `incomparable` for these +layouts, with the limit in `unchanged_limits` exactly as a limit at its own +path has been published since `1.0.0`. + ## Migration Note: 1.1.0 — what each host comparison established (verifier `0.20`, capability diff `0.3`, contract v40, #812) diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index 95a975151..c7ac50839 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A surface the reader reaches through an in-tree link it reads through qualifies only when that link, a link with the same text at each link on the way, and the file it lands on, the same blob at the same path, are both unchanged, read from the base's Git tree entries against a commit's or, without following any link, the working tree's; any change to either refuses as before (#822, `tests/test_linked_unchanged_limits.py`). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 62d780c59..7787e607f 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -26,7 +26,12 @@ A registered adapter reports `complete`, `not_applicable`, `partial`, or external, unresolved-symlink, or unsupported input prevents a complete control result. An in-tree symlink at a boundary path is read at its target and recorded in `resolved_through` (#700), and so is a link anywhere that points at an -in-tree file. A dangling link, a link that leaves the repository, or a +in-tree file. A limit on a file read that way, such as a skill whose +structure could not be established behind `.claude/skills -> ../.agents/skills`, +is named as unchanged, like one at its own path, only when the link, with the +same text at each link on the way, and the file it lands on are both +unchanged between the compared sides (#822); any change to either still +refuses the comparison. A dangling link, a link that leaves the repository, or a directory link outside the boundary paths refuses the whole comparison, even when no host file sits behind it (#688). That refusal is deliberate (#659), and it caused every widening the 1.0 host-config measurement missed. diff --git a/src/agents_shipgate/cli/verify/git.py b/src/agents_shipgate/cli/verify/git.py index a3b9e638b..8ab913c5d 100644 --- a/src/agents_shipgate/cli/verify/git.py +++ b/src/agents_shipgate/cli/verify/git.py @@ -1061,27 +1061,94 @@ def path_present_at_ref(workspace: Path, ref: str, path: Path) -> bool | None: return False -def blob_path_unchanged(workspace: Path, base: str, head: str | None, path: str) -> bool: - """Whether ``path`` is the same regular file at ``base`` and at ``head``. - - ``head=None`` means the working tree. The answer is ``False`` whenever - identity cannot be proven: a path absent on either side, a symlink or a - symlinked parent, a tree or submodule, an unreadable file, or any Git - failure. Blob object IDs are compared rather than `git diff` output, so a - ``.gitattributes`` filter or textconv cannot make two different byte - sequences read as equal (#721). +#: The longest link text a tree entry is read for. No platform accepts a +#: longer path, so a longer text is never followed. +_MAX_LINK_TEXT_BYTES = 4096 + +#: How the host reader reaches a file: each link it follows, with that link's +#: own text, and the in-tree path of the regular file it opens (#822). +_ReaderPath = tuple[tuple[tuple[str, str], ...], str] + + +def _reader_path( + path: str, + kind: Callable[[str], str | None], + link_text: Callable[[str], str | None], +) -> _ReaderPath | None: + """How the host reader reaches the file at ``path`` on one side, or ``None`` (#822). + + ``kind`` names the entry at an in-tree path without following it: + ``"tree"``, ``"blob"`` (a regular file) or ``"link"``, and ``None`` for + anything else or nothing. ``link_text`` is a link entry's own text. + + A link is followed only as the reader follows one (#700, the reader's + ``_resolve_in_tree_link``) and the archive would also resolve it (#711, + :func:`_resolve_tree_link`): its text is relative and lands inside the + tree after normalization, every directory above where it lands is a + directory and not a link, and the chain ends at a directory or regular file + within :data:`_MAX_TREE_LINK_HOPS` links. A path holds links at one + component only, because a directory the reader reads through holds no + link, and every other component must be a directory. ``None`` means the + reader would not reach a regular file at ``path`` that way, so nothing + about it can be proven. """ - from pathlib import PurePosixPath + parts = path.split("/") + real: list[str] = [] + links: list[tuple[str, str]] = [] + for index, part in enumerate(parts): + current = "/".join((*real, part)) + current_kind = kind(current) + if current_kind == "link": + if links: + return None + for _ in range(_MAX_TREE_LINK_HOPS): + text = link_text(current) + if not text or "\0" in text or os.path.isabs(text) or text.startswith(("/", "\\")): + return None + joined = posixpath.normpath(posixpath.join(posixpath.dirname(current), text)) + if joined in {".", ".."} or joined.startswith("../"): + return None + links.append((current, text)) + landed = joined.split("/") + if any(kind("/".join(landed[:depth])) != "tree" for depth in range(1, len(landed))): + return None + current, current_kind = joined, kind(joined) + if current_kind != "link": + break + else: + return None + real = current.split("/") + else: + real.append(part) + if index == len(parts) - 1: + return (tuple(links), current) if current_kind == "blob" else None + if current_kind != "tree": + return None + return None - relative = PurePosixPath(path) - if not path or relative.is_absolute() or ".." in relative.parts or "\\" in path: - return False - def entry(commit: str) -> tuple[str, str, str] | None: +class _CommitEntries: + """One commit's tree entries, each read exactly, by its own `ls-tree`.""" + + def __init__(self, workspace: Path, commit: str) -> None: + self._workspace = workspace + self._commit = commit + self._entries: dict[str, tuple[str, str, str] | None] = {} + + def entry(self, path: str) -> tuple[str, str, str] | None: + """``(mode, type, oid)`` of exactly this path; ``None`` for none, or a Git failure.""" + + if path not in self._entries: + self._entries[path] = self._read(path) + return self._entries[path] + + def _read(self, path: str) -> tuple[str, str, str] | None: + # One path per listing: a pathspec that names a directory another + # pathspec is under lists that directory's children, not the directory. result = _run_git( - workspace, - ["--literal-pathspecs", "ls-tree", "-z", "--full-tree", commit, "--", path], + self._workspace, + ["--literal-pathspecs", "ls-tree", "-z", "--full-tree", self._commit, "--", path], check=False, text=False, ) @@ -1094,27 +1161,128 @@ def entry(commit: str) -> tuple[str, str, str] | None: fields = header.split() if not separator or len(fields) != 3 or name != path.encode("utf-8"): return None - mode, kind, oid = (field.decode("ascii") for field in fields) - if kind != "blob" or mode not in {"100644", "100755"}: + mode, object_type, oid = (field.decode("ascii", errors="replace") for field in fields) + if not _GIT_OBJECT_RE.fullmatch(oid): + return None + return mode, object_type, oid + + def kind(self, path: str) -> str | None: + entry = self.entry(path) + if entry is None: + return None + mode, object_type, _oid = entry + if object_type == "tree" and mode == "040000": + return "tree" + if object_type == "blob" and mode in {"100644", "100755"}: + return "blob" + if object_type == "blob" and mode == "120000": + return "link" + return None + + def link_text(self, path: str) -> str | None: + entry = self.entry(path) + if entry is None or entry[:2] != ("120000", "blob"): + return None + data = _run_git_bounded_output( + self._workspace, ["cat-file", "blob", entry[2]], max_output_bytes=_MAX_LINK_TEXT_BYTES + ) + if data is None: + return None + try: + return data.decode("utf-8", errors="strict") + except UnicodeDecodeError: + return None + + def reader_path(self, path: str) -> _ReaderPath | None: + # A regular blob listed under the path's own name has no link above + # it, because a listing never descends through one. That is the one + # listing a path with no link took before #822, and still takes. + if self.kind(path) == "blob": + return (), path + return _reader_path(path, self.kind, self.link_text) + + +def _worktree_reader_path(workspace: Path, path: str) -> _ReaderPath | None: + """How the reader reaches ``path`` in the working tree, read without following a link.""" + + from agents_shipgate.core.trust_roots import _directory_member_kind + + kinds = {"directory": "tree", "file": "blob", "symlink": "link"} + + def kind(relative: str) -> str | None: + try: + return kinds.get(_directory_member_kind(workspace / relative)) + except OSError: + return None + + def link_text(relative: str) -> str | None: + try: + return os.readlink(workspace / relative) + except (OSError, ValueError): return None - return mode, kind, oid + + return _reader_path(path, kind, link_text) + + +def blob_path_unchanged(workspace: Path, base: str, head: str | None, path: str) -> bool: + """Whether the file the host reader opens at ``path`` is the same at ``base`` and ``head``. + + ``head=None`` means the working tree. Both sides must reach that file the + same way. With no link on the path, it is the same regular file at + ``path``, as before. Through an in-tree link (#822), such as + ``.claude/skills -> ../.agents/skills``, every link on the way must be a + link on both sides with the same text, every other component a + directory, and the file it lands on the same regular file at the same + in-tree path. Links are followed only as :func:`_reader_path` states: the + rules the reader and the archive already follow them by, so a link either + of them would not read through is never followed here. + + The answer is ``False`` whenever identity cannot be proven: a path absent + on either side, a link added, removed or given another text (even one that + lands on the same file), a link that leaves the repository, dangles, loops + or passes the hop bound, a tree or submodule, an unreadable file, or any + Git failure. The base is always read from its Git tree entries, and so is a + commit head. A working tree is read without following any link, each link + by its own text and the file by its unfiltered hash, against the base's + entries. Blob object IDs are compared rather than `git diff` output, so a + ``.gitattributes`` filter or textconv cannot make two different byte + sequences read as equal (#721). + """ + + from pathlib import PurePosixPath + + relative = PurePosixPath(path) + if ( + not path + or relative.is_absolute() + or relative.as_posix() != path + or ".." in relative.parts + or "\\" in path + or "\0" in path + ): + return False + try: + path.encode("utf-8") + except UnicodeEncodeError: + return False base_commit = commit_sha(workspace, base) - base_entry = entry(base_commit) if base_commit else None - if base_entry is None: + if base_commit is None: + return False + base_tree = _CommitEntries(workspace, base_commit) + before = base_tree.reader_path(path) + base_entry = base_tree.entry(before[1]) if before is not None else None + if before is None or base_entry is None: return False if head is not None: head_commit = commit_sha(workspace, head) - head_entry = entry(head_commit) if head_commit else None - return head_entry == base_entry - target = workspace / path - try: - parents = [workspace / Path(*relative.parts[:index]) for index in range(1, len(relative.parts))] - if any(parent.is_symlink() for parent in parents) or target.is_symlink() or not target.is_file(): + if head_commit is None: return False - except OSError: + head_tree = _CommitEntries(workspace, head_commit) + return head_tree.reader_path(path) == before and head_tree.entry(before[1]) == base_entry + if _worktree_reader_path(workspace, path) != before: return False - hashed = _run_git(workspace, ["hash-object", "--no-filters", "--", path], check=False) + hashed = _run_git(workspace, ["hash-object", "--no-filters", "--", before[1]], check=False) return hashed.returncode == 0 and hashed.stdout.strip() == base_entry[2] @@ -1306,7 +1474,7 @@ def blob_path_identities( ``head=None`` means the working tree. Every path requested has an answer: - ``True``: identical. The same regular-file blob on both sides, exactly - as :func:`blob_path_unchanged` proves it. + as :func:`blob_path_unchanged` proves it for a path no link reaches. - ``False``: differs. Between commits, two regular-file blobs with different object IDs. In the working tree, a regular file whose unfiltered hash differs from the base blob, on a path no `filter`, @@ -1322,7 +1490,8 @@ def blob_path_identities( No clean or smudge filter runs: a configured filter driver is a command line, so a converted path is answered ``None`` rather than converted. ``blob_path_unchanged`` is left as it is for ``unchanged_limits``, where - anything short of identity refuses. + anything short of identity refuses; only it follows an in-tree link + (#822), so a path through one is ``None`` here. The cost is bounded per call, not per path: one tree listing per commit, one `hash-object --stdin-paths` and, only for paths whose hashes differ, diff --git a/src/agents_shipgate/core/host_comparison.py b/src/agents_shipgate/core/host_comparison.py index 83e527c72..a39bef5e5 100644 --- a/src/agents_shipgate/core/host_comparison.py +++ b/src/agents_shipgate/core/host_comparison.py @@ -56,7 +56,10 @@ #: Blocking issue kinds an unchanged source may carry without refusing the #: comparison (#721). `unreadable` is deliberately absent: an unchanged symlink #: whose in-tree target changed would read as an unchanged limit and hide the -#: change it points to, and when a link is truly unchanged is #700's to decide. +#: change it points to, and a link the reader does not read through (#700) is +#: `unreadable` with no target read on either side. A source reached through +#: a link the reader does read through qualifies only when ``unchanged`` proves +#: the link and the file it lands on both unchanged (#822). UNCHANGED_LIMIT_ISSUE_KINDS = frozenset({"unsupported", "parse_failed"}) @@ -753,8 +756,10 @@ def compare_host_inventories( """Compare two inventories, refusing unless every limit is proven unchanged. ``unchanged`` answers whether one repository-relative source is identical - on both sides; anything short of proof is ``False``. Without it, an - incomplete inventory refuses the comparison as it always has. + on both sides, and for a source reached through an in-tree link, that the + link and the file it lands on both are (#822); anything short of proof is + ``False``. Without it, an incomplete inventory refuses the comparison as it + always has. ``identities`` is coverage's own question, asked once with every path it needs (#812): for each, ``True`` when its bytes are proven identical, diff --git a/tests/test_adapter_static_only.py b/tests/test_adapter_static_only.py index 1955aadda..a6474a1da 100644 --- a/tests/test_adapter_static_only.py +++ b/tests/test_adapter_static_only.py @@ -257,7 +257,7 @@ class AllowedException: AllowedException( relative_path="cli/verify/git.py", surface="attr_call:subprocess.Popen", - line=2686, + line=2855, snippet=( "subprocess.Popen(cmd, env=env, stderr=subprocess.PIPE, " "stdin=subprocess.PIPE if input is not None else " @@ -267,9 +267,11 @@ class AllowedException: "The shared bounded Git collector drains fixed local Git argv " "incrementally and kills them at a hard byte or wall-clock " "bound. It covers diff/name/attribute/inventory reads, " - "retained-manifest discovery, and the bounded host-path identity " + "retained-manifest discovery, the bounded host-path identity " "reads (ls-tree -l, hash-object --no-filters --stdin-paths, " - "check-attr --stdin -z, cat-file --batch) without a shell, user-code " + "check-attr --stdin -z, cat-file --batch), and one link entry's " + "text for the unchanged-limit link proof (cat-file blob, #822) " + "without a shell, user-code " "execution, or fetch. stderr is piped (not discarded) and " "drained on its own thread under a small cap so an input " "failure can be classified; the excerpt is diagnostic only." @@ -278,7 +280,7 @@ class AllowedException: AllowedException( relative_path="cli/verify/git.py", surface="attr_call:subprocess.run", - line=3086, + line=3255, snippet=( "subprocess.run(cmd, capture_output=capture_output, check=check, " "env=env, input=input, stderr=stderr, stdin=stdin, stdout=stdout, " diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 9056ebbff..2f8f464d8 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -226,7 +226,12 @@ def paths(self) -> list[Path]: # reserved coverage `scope`, and is read as incomparable by every # control route, so it adds no claim; every route to the same object, # and every refusal it must keep, is held by - # `tests/test_partial_host_comparison.py`. + # `tests/test_partial_host_comparison.py`. Naming an unchanged limit + # the reader reached through an in-tree link (#822) rests on a Git + # identity fact about the link and the file it lands on, the proof + # every unchanged limit already rested on, so it restates no engine + # answer and adds no claim; `tests/test_linked_unchanged_limits.py` + # holds diff, verify, the PR comment and `check` to it. {}, ), Surface( diff --git a/tests/test_host_comparison_coverage.py b/tests/test_host_comparison_coverage.py index 627066637..48e47068b 100644 --- a/tests/test_host_comparison_coverage.py +++ b/tests/test_host_comparison_coverage.py @@ -819,13 +819,11 @@ def _linked_skill_with_nested_link(repo: Path) -> None: _dangling_link, WIDENED, BOTH, {("NOTES.md", "unreadable", "both")}, ), - "skill-metadata-through-link": ( - _linked_skill, {"permissions": {"allow": ["Read(**)"], "deny": []}}, BOTH, - { - (".agents/skills/demo/SKILL.md", "unsupported", "both"), - (".claude/skills/demo/SKILL.md", "unsupported", "both"), - }, - ), + # The same skill through `.claude/skills -> ../.agents/skills` alone was a + # case here until #822: its link and target are unchanged, so it is now + # named in `unchanged_limits` and the rest compared + # (`tests/test_linked_unchanged_limits.py`). With a link inside the linked + # directory the reader does not read through it, and that still refuses. "skill-metadata-and-nested-link": ( _linked_skill_with_nested_link, {"permissions": {"allow": ["Read(**)"], "deny": []}}, BOTH, { diff --git a/tests/test_linked_unchanged_limits.py b/tests/test_linked_unchanged_limits.py new file mode 100644 index 000000000..05e0fbe5f --- /dev/null +++ b/tests/test_linked_unchanged_limits.py @@ -0,0 +1,427 @@ +"""#822: an unchanged limit reached through an in-tree link is named, not a veto. + +A `SKILL.md` whose `metadata` holds a non-string value is an `unsupported` +limit this entry cannot resolve. Untouched by a change, it is named as an +unchanged limit and the rest is compared (#721). Reached through an in-tree +link the reader reads through (#700) — `.claude/skills -> ../.agents/skills`, +or a per-skill link — the same untouched file refused the whole comparison +instead, hiding a removed `deny` rule beside it: the unchanged proof asked +whether the link-resolved path was a regular file in Git, and a path through a +link never is. + +The proof now follows the link as the reader does, from Git tree entries on +both sides, and holds only when the link (its entry type and text at each +link component) and the file it lands on (its blob) are both unchanged. +Anything else still refuses: a skill added or edited behind the link, the link +retargeted or its text rewritten, a link replaced by a directory or the +reverse, and every link the reader does not read through (dangling, looping, +external, escaping, past the hop bound). The metadata value is never coerced: +`internal: true` stays an `unsupported` structure (#811). + +Every route case is a real repository driven through `diff` (text and +`--json`, whose head is the working tree), `verify` (`verifier.json` and its +text, whose head is a commit), the PR comment `verify` writes, and `check`. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +from pathlib import Path + +import pytest +from typer.testing import CliRunner + +from agents_shipgate.cli.main import app +from agents_shipgate.cli.verify.git import blob_path_unchanged +from agents_shipgate.core.host_grants import host_audit_inventory + +pytestmark = pytest.mark.skipif(os.name == "nt", reason="symbolic link fixtures") + +SETTINGS = ".claude/settings.json" +BASE_SETTINGS = {"permissions": {"allow": ["Read"], "deny": ["Bash(curl *)"]}} +DENY_DROPPED = {"permissions": {"allow": ["Read"]}} +#: The issue's skill: `metadata.internal` is a boolean, which this entry's +#: bounded profile does not accept, so the structure is unresolved. +SKILL = "---\nname: review\ndescription: Review a change.\nmetadata:\n internal: true\n---\nReview the diff.\n" +#: The issue's control: the same value as a string is accepted. +STRING_SKILL = SKILL.replace("internal: true", 'internal: "true"') +SOURCE = ".claude/skills/review/SKILL.md" +DENY_REMOVED = [("claude-code .claude/settings.json", "Bash(curl *)", "removed")] +NOT_COMPARED = "Not compared: unchanged in this change and not read, so no claim is made about them:" +BOTH = ["base_inventory_incomplete", "head_inventory_incomplete"] +_GIT_ENV = { + **os.environ, + "GIT_AUTHOR_NAME": "fixture", "GIT_AUTHOR_EMAIL": "fixture@example.invalid", + "GIT_COMMITTER_NAME": "fixture", "GIT_COMMITTER_EMAIL": "fixture@example.invalid", + "GIT_CONFIG_GLOBAL": os.devnull, "GIT_CONFIG_SYSTEM": os.devnull, +} + +#: Where the skill the host reads at `SOURCE` lives: files, and links by their +#: text. `direct` is the #721 shape every other layout must match. +LAYOUTS: dict[str, tuple[dict[str, str], dict[str, str]]] = { + "direct": ({SOURCE: SKILL}, {}), + "directory-link": ( + {".agents/skills/review/SKILL.md": SKILL}, + {".claude/skills": "../.agents/skills"}, + ), + "skill-link": ({"skills/review/SKILL.md": SKILL}, {".claude/skills/review": "../../skills/review"}), + "file-link": ({"shared/review.md": SKILL}, {SOURCE: "../../../shared/review.md"}), + "link-chain": ( + {"shared/review.md": SKILL}, + {SOURCE: "../../../docs/review.md", "docs/review.md": "../shared/review.md"}, + ), +} + + +def _git(repo: Path, *args: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *args], check=True, capture_output=True, text=True, env=_GIT_ENV + ).stdout.strip() + + +def _write(repo: Path, name: str, value: object) -> None: + path = repo / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(value if isinstance(value, str) else json.dumps(value), encoding="utf-8") + + +def _remove(repo: Path, name: str) -> None: + """Remove what is at ``name`` itself, never following a link there.""" + + path = repo / name + if path.is_symlink() or path.is_file(): + path.unlink() + elif path.is_dir(): + shutil.rmtree(path) + + +def _link(repo: Path, name: str, target: str) -> None: + _remove(repo, name) + path = repo / name + path.parent.mkdir(parents=True, exist_ok=True) + os.symlink(target, path) + + +def _commit(repo: Path, message: str) -> None: + _git(repo, "add", "-A", "-f") + _git(repo, "commit", "-q", "--allow-empty", "-m", message) + + +def _repository( + tmp_path: Path, + files: dict[str, str], + links: dict[str, str], + *, + head_removed: tuple[str, ...] = (), + head_files: dict[str, str] | None = None, + head_links: dict[str, str] | None = None, + head_settings: object = DENY_DROPPED, +) -> Path: + """`main` holds the deny rule beside ``files`` and ``links``; branch `change` drops it. + + On `change`, ``head_removed`` paths are removed first, then ``head_files`` + and ``head_links`` are written, and all of it is committed. A link text + ``{outside}`` names a directory beside the repository that holds the same + skill, so a link that leaves the repository lands on real content. + """ + + outside = tmp_path / "outside" + _write(outside, "skills/review/SKILL.md", SKILL) + repo = tmp_path / "repo" + repo.mkdir(parents=True) + _git(repo, "init", "-q", "-b", "main") + _write(repo, SETTINGS, BASE_SETTINGS) + _write(repo, "README.md", "# demo\n") + for name, value in files.items(): + _write(repo, name, value) + for name, target in links.items(): + _link(repo, name, target.format(outside=outside)) + _commit(repo, "base") + _git(repo, "checkout", "-q", "-b", "change") + _write(repo, SETTINGS, head_settings) + for name in head_removed: + _remove(repo, name) + for name, value in (head_files or {}).items(): + _write(repo, name, value) + for name, target in (head_links or {}).items(): + _link(repo, name, target.format(outside=outside)) + _commit(repo, "change") + return repo + + +def _layout(tmp_path: Path, name: str, **head) -> Path: + files, links = LAYOUTS[name] + return _repository(tmp_path, files, links, **head) + + +def _invoke(args: list[str]) -> str: + result = CliRunner().invoke(app, args, env={"NO_COLOR": "1", "COLUMNS": "200"}) + assert result.exit_code == 0, result.output + return result.output + + +def _diff(repo: Path) -> dict: + return json.loads(_invoke(["diff", "--workspace", str(repo), "--base", "main", "--json"])) + + +def _all_routes(repo: Path, tmp_path: Path) -> tuple[str, dict, str, str]: + """diff text and JSON, and verify's comparison, text and PR comment, agreeing on one comparison.""" + + command = ["diff", "--workspace", str(repo), "--base", "main"] + text, payload = _invoke(command), json.loads(_invoke([*command, "--json"])) + out = tmp_path / "out" + verify_text = _invoke([ + "verify", "--workspace", str(repo), "--config", "shipgate.yaml", "--ci-mode", "advisory", + "--out", str(out), "--format", "text", "--base", "main", + "--head", _git(repo, "rev-parse", "HEAD"), + ]) + comparison = json.loads((out / "verifier.json").read_text(encoding="utf-8"))["host_comparison"] + for key in ("comparison_status", "incomparable_reasons", "rows", "unchanged_limits", "coverage"): + assert comparison[key] == payload[key], key + if payload["review"] is not None: + # The reproduction names the working tree for `diff` and the commit for + # `verify`; the presented changes are the same. + assert comparison["review"]["changes"] == payload["review"]["changes"] + comment = (out / "pr-comment.md").read_text(encoding="utf-8") + return text, payload, verify_text, comment + + +def _rows(payload: dict) -> list[tuple[str, str, str]]: + return [ + (row["subject"], row["before"] if row["direction"] == "removed" else row["after"], row["direction"]) + for row in payload["rows"] + ] + + +def _limits(payload: dict) -> set[tuple[str, str, str]]: + return {(limit["host"], limit["source"], limit["limit"]) for limit in payload["unchanged_limits"]} + + +# --- the issue's reproduction: every layout matches `direct` --------------------- + + +@pytest.mark.parametrize("layout", list(LAYOUTS)) +def test_an_unchanged_limit_through_an_in_tree_link_is_named_like_a_direct_one( + tmp_path: Path, layout: str +) -> None: + repo = _layout(tmp_path, layout) + + text, payload, verify_text, comment = _all_routes(repo, tmp_path) + + assert payload["comparison_status"] == "comparable" + assert payload["incomparable_reasons"] == [] + assert _rows(payload) == DENY_REMOVED + assert ("claude-code", SOURCE, "unsupported") in _limits(payload) + # Only the skill is a limit; a file the link lands on that a host also + # reads directly (`.agents/skills/...`) is named on its own path. + assert {limit["limit"] for limit in payload["unchanged_limits"]} == {"unsupported"} + assert all( + "frontmatter_invalid_structure" in limit["detail"] for limit in payload["unchanged_limits"] + ) + assert NOT_COMPARED in text and f" claude-code {SOURCE} — unsupported" in text + assert "Not compared: unchanged in this change" in verify_text + assert NOT_COMPARED in comment and f"` {SOURCE} `" in comment + assert "Bash(curl *)" in text and "Bash(curl *)" in comment + + +@pytest.mark.parametrize("layout", list(LAYOUTS)) +def test_the_proof_holds_between_commits_and_against_the_working_tree( + tmp_path: Path, layout: str +) -> None: + repo = _layout(tmp_path, layout) + + assert blob_path_unchanged(repo, "main", "HEAD", SOURCE) + assert blob_path_unchanged(repo, "main", None, SOURCE) + + +def test_a_metadata_value_is_never_coerced(tmp_path: Path) -> None: + """#811: a boolean stays an unsupported structure through a link, and the + string the issue used as its control reads as supported, with no limit.""" + + boolean = _layout(tmp_path / "boolean", "directory-link") + string = _repository( + tmp_path / "string", + {".agents/skills/review/SKILL.md": STRING_SKILL}, + {".claude/skills": "../.agents/skills"}, + ) + + assert ("claude-code", SOURCE, "unsupported") in _limits(_diff(boolean)) + controlled = _diff(string) + assert controlled["comparison_status"] == "comparable" + assert controlled["unchanged_limits"] == [] + assert _rows(controlled) == DENY_REMOVED + + +@pytest.mark.parametrize("layout", list(LAYOUTS)) +def test_check_refuses_exactly_as_it_does_for_a_direct_limit(tmp_path: Path, layout: str) -> None: + """`check`'s boundary result cannot name a limit (#721), so it refuses its + comparison with the reason a direct limit gives; its decision comes from its + own routing and does not move.""" + + repo = _layout(tmp_path, layout) + + payload = json.loads(_invoke([ + "check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", + "--format", "agent-boundary-json", + ])) + + assert payload["comparison_status"] == "incomparable" + assert payload["incomparable_reasons"] == ["unchanged_limits_not_representable"] + assert payload["rows"] == [] + assert payload["decision"] == "require_review" + assert "HOST-PERMISSION-DENY-REMOVED" in [item["id"] for item in payload["violations"]] + + +# --- negative controls: what the change touched still refuses --------------------- + +EDITED = SKILL.replace("Review the diff.", "Review it again.") +#: ``(base files, base links, what the head changes, reasons)``. Each head also +#: drops the deny rule, so a comparison that compared past the limit would +#: publish that row. +CHANGED: dict[str, tuple[dict[str, str], dict[str, str], dict[str, object], list[str]]] = { + "skill-added-behind-the-link": ( + {".agents/skills/other/SKILL.md": "---\nname: other\ndescription: d\n---\nbody\n"}, + {".claude/skills": "../.agents/skills"}, + {"head_files": {".agents/skills/review/SKILL.md": SKILL}}, + ["head_inventory_incomplete"], + ), + "skill-edited-behind-the-link": ( + *LAYOUTS["directory-link"], + {"head_files": {".agents/skills/review/SKILL.md": EDITED}}, + BOTH, + ), + # The same bytes behind another link text: the link changed, so the proof + # cannot say what the reader opens is what it opened before. + "link-retargeted-to-an-identical-copy": ( + {"skills/review/SKILL.md": SKILL, "copy/review/SKILL.md": SKILL}, + {".claude/skills/review": "../../skills/review"}, + {"head_links": {".claude/skills/review": "../../copy/review"}}, + BOTH, + ), + "link-text-rewritten-to-land-on-the-same-file": ( + *LAYOUTS["skill-link"], + {"head_links": {".claude/skills/review": "../../skills/./review"}}, + BOTH, + ), + "link-replaced-by-a-directory-with-the-same-bytes": ( + *LAYOUTS["skill-link"], + {"head_removed": (".claude/skills/review",), "head_files": {SOURCE: SKILL}}, + BOTH, + ), + "directory-replaced-by-a-link-to-the-same-bytes": ( + {SOURCE: SKILL, "skills/review/SKILL.md": SKILL}, + {}, + {"head_links": {".claude/skills/review": "../../skills/review"}}, + BOTH, + ), + "file-link-target-edited": ( + *LAYOUTS["file-link"], + {"head_files": {"shared/review.md": EDITED}}, + BOTH, + ), + "second-hop-retargeted": ( + {**LAYOUTS["link-chain"][0], "shared/copy.md": SKILL}, + LAYOUTS["link-chain"][1], + {"head_links": {"docs/review.md": "../shared/copy.md"}}, + BOTH, + ), +} + + +@pytest.mark.parametrize("case", list(CHANGED)) +def test_a_change_to_the_link_or_what_it_lands_on_still_refuses(tmp_path: Path, case: str) -> None: + files, links, head, reasons = CHANGED[case] + repo = _repository(tmp_path, files, links, **head) + + _text, payload, _verify_text, _comment = _all_routes(repo, tmp_path) + + assert payload["comparison_status"] == "incomparable" + assert payload["incomparable_reasons"] == reasons + assert payload["rows"] == [] and payload["unchanged_limits"] == [] + assert not blob_path_unchanged(repo, "main", "HEAD", SOURCE) + assert not blob_path_unchanged(repo, "main", None, SOURCE) + + +def test_a_working_tree_change_to_the_link_or_its_target_is_not_unchanged(tmp_path: Path) -> None: + """`diff` reads the working tree: an uncommitted retarget or edit refuses as a committed one does.""" + + repo = _layout(tmp_path, "skill-link") + assert blob_path_unchanged(repo, "main", None, SOURCE) + + _write(repo, "copy/review/SKILL.md", SKILL) + _link(repo, ".claude/skills/review", "../../copy/review") + assert not blob_path_unchanged(repo, "main", None, SOURCE) + assert _diff(repo)["comparison_status"] == "incomparable" + + _link(repo, ".claude/skills/review", "../../skills/review") + assert blob_path_unchanged(repo, "main", None, SOURCE) + _write(repo, "skills/review/SKILL.md", EDITED) + assert not blob_path_unchanged(repo, "main", None, SOURCE) + assert _diff(repo)["comparison_status"] == "incomparable" + + +# --- the proof follows exactly the links the reader reads through ------------------ + + +def _chain(length: int) -> tuple[dict[str, str], dict[str, str]]: + """``SOURCE`` reaching the skill through ``length`` file links in a row.""" + + links = {f"hop{index}.md": f"hop{index - 1}.md" for index in range(1, length - 1)} + links["hop0.md"] = "shared/review.md" + links[SOURCE] = f"../../../hop{length - 2}.md" + return {"shared/review.md": SKILL}, links + + +#: Shapes the reader does or does not read `SOURCE` through, unchanged between +#: the commits. ``True`` where the reader reads the skill at `SOURCE`. +READER_SHAPES: dict[str, tuple[dict[str, str], dict[str, str], bool]] = { + **{name: (*LAYOUTS[name], True) for name in LAYOUTS}, + "eight-links-in-a-row": (*_chain(8), True), + "nine-links-in-a-row": (*_chain(9), False), + "dangling": ({}, {".claude/skills": "../.agents/skills"}, False), + "loop": ({}, {".claude/skills": "../.agents/skills", ".agents/skills": "../.claude/skills"}, False), + "escaping": ({}, {".claude/skills": "../../outside/skills"}, False), + "external": ({}, {".claude/skills": "{outside}/skills"}, False), + # A link as an intermediate component of where the first link lands. + "link-above-the-target": ( + {".agents/skills/review/SKILL.md": SKILL}, + {".claude/skills": "../alias/skills", "alias": ".agents"}, + False, + ), + # A link inside a linked directory: the reader does not read through it. + "link-inside-the-linked-directory": ( + {"skills/review/SKILL.md": SKILL}, + {".claude/skills": "../.agents/skills", ".agents/skills/review": "../../skills/review"}, + False, + ), +} + + +@pytest.mark.parametrize("shape", list(READER_SHAPES)) +def test_the_proof_holds_exactly_where_the_reader_reads_through(tmp_path: Path, shape: str) -> None: + files, links, reads_through = READER_SHAPES[shape] + repo = _repository(tmp_path, files, links, head_settings=BASE_SETTINGS) + + issues = host_audit_inventory(repo)["issues"] + read = any(issue["source"] == SOURCE and issue["kind"] == "unsupported" for issue in issues) + + assert read is reads_through, [(issue["kind"], issue["source"]) for issue in issues] + assert blob_path_unchanged(repo, "main", "HEAD", SOURCE) is reads_through + assert blob_path_unchanged(repo, "main", None, SOURCE) is reads_through + + +@pytest.mark.parametrize("shape", [name for name, (*_, reads) in READER_SHAPES.items() if not reads]) +def test_a_link_the_reader_does_not_read_through_still_refuses(tmp_path: Path, shape: str) -> None: + """Unchanged, and still refused: the reader read nothing behind it on either side.""" + + files, links, _reads_through = READER_SHAPES[shape] + repo = _repository(tmp_path, files, links) + + payload = _diff(repo) + + assert payload["comparison_status"] == "incomparable" + assert payload["rows"] == [] and payload["unchanged_limits"] == [] + assert "unreadable" in {item["limit"] for item in payload["coverage"]["items"]} From 8c6b816a0e1783e6e67137502e317a15a08e026e Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 08:38:00 -0700 Subject: [PATCH 10/14] Address review cycle 1 on linked unchanged limits (#822) The STABILITY migration note said that where the other routes now compare past a limit reached through an in-tree link, `check` refuses with `unchanged_limits_not_representable` and its rows stay empty. That holds for an instruction file's limit, which `check` cannot leave out, but not for a plugin-reference limit: `_without_shared_plugin_reference_limits` asks the same unchanged proof, so a `parse_failed` `plugin.json` that is a file link to an unchanged target is now left out exactly as one at its own path has been since #714. `check` then compares and publishes the removed `deny` row in its boundary result and its control envelope's `capability_rows`, where it refused with `base_inventory_incomplete` / `head_inventory_incomplete` and no row, and `diff`, `verify` and the PR comment, which withheld `plugins/demo` as `partial` (#808), are `comparable` with the limit in `unchanged_limits`. `check`'s decision, violations and control state, and `verify`'s control state and next action, do not move. The note now splits `check` by whether it may leave the limit out, adds the `partial` to `comparable` move, and says "every row's value" where it said "every row". The CHANGELOG entry mirrors it, the distribution-surfaces row and `docs/host-boundary-support.md` no longer say any change behind the link refuses the comparison (an edited plugin manifest keeps it `partial`), and `tests/test_linked_unchanged_limits.py` pins the plugin manifest at its own path and behind a file link, on every route, with the edited-target control. The `blob_path_unchanged` and `_reader_path` docstrings now state the proof as necessary, not sufficient: it follows the links on the way to the path, not every condition the reader puts on reading a whole linked directory, and a side whose reader does not read the path carries no limit there. A test holds a head that adds a link inside the linked directory to a refusal on every route. The two pinned call-site lines in `cli/verify/git.py` move with the docstrings. --- CHANGELOG.md | 39 ++++-- STABILITY.md | 62 +++++++-- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 6 +- src/agents_shipgate/cli/verify/git.py | 35 +++-- tests/test_adapter_static_only.py | 4 +- tests/test_distribution_surface_parity.py | 3 +- tests/test_linked_unchanged_limits.py | 152 +++++++++++++++++++++- 8 files changed, 254 insertions(+), 49 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 82fc687e1..882f9dac4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -134,13 +134,13 @@ - No schema, contract, member, error kind, refusal code or exit code moves. See the `STABILITY.md` migration note. (#807) -- An unchanged instruction file reached through an in-tree link no longer - refuses the whole host comparison. A `SKILL.md` whose `metadata` holds a - non-string value, which this entry cannot resolve, is named as an unchanged - limit when the change leaves it alone (#721). Read through - `.claude/skills -> ../.agents/skills`, a per-skill link or a file link, the - same untouched file made `diff`, `verify` and the manifest-free PR comment - print `Cannot compare against HEAD~1: base_inventory_incomplete; +- An unchanged instruction file or plugin manifest reached through an in-tree + link no longer refuses the whole host comparison. A `SKILL.md` whose + `metadata` holds a non-string value, which this entry cannot resolve, is + named as an unchanged limit when the change leaves it alone (#721). Read + through `.claude/skills -> ../.agents/skills`, a per-skill link or a file + link, the same untouched file made `diff`, `verify` and the manifest-free PR + comment print `Cannot compare against HEAD~1: base_inventory_incomplete; head_inventory_incomplete` and no row, hiding a removed `deny` rule beside it. - **The cause.** The unchanged proof asked whether the path the link is read under, `.claude/skills/review/SKILL.md`, was one regular file in Git, and a @@ -163,11 +163,26 @@ directory), which stays `unreadable` and is never an unchanged limit. The metadata value is not coerced: `internal: true` stays an `unsupported` structure (#811). - - **`check`.** Its boundary result still cannot name a limit, so it now - refuses such a comparison with `unchanged_limits_not_representable`, the - reason it gives for a limit at its own path, instead of - `base_inventory_incomplete` / `head_inventory_incomplete`. Rows stay empty - and its decision and violations do not move. + - **A plugin manifest behind a link.** The same proof covers a + plugin-reference limit, such as a `parse_failed` + `plugins/demo/.claude-plugin/plugin.json` that is a file link to + `../../../vendor/plugin.json`. Unchanged on both sides, it is now named in + `unchanged_limits` like one at its own path, so a comparison that was + `partial` only because of it (#808, `Not compared: plugins/demo`) is + `comparable` on `diff`, `verify` and the PR comment, with the directory + compared; `1.1.0` refused it. Edited behind the link, it stays `partial`. + - **`check`.** Its boundary result still cannot name a limit, so it does + what it does for the same limit at its own path. An instruction file's + limit, which it cannot leave out, now refuses its comparison with + `unchanged_limits_not_representable` instead of + `base_inventory_incomplete` / `head_inventory_incomplete`, and its rows + stay empty. A plugin-reference limit both sides share on an unchanged + source it leaves out, as it has since #714, so behind a link it now + compares and publishes the rows it finds where it refused with no row: the + plugin manifest above gives `comparable` with the removed `deny` row, in + the boundary result and in the control envelope's `capability_rows`. + Edited behind the link, the limit is kept and `check` refuses as before. + Its decision, violations and control state do not move. - No schema, contract, member, reason code or check id is added, and host-grants stays `0.6`. See the `STABILITY.md` migration note. (#822) diff --git a/STABILITY.md b/STABILITY.md index c757f7a8c..c383471bc 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -602,7 +602,8 @@ No schema, member, reason code, check id or `minimum_control_contract_version` moves, and host-grants stays `0.6`. This change adds no version of its own: capability diff, verifier and the runtime contract are what the rest of the unreleased tree carries. What moves is which comparisons name an unchanged -limit instead of refusing, and so which are `comparable`. +limit instead of refusing, and so which are `comparable`, and which `check` +compares. Since `1.0.0`, a per-source `unsupported` or `parse_failed` limit that both sides share on an untouched file is named in `unchanged_limits` and the rest @@ -662,23 +663,56 @@ accept, and a change that only drops `deny: Bash(curl *)` from `verify` text and the PR comment; the unchanged limits a `partial` comparison may carry beside its withheld directories ([#808](#partial-host-comparison-808)); and the shared plugin-reference - limits `check` leaves out of its comparison (#714). `check`'s boundary result - still cannot name a limit, so where the other routes now compare past one it - refuses with `unchanged_limits_not_representable`, the reason it already - gives for a limit at its own path, instead of `base_inventory_incomplete` / - `head_inventory_incomplete`. Its rows stay empty, and its decision, - violations and control do not move. + limits `check` leaves out of its comparison (#714). +- **`check`.** Its boundary result still cannot name a limit, so what it does + depends on whether it may leave the limit out, exactly as for a limit at its + own path: + - **A limit it cannot leave out**, such as the skill above: where the other + routes now compare past one it refuses with + `unchanged_limits_not_representable`, the reason it already gives for a + limit at its own path, instead of `base_inventory_incomplete` / + `head_inventory_incomplete`. Its rows stay empty. + - **A plugin-reference limit both sides share**, such as a `parse_failed` + `plugins/demo/.claude-plugin/plugin.json` that is a file link to + `../../../vendor/plugin.json`: `check` leaves it out of its comparison + once the proof holds, as it has left out one at its own path since #714. + So it now compares, and publishes the rows it finds where it refused with + `base_inventory_incomplete` / `head_inventory_incomplete` and no row. With + the change above, the boundary result and the control envelope's + `capability_rows` are `comparable` with the removed `deny` row, the same + row `check` publishes when `plugin.json` sits at its own path. Edited + behind the link, the limit is kept and `check` refuses as before. + + Either way, `check`'s decision, violations and control state do not move: + the change above is still `require_review` with + `HOST-PERMISSION-DENY-REMOVED`. +- **`partial` becomes `comparable`.** On this unreleased tree, a comparison + refused only by such a plugin-reference limit is `partial` (#808): `diff`, + `verify` and the PR comment withhold its plugin directory, as + `Not compared: plugins/demo`, and publish the rows outside it. Once the + proof holds, the limit is an unchanged one, so a comparison that was + `partial` only because of it is `comparable` instead: the directory is + compared, and the limit is named in `unchanged_limits` and the + `Not compared: unchanged in this change and not read` list, with no + `scope`, exactly as for a `plugin.json` at its own path. `1.1.0`, which has + no `partial`, refused it outright. `verify`'s control state and next action + do not move; its `reason` is the comparable result's sentence. Edited behind + the link, it is still `partial`. - **Not changed.** The inventory, its issues and `resolved_through`, saved - host-grants baselines, drift, `audit --host`, every row and every control - state and route. The coverage block asks its own identity question and asks - none for a file read through a link, so a zero-row file read that way is - still `unchanged_not_proven`. The host-config and cold-start benchmark - replays reproduce their run-of-record scores. + host-grants baselines, drift, `audit --host`, every row's value and every + control state and route. The coverage block asks its own identity question + and asks none for a file read through a link, so a zero-row file read that + way is still `unchanged_not_proven`. The host-config and cold-start + benchmark replays reproduce their run-of-record scores. **Compatibility.** No field changes shape. A consumer that switches on `comparison_status` reads `comparable` where it read `incomparable` for these -layouts, with the limit in `unchanged_limits` exactly as a limit at its own -path has been published since `1.0.0`. +layouts, or `partial` on this unreleased tree for a plugin-reference limit, +with the limit in `unchanged_limits` exactly as a limit at its own path has +been published since `1.0.0`. Where such a limit was its only refusal, +`check`'s boundary result reads `unchanged_limits_not_representable` for a +limit it cannot leave out, and for a shared plugin-reference limit it +compares, with the rows it finds, as for the same limit at its own path. diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index c7ac50839..11dfb2e61 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A surface the reader reaches through an in-tree link it reads through qualifies only when that link, a link with the same text at each link on the way, and the file it lands on, the same blob at the same path, are both unchanged, read from the base's Git tree entries against a commit's or, without following any link, the working tree's; any change to either refuses as before (#822, `tests/test_linked_unchanged_limits.py`). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A surface the reader reaches through an in-tree link it reads through qualifies only when that link, a link with the same text at each link on the way, and the file it lands on, the same blob at the same path, are both unchanged, read from the base's Git tree entries against a commit's or, without following any link, the working tree's; any change to either is treated as before. The same proof decides which shared plugin-reference limits `check` leaves out, so behind such a link `check` compares, and publishes the rows it finds, exactly as for a limit at its own path, and a comparison `partial` only because of such a limit is `comparable` with it in `unchanged_limits` (#822, `tests/test_linked_unchanged_limits.py`). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 7787e607f..51cb1f544 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -30,9 +30,9 @@ in-tree file. A limit on a file read that way, such as a skill whose structure could not be established behind `.claude/skills -> ../.agents/skills`, is named as unchanged, like one at its own path, only when the link, with the same text at each link on the way, and the file it lands on are both -unchanged between the compared sides (#822); any change to either still -refuses the comparison. A dangling link, a link that leaves the repository, or a -directory link outside the boundary paths refuses the whole comparison, even +unchanged between the compared sides (#822); any change to either keeps it a +blocking limit, as before. A dangling link, a link that leaves the repository, +or a directory link outside the boundary paths refuses the whole comparison, even when no host file sits behind it (#688). That refusal is deliberate (#659), and it caused every widening the 1.0 host-config measurement missed. Path classification is case-insensitive so protected files cannot evade review diff --git a/src/agents_shipgate/cli/verify/git.py b/src/agents_shipgate/cli/verify/git.py index 8ab913c5d..5cb88e50c 100644 --- a/src/agents_shipgate/cli/verify/git.py +++ b/src/agents_shipgate/cli/verify/git.py @@ -1081,16 +1081,23 @@ def _reader_path( ``"tree"``, ``"blob"`` (a regular file) or ``"link"``, and ``None`` for anything else or nothing. ``link_text`` is a link entry's own text. - A link is followed only as the reader follows one (#700, the reader's - ``_resolve_in_tree_link``) and the archive would also resolve it (#711, - :func:`_resolve_tree_link`): its text is relative and lands inside the - tree after normalization, every directory above where it lands is a - directory and not a link, and the chain ends at a directory or regular file - within :data:`_MAX_TREE_LINK_HOPS` links. A path holds links at one + A link is followed only by rules the reader also applies (#700, the + reader's ``_resolve_in_tree_link``) and the archive would also resolve it + by (#711, :func:`_resolve_tree_link`): its text is relative and lands + inside the tree after normalization, every directory above where it lands + is a directory and not a link, and the chain ends at a directory or regular + file within :data:`_MAX_TREE_LINK_HOPS` links. A path holds links at one component only, because a directory the reader reads through holds no link, and every other component must be a directory. ``None`` means the reader would not reach a regular file at ``path`` that way, so nothing about it can be proven. + + These are the rules for the links on the way to ``path``, not every + condition the reader puts on reading a whole linked directory (the + reader's ``_reads_through_directory_link``: no link anywhere beneath the + target, no skipped name on the way, a boundary location). So an answer + other than ``None`` is necessary for the reader to open the file there, + not sufficient: see :func:`blob_path_unchanged`. """ parts = path.split("/") @@ -1233,9 +1240,19 @@ def blob_path_unchanged(workspace: Path, base: str, head: str | None, path: str) ``.claude/skills -> ../.agents/skills``, every link on the way must be a link on both sides with the same text, every other component a directory, and the file it lands on the same regular file at the same - in-tree path. Links are followed only as :func:`_reader_path` states: the - rules the reader and the archive already follow them by, so a link either - of them would not read through is never followed here. + in-tree path. Links are followed only by the rules :func:`_reader_path` + states, which the reader and the archive also apply to the links on the + way. + + ``True`` is necessary for naming a limit on ``path`` as unchanged, not + sufficient: this does not model every condition the reader puts on reading + a whole linked directory, so it can answer ``True`` where one side's reader + does not read ``path`` at all, such as a head that adds a link inside + ``.agents/skills``. Its callers ask only about a limit both sides carry on + ``path``, and a side whose reader does not read ``path`` carries none + there: in that example it carries an ``unreadable`` limit on the link + instead, which is never named as unchanged, so the comparison still + refuses whatever this answers. The answer is ``False`` whenever identity cannot be proven: a path absent on either side, a link added, removed or given another text (even one that diff --git a/tests/test_adapter_static_only.py b/tests/test_adapter_static_only.py index a6474a1da..43626c14a 100644 --- a/tests/test_adapter_static_only.py +++ b/tests/test_adapter_static_only.py @@ -257,7 +257,7 @@ class AllowedException: AllowedException( relative_path="cli/verify/git.py", surface="attr_call:subprocess.Popen", - line=2855, + line=2872, snippet=( "subprocess.Popen(cmd, env=env, stderr=subprocess.PIPE, " "stdin=subprocess.PIPE if input is not None else " @@ -280,7 +280,7 @@ class AllowedException: AllowedException( relative_path="cli/verify/git.py", surface="attr_call:subprocess.run", - line=3255, + line=3272, snippet=( "subprocess.run(cmd, capture_output=capture_output, check=check, " "env=env, input=input, stderr=stderr, stdin=stdin, stdout=stdout, " diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 2f8f464d8..d725a5e05 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -231,7 +231,8 @@ def paths(self) -> list[Path]: # identity fact about the link and the file it lands on, the proof # every unchanged limit already rested on, so it restates no engine # answer and adds no claim; `tests/test_linked_unchanged_limits.py` - # holds diff, verify, the PR comment and `check` to it. + # holds diff, verify, the PR comment and `check` to it, including the + # shared plugin-reference limits `check` leaves out by the same proof. {}, ), Surface( diff --git a/tests/test_linked_unchanged_limits.py b/tests/test_linked_unchanged_limits.py index 05e0fbe5f..e24856d90 100644 --- a/tests/test_linked_unchanged_limits.py +++ b/tests/test_linked_unchanged_limits.py @@ -18,6 +18,13 @@ external, escaping, past the hop bound). The metadata value is never coerced: `internal: true` stays an `unsupported` structure (#811). +The same proof decides which shared plugin-reference limits `check` leaves +out of its comparison (#714). A `parse_failed` `plugin.json` behind a file +link was kept, so `check` refused with no row, and `diff` and `verify` +withheld its plugin directory as `partial` (#808). Now `check` compares and +publishes its rows, and the others are `comparable` with the limit named, +exactly as for a `plugin.json` at its own path. + Every route case is a real repository driven through `diff` (text and `--json`, whose head is the working tree), `verify` (`verifier.json` and its text, whose head is a commit), the PR comment `verify` writes, and `check`. @@ -257,16 +264,14 @@ def test_a_metadata_value_is_never_coerced(tmp_path: Path) -> None: @pytest.mark.parametrize("layout", list(LAYOUTS)) def test_check_refuses_exactly_as_it_does_for_a_direct_limit(tmp_path: Path, layout: str) -> None: - """`check`'s boundary result cannot name a limit (#721), so it refuses its - comparison with the reason a direct limit gives; its decision comes from its - own routing and does not move.""" + """`check`'s boundary result cannot name a limit (#721), and an instruction + file's limit is not one it may leave out, so it refuses its comparison with + the reason a direct limit gives; its decision comes from its own routing and + does not move.""" repo = _layout(tmp_path, layout) - payload = json.loads(_invoke([ - "check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", - "--format", "agent-boundary-json", - ])) + payload = _check(repo, "agent-boundary-json") assert payload["comparison_status"] == "incomparable" assert payload["incomparable_reasons"] == ["unchanged_limits_not_representable"] @@ -275,6 +280,111 @@ def test_check_refuses_exactly_as_it_does_for_a_direct_limit(tmp_path: Path, lay assert "HOST-PERMISSION-DENY-REMOVED" in [item["id"] for item in payload["violations"]] +def _check(repo: Path, format_: str) -> dict: + return json.loads(_invoke([ + "check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", format_, + ])) + + +# --- a plugin-reference limit behind a link: as at its own path -------------------- + +PLUGIN_MANIFEST = "plugins/demo/.claude-plugin/plugin.json" +PLUGIN_HOOKS = json.dumps( + {"hooks": {"PreToolUse": [{"matcher": "Bash", "hooks": [{"type": "command", "command": "echo hi"}]}]}} +) +#: An unparseable plugin manifest beside a plugin hook file: at its own path, +#: where `check` has left the shared limit out since #714, or behind a file +#: link the reader reads through. +PLUGIN_LAYOUTS: dict[str, tuple[dict[str, str], dict[str, str]]] = { + "direct": ({PLUGIN_MANIFEST: "{not json", "plugins/demo/cfg/hooks.json": PLUGIN_HOOKS}, {}), + "file-link": ( + {"vendor/plugin.json": "{not json", "plugins/demo/cfg/hooks.json": PLUGIN_HOOKS}, + {PLUGIN_MANIFEST: "../../../vendor/plugin.json"}, + ), +} +#: `check` redacts rule arguments. +CHECK_DENY_REMOVED = [("claude-code .claude/settings.json", "Bash()", "removed")] +#: #808's line for a withheld plugin directory, as text and as the PR comment +#: spells it. +PLUGIN_WITHHELD = ("Not compared: plugins/demo, a plugin directory", "Not compared: ` plugins/demo `") + + +@pytest.mark.parametrize("layout", list(PLUGIN_LAYOUTS)) +def test_check_leaves_out_a_shared_plugin_reference_limit_behind_a_link_as_at_its_own_path( + tmp_path: Path, layout: str +) -> None: + """`check` leaves out a plugin-reference limit both sides share on an + unchanged source (#714), which the proof now establishes behind a link. So + it compares and publishes the row it finds, where it refused with + `base_inventory_incomplete` / `head_inventory_incomplete` and no row; its + decision, violations and control state are what they were.""" + + repo = _repository(tmp_path, *PLUGIN_LAYOUTS[layout]) + + boundary = _check(repo, "agent-boundary-json") + control = _check(repo, "agent-control-json") + + assert blob_path_unchanged(repo, "main", "HEAD", PLUGIN_MANIFEST) + assert boundary["comparison_status"] == "comparable" + assert boundary["incomparable_reasons"] == [] + assert _rows(boundary) == CHECK_DENY_REMOVED + assert boundary["decision"] == "require_review" + assert "HOST-PERMISSION-DENY-REMOVED" in [item["id"] for item in boundary["violations"]] + rows = control["capability_rows"] + assert rows["comparison_status"] == "comparable" + assert rows["incomparable_reasons"] == [] + assert _rows(rows) == CHECK_DENY_REMOVED + assert control["control_state"] == "review_publishable" + + +@pytest.mark.parametrize("layout", list(PLUGIN_LAYOUTS)) +def test_a_shared_plugin_reference_limit_behind_a_link_is_named_not_withheld( + tmp_path: Path, layout: str +) -> None: + """Behind a link the limit could not be proven unchanged, so #808 withheld + its plugin directory and the comparison was `partial`. Now it is named in + `unchanged_limits` and the directory compared, as at its own path.""" + + repo = _repository(tmp_path, *PLUGIN_LAYOUTS[layout]) + + text, payload, verify_text, comment = _all_routes(repo, tmp_path) + + assert payload["comparison_status"] == "comparable" + assert payload["incomparable_reasons"] == [] + assert _rows(payload) == DENY_REMOVED + assert _limits(payload) == {("claude-code", PLUGIN_MANIFEST, "parse_failed")} + assert all(item["scope"] is None for item in payload["coverage"]["items"]) + assert NOT_COMPARED in text and f" claude-code {PLUGIN_MANIFEST} — parse_failed" in text + assert NOT_COMPARED in comment + for output in (text, verify_text, comment): + assert not any(line in output for line in PLUGIN_WITHHELD) + assert "Partial comparison" not in output and "comparison partial" not in output + + +def test_a_plugin_manifest_edited_behind_the_link_is_still_withheld(tmp_path: Path) -> None: + """The negative control: the file the link lands on changed, so the limit is + kept. `diff` and `verify` withhold the directory as before (#808), and + `check` refuses with no row as before.""" + + files, links = PLUGIN_LAYOUTS["file-link"] + repo = _repository(tmp_path, files, links, head_files={"vendor/plugin.json": "{still not json"}) + + text, payload, _verify_text, comment = _all_routes(repo, tmp_path) + boundary = _check(repo, "agent-boundary-json") + + assert not blob_path_unchanged(repo, "main", "HEAD", PLUGIN_MANIFEST) + assert payload["comparison_status"] == "partial" + assert payload["incomparable_reasons"] == BOTH + assert _rows(payload) == DENY_REMOVED + assert payload["unchanged_limits"] == [] + assert PLUGIN_WITHHELD[0] in text and PLUGIN_WITHHELD[1] in comment + assert boundary["comparison_status"] == "incomparable" + assert boundary["incomparable_reasons"] == BOTH + assert boundary["rows"] == [] + assert boundary["decision"] == "require_review" + assert "HOST-PERMISSION-DENY-REMOVED" in [item["id"] for item in boundary["violations"]] + + # --- negative controls: what the change touched still refuses --------------------- EDITED = SKILL.replace("Review the diff.", "Review it again.") @@ -425,3 +535,31 @@ def test_a_link_the_reader_does_not_read_through_still_refuses(tmp_path: Path, s assert payload["comparison_status"] == "incomparable" assert payload["rows"] == [] and payload["unchanged_limits"] == [] assert "unreadable" in {item["limit"] for item in payload["coverage"]["items"]} + + +def test_a_link_added_inside_the_linked_directory_still_refuses(tmp_path: Path) -> None: + """The path proof covers the links on the way to `SOURCE`, not every + condition the reader puts on reading a whole linked directory, so it is + necessary for naming a limit, not sufficient. A head that adds a link + inside `.agents/skills` leaves every link on the way unchanged, but the + reader no longer reads through `.claude/skills`: that side carries an + `unreadable` limit, never an unchanged one, and every route still refuses.""" + + files, links = LAYOUTS["directory-link"] + repo = _repository( + tmp_path, + {**files, "img/logo.txt": "logo\n"}, + links, + head_links={".agents/skills/review/logo": "../../../img/logo.txt"}, + ) + + _text, payload, _verify_text, _comment = _all_routes(repo, tmp_path) + boundary = _check(repo, "agent-boundary-json") + + assert payload["comparison_status"] == "incomparable" + assert payload["rows"] == [] and payload["unchanged_limits"] == [] + assert ("unreadable", "head") in { + (item["limit"], item["side"]) for item in payload["coverage"]["items"] + } + assert boundary["comparison_status"] == "incomparable" + assert boundary["rows"] == [] From 62cadf4d6ab023c933c5a80ce5be6a5f796d8f37 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 09:49:38 -0700 Subject: [PATCH 11/14] Address review cycle 4 on agent launches in CI (#823) Four review cycles each found a new shell form (quoted words, $(...), here-docs, comments, reserved words) that the run: reader mis-read. The cycle-4 finding was the fourth: a `#` comment was split as commands, so an apostrophe in a comment hid a launch and a command line in a comment invented one. Parsing arbitrary shell cannot converge, so the reader now claims only forms it reads exactly and names every other one as a limit. - A run: step is an agent launch only when it is one line of plain words (letters, digits and `_ . / : = , % + -`, separated by spaces or tabs), run by bash, sh or no declared shell:, whose program's file name is `claude` with -p/--print or `codex` followed by exec. Every POSIX shell runs such text as exactly those words; a cross-check of 2251 generated texts in sh, bash, dash and ksh agrees on every one. - Any other run: that mentions claude or codex as a word of its own is an `unread_agent_runs[]` entry (job, step, agent) on the workflow grant: a non-blocking coverage issue in audit --host that publishes none of its text, is never compared, so it gives no row, and never says whether the step starts an agent. It takes no gain from another launch; only when an unread step goes and a read launch is added in the same job is a rule that launch meets named and not claimed, since it may be that step rewritten. - claude_args and codex-args are read only as a plain list of words (the same characters and parentheses, across blanks and newlines, with no --settings or --mcp-config flag). Any other value, a ${{ }} expression included, is `unread_arguments`: published only as a digest, so an edit is a changed row, and read for no rule; a rule a launch gains where that input was unread before is named and not claimed. - A codex --config override publishes its key; its value is under env, headers or a secret-named key, as written for sandbox_mode, default_permissions, approval_policy and model, and a digest otherwise. The word after a secret-named word such as --token is and the value is then published redacted, a named limit. - The shell tokenizer, the command-substitution and here-doc scanners, the reserved-word reader and the shell-quote and string-argv emulations are removed, with the expression-prefix reading of argument inputs. JSON is read only in the settings and mcp_config inputs. Host-grants 0.7 is unreleased, so its schema changes in place: the launch `unresolved_reason` is only inputs_not_a_mapping, a setting may be unread_arguments, and the workflow grant adds unread_agent_runs. The support page, STABILITY, CHANGELOG, the current contract page and llms-full.txt describe the tightened reader. --- CHANGELOG.md | 2 +- STABILITY.md | 77 +- docs/agent-contract-current.md | 62 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 303 ++-- docs/host-grants-baseline-schema.v0.7.json | 54 +- docs/host-grants-inventory-schema.v0.7.json | 54 +- docs/integrations.md | 14 +- llms-full.txt | 62 +- .../core/capability_diff_rows.py | 51 +- src/agents_shipgate/core/host_grants.py | 1233 ++++----------- src/agents_shipgate/schemas/contract.py | 13 +- src/agents_shipgate/schemas/host_grants.py | 167 +- tests/test_distribution_surface_parity.py | 2 +- tests/test_workflow_agent_launches.py | 1401 ++++++++--------- 15 files changed, 1496 insertions(+), 2001 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ef4a84de..ceaae396c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one literal `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing `claude_args` from `--allowedTools "Read"` to `--permission-mode bypassPermissions --allowedTools "Bash(*)"`, adding a `claude -p --permission-mode acceptEdits` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. `claude_args` and `codex-args` are split as each action splits them — on several lines, with unquoted `Bash(...)`, without the full-line `#` comments the Claude actions drop — and the rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; a rule is read only from literal text a `${{ }}` expression cannot reach, and one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, while a job that remains has not been left, since it may still run its launch in a form this audit does not read. A compound command, an expansion or an expression — an agent CLI inside a quoted `$(…)` or a backtick substitution, or after a reserved word such as `then`, included — is a named non-blocking limit and publishes none of its text, while a here-doc's body is input and never a command, so a Markdown `` `claude -p …` `` in a comment drafted with `cat <<'EOF'` is no launch and only an unquoted `<` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, and one the job's unread step or unread input may already have met is named, not claimed, while an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index f692b49c2..b2247d879 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -21,32 +21,34 @@ workspace too. Also in unreleased runtime contract v41, extended in place: the host inventory reads how a coding agent is launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and -drift schemas move to `0.7`: a workflow -grant adds `agent_launches[]` — a documented agent action's permission inputs, -or the permission flags of a `run:` that is one literal `claude -p` or -`codex exec` command, compared as text and never executed — and -`checkout_refs[]`, each `actions/checkout` step's `with.ref`. Only a documented -rule a job's launches gain widens (`workflow_agent_widened_`); -every other edit is a `changed` row naming `job/step`, and a workflow row that -runs an agent ends with the job facts beside each agent step. The rules each -launch meets are read from its declared text and published as -`widening_rules`, and `claude_args` and `codex-args` are split as each action -splits them, and only from literal text a `${{ }}` expression cannot reach (a -setting holding one says so with `holds_expression`); a rule a launch already -met in a job it left is moved, not gained. A JSON object in a setting, however -it is attached to its flag, publishes its shape and none of its free text: key -names, with each string a `` digest except those a host reader -publishes (a permission rule, a documented setting's value, an MCP server's -command name and URL host), so an MCP server's arguments and a hook's command -are compared but never published; a URL publishes its scheme and host. A -compound command, an expansion or an expression in -`run:` is `unresolved`, a named non-blocking limit that leaves coverage -complete; a setting holding credential-shaped text, prose included, is -published redacted, compared as published and named the same way, while a -checkout ref holding it refuses as a redacted step reference does. A `0.4`–`0.6` baseline holding a workflow -grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one -without a workflow stays comparable. It moves neither #821's verifier `0.21` -nor its capability diff `0.4`, and `minimum_control_contract_version` stays +drift schemas move to `0.7`: a workflow grant adds `agent_launches[]` — a +documented agent action's permission inputs, or the permission flags of a +`run:` that is one line of plain words running `claude -p` or `codex exec`, +compared as text and never executed — `unread_agent_runs[]`, each other +`run:` that mentions an agent CLI, and `checkout_refs[]`, each +`actions/checkout` step's `with.ref`. Shell is not parsed: only a plain list +of words is read, as a `run:` command or as `claude_args` / `codex-args`, and +any other form is a named non-blocking limit that publishes none of its text — +an unread `run:` is never compared and gives no row, and an unread argument +input is compared by a digest and read for no rule. Only a documented rule a +job's launches gain widens (`workflow_agent_widened_`); every +other edit is a `changed` row naming `job/step`, and a workflow row that runs +an agent ends with the job facts beside each agent step. The rules each launch +meets are read from its declared text and published as `widening_rules`; a +rule a launch already met in a job it left is moved, not gained, and one the +job's unread step or unread input may already have met is named, not claimed. +A JSON object in a `settings` or `mcp_config` input publishes its shape and +none of its free text: key names, with each string a `` digest +except those a host reader publishes (a permission rule, a documented +setting's value, an MCP server's command name and URL host), so an MCP +server's arguments and a hook's command are compared but never published; a +URL publishes its scheme and host. A setting holding credential-shaped text, +prose included, is published redacted, compared as published and named as a +non-blocking limit, while a checkout ref holding it refuses as a redacted step +reference does. A `0.4`–`0.6` baseline holding a workflow grant is +incomparable (`baseline_workflow_agent_launches_unavailable`); one without a +workflow stays comparable. It moves neither #821's verifier `0.21` nor its +capability diff `0.4`, and `minimum_control_contract_version` stays `21`. See [the migration note](#workflow-agent-launches-contract-v41-823). Also unreleased, and moving no version of its own: a Claude Code setting that @@ -335,7 +337,7 @@ is added. ## Migration Note: Unreleased — workflow agent launches (host-grants `0.7`, contract v41, #823) -Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` rather than extending it in place, and extends in place the unreleased runtime contract `41` that #821 minted ([its note](#unread-changed-inputs-821)). The `0.6` schema files stay published and unchanged. A workflow grant adds two members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: +Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` rather than extending it in place, and extends in place the unreleased runtime contract `41` that #821 minted ([its note](#unread-changed-inputs-821)). The `0.6` schema files stay published and unchanged. A workflow grant adds three members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: ```json { @@ -344,26 +346,31 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin "job": "review", "step": "steps[1]", "agent": "anthropics/claude-code-action", "form": "read", "unresolved_reason": null, "settings": [ - {"name": "claude_args", "value": "--permission-mode bypassPermissions --allowedTools \"Bash(*)\"", "unresolved_reason": null} + {"name": "claude_args", "value": "--permission-mode bypassPermissions --allowedTools Bash", "unresolved_reason": null} ], "widening_rules": [{"rule": "bypass_permissions", "setting": "claude_args"}], "job_secrets": ["CLAUDE_CODE_OAUTH_TOKEN"] } ], + "unread_agent_runs": [ + {"job": "review", "step": "steps[2]", "agent": "claude"} + ], "checkout_refs": [ {"job": "review", "step": "steps[0]", "ref": "${{ github.event.pull_request.head.sha }}", "unresolved_reason": null} ] } ``` -- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets; a `run:` that is one literal simple command starting with `claude` and passing `-p`/`--print`, or with `codex exec` (`codex e`), lists its documented permission flags under their primary spelling. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). -- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`, `--mcp-config`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. An agent action's `claude_args` or `codex-args` is compared whole, as the text the action parses: the Claude actions drop full-line `#` comments, which are therefore neither published nor compared. -- **What is withheld.** A structured value publishes its shape and none of its free text. A JSON object — a `settings` or `mcp_config` value, a `--settings` or `--mcp-config` value written as its own word or attached as `--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex `--config` table or array publish, as canonical JSON, their key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. A codex `--config` override (`-c`, `--config=`, `-c`, `-c=`) under `env`, `headers` or a secret-named key publishes `` for its value. Other argument text — a prompt, a flag's value, a codex `--config` override's scalar value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path and no query, as an MCP server's URL does (#723), so a change only to such a URL's path or query is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text that starts like JSON, or a codex `--config` table or array, and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions`, `--permission-mode bypassPermissions`, or Claude Code settings written as JSON — the `settings` input, or a `--settings` value — whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one; `claude_args` and `codex-args` are split as each action splits them, so a rule on any line of `claude_args: |` counts and one in a full-line `#` comment does not. It is read from literal text only: GitHub substitutes a `${{ }}` expression before the action reads the input, so a rule is read from the words of `claude_args` or `codex-args` before the first expression (less the word it touches and a quoted run open at it; in a JSON-array `codex-args`, the elements before the one holding it) and from the gate entries that hold none, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none. A setting holding one is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where the job launched that agent before only in a form this audit does not read, as for a job whose permissions were not explicit; where the job's launch held a `${{ }}` expression before in an input the rule is read from, whose substituted text may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it), as a step reference moved between jobs adds no scope. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools "Bash(*)"` (rating its reach is #824's), `acceptEdits`, a new plugin and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. -- **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. A removed workflow gets none. -- **Unresolved and unreadable values are a named limit, not a blocking one.** A `run:` holding more than one command or a shell reserved word such as `then` or `!`, a shell expansion — an agent CLI heading a command inside a `$(…)` or backtick substitution, double-quoted or not, included — or a `${{ }}` expression is `form: unresolved` (`compound_command`, `shell_expansion`, `expression`) with no settings, as is an agent action whose `with:` is not a mapping (`inputs_not_a_mapping`). A here-doc's body is input to its command, never a command: a `claude -p` line in it, or a backticked `` `claude -p …` `` in a body whose delimiter is quoted (`<<'EOF'`, `<<"EOF"`, `<<\EOF`), is no launch, while an agent CLI heading a `$(…)` or backtick substitution in an unquoted `<`, a short digest, so an edit is a `changed` row showing the digest, and no documented widening rule is read from it. So a quoted `claude_args` such as the #823 reproduction's `--allowedTools "Read"` is compared only by its digest. +- **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. `unread_agent_runs` is never compared: adding, removing or editing an unread step gives no row. +- **What is withheld.** A JSON object in a `settings` or `mcp_config` input publishes, as canonical JSON, its key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so never published. A codex `--config` override in a plain list of words (`-c`, `--config=`, `-c`, `-c=`) publishes its key, and its value as `` under `env`, `headers` or a secret-named key, as written for `sandbox_mode`, `default_permissions`, `approval_policy` and `model`, and as a `` digest under any other key, such as an MCP server's `command` or `url`. The word after a secret-named word such as `--token` or `password` is ``, as the host readers redact it among an MCP server's arguments, and the value is then published redacted (`redacted`, below). Other argument text — a prompt word, a flag's value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path, as an MCP server's URL does (#723), so a change only to such a URL's path is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text in a `settings` or `mcp_config` input that starts like JSON and does not parse is withheld whole (`unparsed_json`). +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. An unread step is not an agent step here. A removed workflow gets none. +- **Unread and unreadable values are a named limit, not a blocking one.** An unread `run:` step, an argument input that is not a plain list of words (`unread_arguments`), an agent action whose `with:` is not a mapping (`form: unresolved`, `inputs_not_a_mapping`), a setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), a setting holding credential-shaped text (`redacted`), and a ref that is not a string each record a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`. An unread `run:` step is never compared, so it gives no row whatever is edited, and it never says the step starts or does not start an agent. For the others, adding, removing or re-forming the entry, or its gaining a rule, is still a row; only an edit inside it that gains no rule is not reported (an unread argument input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them: a change that only adds an unread step prints `No static host-grant changes detected`, and `audit --host` names the step. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. -- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through another command (`npx`, `timeout`, `sudo`, `bash -c`, a path), `codex` with an option before `exec`, and a step's `env:`, `shell:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. +- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through a variable or a function, and a step's `env:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them, or an unread `run:`, is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. **Compatibility.** - **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index 857f54c50..8d2a49b5c 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -47,38 +47,44 @@ comparison or a `scope` is refused. See Runtime contract v41, extended in place, also reads how a coding agent is launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow -grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. -An agent launch is a step whose `uses:` is a documented agent action -(`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, -`openai/codex-action`) with the permission inputs it declares, or a `run:` -that is one literal `claude -p` / `codex exec` command with its documented -permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, -`holds_expression`), -`widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is -each `actions/checkout` step's `with.ref`, `null` for the default. Values are -compared as text and never executed; `claude_args` and `codex-args` are split -as each action splits them. A JSON object in a setting publishes its shape and -none of its free text — key names, with each string a `` digest -except those a host reader publishes (a permission rule, a documented -setting's value, an MCP server's command name and URL host) — so an MCP -server's arguments and a hook's command are compared but never published; a -URL publishes its scheme and host. Only a documented rule a job's launches -gain — bypassed permission checks (a flag, or JSON settings whose +grant adds `agent_launches[]`, `unread_agent_runs[]` and `checkout_refs[]`, +each omitted when empty. An agent launch is a step whose `uses:` is a +documented agent action (`anthropics/claude-code-action`, +`anthropics/claude-code-base-action`, `openai/codex-action`) with the +permission inputs it declares, or a `run:` that is one line of plain words +running `claude -p` / `codex exec` under `bash` or `sh`, with its documented +permission flags; its `job`, `step`, `agent`, `form` (`read`, or `unresolved` +with `inputs_not_a_mapping`), `settings[]` (`name`, `value`, +`unresolved_reason`, `holds_expression`), `widening_rules[]` (`rule`, +`setting`) and `job_secrets[]`. Shell is not parsed: any other `run:` that +mentions `claude` or `codex` is an `unread_agent_runs[]` entry (`job`, `step`, +`agent`), a named non-blocking limit that publishes none of its text, is never +compared and gives no row; and `claude_args` / `codex-args` are read only as a +plain list of words, any other value being `unread_arguments`, compared by a +digest and read for no rule. A checkout ref is each `actions/checkout` step's +`with.ref`, `null` for the default. Values are compared as text and never +executed. A JSON object in a `settings` or `mcp_config` input publishes its +shape and none of its free text — key names, with each string a +`` digest except those a host reader publishes (a permission rule, +a documented setting's value, an MCP server's command name and URL host) — so +an MCP server's arguments and a hook's command are compared but never +published; a URL publishes its scheme and host. Only a documented rule a job's +launches gain — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included, and in a `codex exec` step without `--sandbox` a `--config` override of `sandbox_mode` -or `default_permissions` that selects it), -`safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the -row `widened`. A rule is read only from literal text a `${{ }}` expression -cannot reach, and one a launch already met in a job it left, or where the job's -launch before was unread or held an expression the rule is read from, is named +or `default_permissions` that selects it), `safety-strategy: unsafe`, or a +user gate opened to `*` — raises `workflow_agent_widened_` and +makes the row `widened`. A rule is read only from text this audit reads +exactly, and one a launch already met in a job it left, one where an unread +step of the job became a read launch, or one where the job's launch before +held an expression or an unread argument input the rule is read from, is named and not claimed. Every other edit is `changed`, and a workflow row that runs an -agent ends its `why` with the job facts beside each agent step. A compound -`run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit; a setting holding credential-shaped text, prose included, is published -redacted, compared as published and a named non-blocking limit, and a checkout -ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +agent ends its `why` with the job facts beside each agent step. An action whose +`with:` is not a mapping is `unresolved` and a named non-blocking limit; a +setting holding credential-shaped text, prose included, is published redacted, +compared as published and a named non-blocking limit, and a checkout ref +holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. It moves neither #821's verifier `0.21` nor its capability diff diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index 0def93f90..2302b3011 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a literal `claude -p` / `codex exec` run step — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from literal text a `${{ }}` expression cannot reach, and a gain the engine does not claim (a rule moved in from a job the launch left, or one the job's unread or expression-holding launch before may already have met) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unresolved launch, an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unread argument input, an unresolved launch, an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 9d6c63461..42da440f1 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -19,7 +19,7 @@ and `audit --host`. | Claude Code | first-class | `.claude/settings.json`, `.claude/settings.local.json`, `.mcp.json`, `CLAUDE.md`, Claude skills | permission modes/rules, sandbox/network, additional paths, MCP restrictions, plugins and their marketplaces (`extraKnownMarketplaces`), hooks | | Cursor | first-class | `.cursor/cli.json`, `.cursor/mcp.json`, `.cursor/rules/**` | Shell/Read/Write rules, MCP declarations, instruction trust roots | | VS Code MCP | first-class | `.vscode/mcp.json` | MCP servers; `sandbox` and per-server `sandboxEnabled`; `${input:…}` references by name, never value; `envFile` recorded as a limit; other top-level keys partial | -| Shared/GitHub | first-class | `AGENTS.md`, Shipgate policies/state, skills, `.github/workflows/*` | instruction/gate weakening, workflow permissions and triggers, remote step action references, named secret sources passed to reusable workflows, agent launches (documented agent action inputs, literal `claude -p` / `codex exec` run steps) and `actions/checkout` refs | +| Shared/GitHub | first-class | `AGENTS.md`, Shipgate policies/state, skills, `.github/workflows/*` | instruction/gate weakening, workflow permissions and triggers, remote step action references, named secret sources passed to reusable workflows, agent launches (documented agent action inputs, `run:` steps that are one plain `claude -p` / `codex exec` command, `run:` steps that mention an agent CLI in shell this audit does not parse, named as a limit) and `actions/checkout` refs | A registered adapter reports `complete`, `not_applicable`, `partial`, or `experimental` coverage. A relevant malformed, unreadable, binary, oversized, @@ -64,18 +64,20 @@ the changed inputs the candidate rules at the end of this section name (#821): them while that subagent runs; no adapter reads the file. A skill's `hooks` frontmatter is type-checked with the skill's instructions, never read as a hook grant, so its events get no hook row (#714). -- **An agent launched any way the workflow reader below does not recognise** +- **An agent launched any way the workflow reader below does not read** (#823): an action outside its table, even one that takes `claude_args`; a - composite action (#701); a script the step runs (`run: ./scripts/review.sh`); - an agent CLI reached through another command (`npx @anthropic-ai/claude-code`, - `timeout 600 claude`, `sudo`, `bash -c`, a path such as - `./node_modules/.bin/claude`); `codex` with an option before `exec` - (`codex -c sandbox_mode=danger-full-access exec`, `codex --yolo exec`), which - can change how `exec` runs; and a step's `env:`, `shell:` and `if:`. - Editing one gives no row and names no limit. A launch this audit read that - becomes one of these is a row saying the step no longer declares an agent - launch this audit reads, and that it may still start one this way; it never - says the step no longer starts an agent. + composite action (#701); a script the step runs; a `run:` that is not one + line of plain words running `claude -p` or `codex exec` under `bash` or `sh` + (more than one line or command, quoting, an expansion, a redirection, a + comment, a here-doc, `npx`, `timeout`, `sudo`, `bash -c`, `codex` with an + option before `exec`); an agent CLI reached through a variable or a + function; and a step's `env:` and `if:`. + A `run:` among these that mentions `claude` or `codex` as a word of its own + is named as a non-blocking limit (below); editing, adding or removing it + gives no row. The rest give no row and name no limit. A launch this audit + read that becomes one of these is a row saying the step no longer declares + an agent launch this audit reads, and that it may still start one this way; + it never says the step no longer starts an agent. Review changes to those files and fields as you would a change to the workflow, hook or server entry that holds them. @@ -131,9 +133,11 @@ redact alike never compare as unchanged. How a coding agent is launched inside a job is read (#823). Every value is compared as the text it declares, less what the host readers withhold (below); -no action is fetched, no command is run and no expression is evaluated. Three -things are listed on the workflow grant, each naming its `job/step` (the step's -`id`, else its `name`, else `steps[N]`): +no action is fetched, no command is run and no expression is evaluated. Shell +is not parsed: a value is read only in a form every parser involved reads the +same way, a plain list of words, and every other form is named as a limit and +never guessed at. Four things are listed on the workflow grant, each naming its +`job/step` (the step's `id`, else its `name`, else `steps[N]`): - **A documented agent action** — a step whose `uses:` is one of these `owner/repo` references, at any ref and in any letter case — with the inputs @@ -142,39 +146,53 @@ things are listed on the workflow grant, each naming its `job/step` (the step's | Action | Inputs compared as text | Documented widening | |---|---|---| - | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` gains `--dangerously-skip-permissions`, `--permission-mode bypassPermissions` or a JSON `--settings` value whose `defaultMode` is `bypassPermissions`; `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | + | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | a plain `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions`; `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` and `settings` as above | - | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access` (`-s`, attached or not); `allow-users` gains a `*` entry. A sandbox `--config` override in `codex-args` meets none: after `codex-args` the action appends its own `--sandbox`, or its own `default_permissions` override for a `permission-profile`, which takes precedence | - - No shell reads `claude_args` or `codex-args`: each action splits its own - input, and a rule is met only by the words the action passes on. The - Claude actions (`base-action/src/parse-sdk-options.ts`) drop each line whose - first non-blank character is `#`, which is then neither published nor - compared, and split the rest with shell-quote, taking `()|&;<>` literally: - newlines separate words as spaces do, so `claude_args: |` on several lines - reads as it would on one; quotes and backslashes work as in a shell; - `$NAME` reads as empty; an unquoted `#` later in the input ends it; and a - word starting with `--` is always a flag, never another flag's value. - `openai/codex-action` reads `codex-args` as a JSON array of strings or, when - it does not start with `[`, as words separated by whitespace, newlines - included, with a quoted string kept together (string-argv). - -- **A literal agent CLI command in `run:`** — only when the whole `run:` is one - simple command, after any literal `NAME=value` assignments (which are skipped - and never published), that starts with `claude` and passes `-p`/`--print`, or - starts with `codex exec` (`codex e`). Its documented permission flags are - listed under their primary spelling, and every other word — the prompt, - `--model`, an undocumented flag — is not compared. For `claude`: - `--permission-mode`, `--dangerously-skip-permissions`, - `--allow-dangerously-skip-permissions`, `--allowedTools`/`--allowed-tools`, - `--disallowedTools`/`--disallowed-tools`, `--add-dir`, `--mcp-config`, - `--settings` and `--permission-prompt-tool`; gaining - `--dangerously-skip-permissions`, `--permission-mode bypassPermissions` or a - JSON `--settings` value whose `defaultMode` is `bypassPermissions` (one - rule, so moving between the spellings is not a widening) widens; a - `--settings` path names a file this audit does not read. - For `codex exec`: `--sandbox`/`-s`, - `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, + | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; a plain `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access` (`-s`, attached or not); `allow-users` gains a `*` entry. A sandbox `--config` override in `codex-args` meets none: after `codex-args` the action appends its own `--sandbox`, or its own `default_permissions` override for a `permission-profile`, which takes precedence | + + `claude_args` and `codex-args` are read only when they are a **plain list + of words**: words made of letters, digits and `_ . / : = , % + - ( )`, + separated by blanks or newlines, with no `--settings` or `--mcp-config` flag + in any spelling. The Claude actions (`base-action/src/parse-sdk-options.ts`, + shell-quote with `()|&;<>` made literal) and `openai/codex-action` + (string-argv) both split such text at its blanks and nowhere else, so it is + published as those words, one space apart — `claude_args: |` on several + lines reads as it would on one, and reformatting it is quiet — and, as the + Claude actions read it, a word starting with `--` is always a flag, never + another flag's value. Any other + value — holding a quote, a `${{ }}` expression, `$`, a backtick, a + backslash, a `#` comment, `;`, `&`, `|`, `<`, `>`, a glob, JSON, a + `--settings` or `--mcp-config` flag, or any other character — is **not + read** (`unresolved_reason: unread_arguments`): none of its text is + published, its `value` is ``, a short digest, so an edit to it + is a `changed` row whose cell shows the digest, no documented widening rule + is read from it, and it is a non-blocking limit (below). So the + reproduction's `--allowedTools "Read"` → + `--permission-mode bypassPermissions --allowedTools "Bash(*)"` is a + `changed` row saying the input is not read, while the same change written + `--allowedTools Read` → `--permission-mode bypassPermissions --allowedTools Bash` + is `widened`. + +- **An agent CLI launched by a `run:`** — only when the whole `run:` is **one + line of plain words**: letters, digits and `_ . / : = , % + -`, separated by + spaces or tabs, so it holds no quote, `$`, backtick, backslash, `#`, `;`, + `&`, `|`, `<`, `>`, parenthesis, brace, glob, `~`, `!`, `@` or second line, + and every POSIX shell runs it as exactly those words. It must run under + `bash`, `sh` or no declared `shell:` (the step's, else its job's or its + workflow's `defaults.run.shell`). After any `NAME=value` assignments, which + are skipped and never published, the program's file name must be `claude` + with `-p`/`--print` among its arguments, or `codex` followed by `exec` + (`codex e`): `claude -p …`, `./node_modules/.bin/claude -p …` and + `CI=1 codex exec …` are read. Its documented permission flags are listed + under their primary spelling, and every other word — the prompt, `--model`, + an undocumented flag — is not compared. For `claude`: `--permission-mode`, + `--dangerously-skip-permissions`, `--allow-dangerously-skip-permissions`, + `--allowedTools`/`--allowed-tools`, `--disallowedTools`/`--disallowed-tools`, + `--add-dir` and `--permission-prompt-tool`; gaining + `--dangerously-skip-permissions` or `--permission-mode bypassPermissions` + (one rule, so moving between the spellings is not a widening) widens, and a + `--settings` or `--mcp-config` flag makes the step unread. For `codex exec`: + `--sandbox`/`-s`, `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, `--approve-for-me`/`--not-so-yolo`, `--dangerously-bypass-hook-trust`, `--add-dir`, `--config`/`-c` and `--profile`/`-p`, a short flag's value read attached as clap reads it (`-sdanger-full-access`, `-s=…`, @@ -185,37 +203,31 @@ things are listed on the workflow grant, each naming its `job/step` (the step's override in any of its four spellings that sets `sandbox_mode` to `danger-full-access` (the setting `--sandbox` sets) or `default_permissions` to `:danger-full-access` (the built-in full-access profile, which the - action's `permission-profile` input passes the CLI the same way), its value - read as TOML and otherwise as text with its quotes trimmed, as the CLI - reads it. The last override of a key counts, and a `default_permissions` - override outranks a `sandbox_mode` one. A key under another table, such as + action's `permission-profile` input passes the CLI the same way). The last + override of a key counts, and a `default_permissions` override outranks a + `sandbox_mode` one. A key under another table, such as `profiles..sandbox_mode`, and a `--profile`, which names a - configuration this audit does not read, meet none. A flag the CLI reads as variadic (`--allowedTools`, `--add-dir`, …) - takes every following word up to the next word starting with `-`, as the CLI - reads it, so a prompt written after it is compared as one of its values; the - row shows it. A `run:` holding more than one command (a newline, `&&`, `;`, - `|`, a redirection or a here-doc), a shell reserved word before a command - (`if … then`, `for … do`, `{ …; }`, `!`, `time`) or quoting that does not - balance, a shell expansion (`$VAR`, `$(…)`, a backtick) or a `${{ }}` - expression is listed as `unresolved` with that reason, once for each agent - CLI it starts at the head of a command (of a line, when its quoting does not - balance), and none of its text is published. A command's head is read after - any reserved words, and inside each `$(…)` or backtick substitution outside - single quotes, so `gh pr comment --body "$(claude -p …)"` and - ``REVIEW=`claude -p …` `` are listed (`shell_expansion`), and so is - `if …; then claude -p …; fi` (`compound_command`). A command that launches no - headless agent — `claude mcp add`, `codex login`, an `echo` that mentions - either, a quoted `"claude -p …"` or a single-quoted `'$(claude -p …)'` — is - not listed. Nor is the body of a here-doc, which the shell passes to its - command as input and never runs: a `claude -p` line, or a Markdown - `` `claude -p …` `` in a comment drafted with `cat <<'EOF'`, starts no agent. - Only when no part of the delimiter is quoted (`<`, the row is `widened` and its `why` names the rule and step. Three gains are named in the `why` and not claimed: -- where the job launched that agent before only in a form this audit does not - read — a compound `run:` that became a literal one — because the unread - launch may already have met it, as a job whose permissions were not explicit - may already have held a write scope; -- where the job's launch of that agent held a `${{ }}` expression before in an - input the rule is read from (`claude_args` or `settings` for bypassed - permission checks, `sandbox`, `permission-profile` or `codex-args` for a - full-access sandbox, the gate for a `*` entry), because the substituted text - may already have met - it — so replacing `--model ${{ vars.M }}` with `--model opus` beside - `--dangerously-skip-permissions` is not a widening; +- where, in the same job, a step that may launch that agent in a form this + audit does not read (an unread agent step, or an action whose `with:` is + not a mapping) is gone and a launch this audit reads is added, because the + added launch may be that step rewritten, which may already have met the + rule — so rewriting `npm ci && claude -p --dangerously-skip-permissions "Review"` + as two plain steps is not a widening. An unread step that remains takes no + gain from another launch: beside it, a new plain + `claude -p --dangerously-skip-permissions Review` step is a widening; +- where the job's launch of that agent held, before, text this audit did not + read for the rule in an input the rule is read from (`claude_args` or + `settings` for bypassed permission checks, `sandbox`, `permission-profile` + or `codex-args` for a full-access sandbox, the gate for a `*` entry): a + `${{ }}` expression, or an argument input that was not a plain list of + words, because it may already have met it — so unquoting + `--dangerously-skip-permissions --append-system-prompt "Review"` is not a + widening; - where the rule moved between jobs: another job met it before and the launch that met it left that job — the job no longer exists, as when it is renamed, or the same launch now runs in this job while each launch of that agent this job had still runs here or in that job, as when an agent step moves or two jobs swap launches — as a step reference moved between jobs adds no scope. A second job gaining a rule a first job keeps, or a different launch - gaining it while the first job still exists, is claimed: that job may still run its launch - in a form this audit does not read (`npx`, a path), so a launch that only - stops being read has not left it, and a launch edited in place into the one - that job had gains the rule. + gaining it while the first job still exists, is claimed: that job may still + run its launch in a form this audit does not read (`npx`, quoting), so a + launch that only stops being read has not left it, and a launch edited in + place into the one that job had gains the rule. -Any other edit — `--allowedTools "Read"` to `--allowedTools "Bash(*)"`, -`acceptEdits`, a new plugin, a flag after a `${{ }}` expression, a head-ref -checkout — is `changed`; rating a tool rule's reach is a job for #824. `access` -and `risk` still describe the token and triggers alone. +Any other edit — `--allowedTools Read` to `--allowedTools Bash`, +`acceptEdits`, a new plugin, an argument input this audit does not read, a +head-ref checkout — is `changed`; rating a tool rule's reach is a job for +issue #824. `access` and `risk` still describe the token and triggers alone. A workflow row whose workflow runs an agent ends its `why` with the job facts beside each agent step, whatever else the row is about: an untrusted-input @@ -283,16 +297,15 @@ code (`github.event.pull_request.head.sha`, `.head.ref` or `.merge_commit_sha`, `github.head_ref`, `github.event.workflow_run.head_sha` or `.head_branch`, or `refs/pull//head` and `/merge`). It is a note, not a verdict: it moves no direction, and `if:` conditions and the default checkout of a `pull_request` -event are not read into it. A removed workflow gets no note. +event are not read into it. An unread agent step is not an agent step here. A +removed workflow gets no note. -A structured value publishes its shape and none of its free text, so a setting +A structured input publishes its shape and none of its free text, so a setting never publishes what `.claude/settings.json` and `.mcp.json` would withhold -(#823 review). A JSON object — a `settings` or `mcp_config` value, a -`--settings` or `--mcp-config` value written as its own word or as -`--settings={…}`, or any word of `claude_args` or `codex-args` — and a codex -`--config` table or array publish, in canonical JSON: +(#823 review). A JSON object — a `settings` or `mcp_config` value — publishes, +in canonical JSON: -- their key names, numbers, booleans and `null`, so reordering keys compares +- its key names, numbers, booleans and `null`, so reordering keys compares as unchanged and adding a key is a change; - `` for `env` and `headers` values (and codex's `http_headers` and `env_http_headers`), `apiKeyHelper`, every other secret-named value and the @@ -313,19 +326,26 @@ never publishes what `.claude/settings.json` and `.mcp.json` would withhold for `npx -y some-server`, a URL's digest for its query. A URL's path is neither published nor compared, as an MCP server's is not (#723). -A codex `--config` override — `-c key=value`, `--config=key=value`, -`-ckey=value` or `-c=key=value` — under `env`, `headers` or a secret-named key -publishes `` for its value; a table or array value is read under its -key path, so `mcp_servers.gh={command="gh", …}` is a server whose command name -is kept. Other argument text — a prompt, a flag's value, a codex `--config` -override's scalar value such as `model="o3"` — is published as written through -the label redaction below, except that a URL in it publishes its scheme, host -and port, with `` for any path and no query, as an MCP server's -URL does (#723); the rest of the setting is compared, so a change only to such -a URL's path or query — which repository a `plugin_marketplaces` URL names, for -one — is not reported, and a zero-row result says redacted values are not -compared. A `${{ }}` expression is one word while this is decided, so one -inside a URL's userinfo is withheld with it. +JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so +it is never published: the argument input or step is not read at all. A codex +`--config` override in a plain list of words — `-c key=value`, +`--config=key=value`, `-ckey=value` or `-c=key=value` — publishes its key; +its value is `` under `env`, `headers` or a secret-named key, so +rotating it is quiet, published as written for `sandbox_mode`, +`default_permissions`, `approval_policy` and `model`, and ``, a +digest, under any other key, such as an MCP server's `command` or `url` or a +`shell_environment_policy` value. The word after a secret-named word such +as `--token` or `password` is ``, as the host readers redact it +among an MCP server's arguments, and the value is then credential-shaped +(below). Other argument text — a prompt word, a +flag's value — is published as written through the label redaction below, +except that a URL in it publishes its scheme, host and port, with +`` for any path, as an MCP server's URL does (#723); the rest +of the setting is compared, so a change only to such a URL's path — which +repository a `plugin_marketplaces` URL names, for one — is not reported, and +a zero-row result says redacted values are not compared. A `${{ }}` +expression is one word while this is decided, so one inside a URL's userinfo +is withheld with it. Any other text the #802 label redaction rewrites is credential-shaped: a token shape, a credential assignment such as `token=…`, a bearer or header value, a @@ -340,17 +360,24 @@ same way a step reference does (#767), because two refs that redact alike cannot be compared apart: GitHub coverage is `partial`, a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. -An unresolved launch, a setting that is not a string (`not_a_string`), that -holds text starting like JSON or a codex `--config` table or array that does -not parse (`unparsed_json`, whose values cannot be told from its keys), or that -is published redacted (`redacted`), and a checkout ref that is not a string or -whose `with:` is not a mapping record a **non-blocking** `unsupported` coverage -issue naming the `job/step`, printed under `audit --host` → Coverage issues; -all but a redacted setting publish nothing of the value. GitHub coverage stays -complete, so `check`, baselines and every other row are unaffected, and adding, -removing or re-forming such an entry, or its gaining a documented rule, is -still a row; only an edit inside it that gains no rule is not reported. `diff`, -`verify` and `check` carry no limit for it, as for an unread secret value (#693). +An unread agent step; an argument input that is not a plain list of words +(`unread_arguments`); an action whose `with:` is not a mapping; a setting that +is not a string (`not_a_string`), that holds text starting like JSON that does +not parse (`unparsed_json`, whose values cannot be told from its keys), or +that is published redacted (`redacted`); and a checkout ref that is not a +string or whose `with:` is not a mapping record a **non-blocking** +`unsupported` coverage issue naming the `job/step`, printed under +`audit --host` → Coverage issues; none but a redacted setting publishes any of +the value's text. GitHub coverage stays complete, so `check`, baselines and +every other row are unaffected. An unread agent step is never compared, so +adding, removing or editing it gives no row; for the others, adding, removing +or re-forming the entry, or its gaining a documented rule, is still a row, and +only an edit inside it that gains no rule is not reported (an unread argument +input's edit is a `changed` row by its digest). `diff`, `verify` and `check` +carry no limit for any of them, as for an unread secret value (#693): a +`diff` of a change that only adds an unread agent step says +`No static host-grant changes detected`, and `audit --host` is where the step +is named. A workflow's labels are published redacted (#802). A job id, a step's `id` or `name`, a trigger and a permission scope name go through the same redaction as diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 16ac7e7c6..1672919e6 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1284,7 +1284,7 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command, a shell reserved word or\nquoting that does not balance, a shell expansion (an agent CLI inside a\ncommand substitution included), a ``${{ }}`` expression, or ``with:``\nthat is not a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``), or a known agent CLI a\n``run:`` launches when the whole ``run:`` is one line of plain words\n(letters, digits and ``_ . / : = , % + -``, separated by spaces or tabs),\nrun by ``bash``, ``sh`` or the runner's default shell, whose program,\nafter any ``NAME=value`` assignments, has the file name ``claude`` and\npasses ``-p``/``--print``, or ``codex`` followed by ``exec`` (``e``).\n``form: read`` lists the documented permission inputs or flags the step\ndeclares in ``settings``, and the documented widening rules they meet in\n``widening_rules``, omitted when none. ``form: unresolved`` is an agent\naction whose ``with:`` is not a mapping (``inputs_not_a_mapping``), with\nno settings, and records a non-blocking coverage issue. Any other\n``run:`` that mentions an agent CLI is not a launch: it is listed in\n``unread_agent_runs``. ``job_secrets`` names the secrets the step's job\nreferences (``${{ secrets.NAME }}``) and the workflow-level ``env``\npasses: context for the row that names this step, never compared.\n``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ @@ -1331,12 +1331,7 @@ "unresolved_reason": { "anyOf": [ { - "enum": [ - "compound_command", - "shell_expansion", - "expression", - "inputs_not_a_mapping" - ], + "const": "inputs_not_a_mapping", "type": "string" }, { @@ -1365,7 +1360,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode or ``settings`` input holding one meets\nnone. Claude Code settings written as JSON \u2014 the ``settings`` input, or a\n``--settings`` value in ``claude_args`` or on the CLI \u2014 meet\n``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``,\nread as the settings reader reads it; a path to a settings file is not\nread. ``setting`` is the input (``claude_args``, ``allowed_bots``,\n``sandbox``, ``permission-profile``, \u2026) or the CLI flag's primary\nspelling. One rule compares as one whatever setting meets it, except\n``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\ntext this reader reads exactly meets one: ``claude_args`` or\n``codex-args`` only when it is a plain list of words (never when it holds\na ``${{ }}`` expression), the entries of a user gate that hold no\nexpression, and a mode or ``settings`` input that holds none. Claude Code\nsettings written as JSON in the ``settings`` input meet\n``bypass_permissions`` when their ``defaultMode`` is\n``bypassPermissions``, read as the settings reader reads it; a path to a\nsettings file is not read. ``setting`` is the input (``claude_args``,\n``allowed_bots``, ``sandbox``, ``permission-profile``, \u2026) or the CLI\nflag's primary spelling. One rule compares as one whatever setting meets\nit, except ``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1392,7 +1387,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. A\nstructured value \u2014 a JSON object (a ``settings`` or ``mcp_config`` value,\na ``--settings`` or ``--mcp-config`` value, any argument word), or a codex\n``--config`` table or array \u2014 publishes its shape and none of its free\ntext: key names, numbers, booleans and ``null``, with each string\nreplaced by ````, a short digest of what the host readers\ndigest for it, so an edit to it is still a change. ``env`` and\n``headers`` values, ``apiKeyHelper`` and every secret-named value are\n````, as the host readers redact them. The strings a host reader\npublishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a\ndocumented Claude Code setting's value such as ``defaultMode``, and an\nMCP server's command name and its URL's scheme and host, each followed by\nthe digest when it drops something the digest reads (a command's\narguments, a URL's query). So an MCP server's arguments and a hook's\ncommand publish nothing, as `.mcp.json` and `.claude/settings.json` do\nnot (#823 review). A codex ``--config`` override under ``env``,\n``headers`` or a secret-named key publishes ```` for its value,\nand a URL elsewhere publishes its scheme and host with\n```` for its path and query (#723). Each is withheld\nhowever it is attached to its flag: ``--settings={\u2026}`` and\n``-c`` as well as a separate word. Other argument text \u2014 a\nprompt, a flag's value, a codex ``--config`` override's scalar value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", "properties": { "holds_expression": { "default": false, @@ -1409,7 +1404,8 @@ "enum": [ "not_a_string", "redacted", - "unparsed_json" + "unparsed_json", + "unread_arguments" ], "type": "string" }, @@ -1490,7 +1486,7 @@ }, "HostWorkflowGrantV7": { "additionalProperties": false, - "description": "A v0.6 workflow grant plus the agent launches and checkout refs its steps declare.\n\nBoth lists are present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``access`` and ``risk`` still describe the workflow's token and\ntriggers alone.", + "description": "A v0.6 workflow grant plus the agent launches, unread agent steps and checkout refs its steps declare.\n\nEach list is present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``unread_agent_runs`` is a named limit and is never compared.\n``access`` and ``risk`` still describe the workflow's token and triggers\nalone.", "properties": { "access": { "enum": [ @@ -1608,6 +1604,13 @@ "title": "Triggers", "type": "array" }, + "unread_agent_runs": { + "items": { + "$ref": "#/$defs/HostWorkflowUnreadAgentRunV7" + }, + "title": "Unread Agent Runs", + "type": "array" + }, "write_all": { "default": false, "title": "Write All", @@ -1734,6 +1737,35 @@ "title": "HostWorkflowStepActionV6", "type": "object" }, + "HostWorkflowUnreadAgentRunV7": { + "additionalProperties": false, + "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", + "properties": { + "agent": { + "enum": [ + "claude", + "codex" + ], + "title": "Agent", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "step": { + "title": "Step", + "type": "string" + } + }, + "required": [ + "job", + "step", + "agent" + ], + "title": "HostWorkflowUnreadAgentRunV7", + "type": "object" + }, "InstructionStructureEvidence": { "additionalProperties": false, "properties": { diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index 3efd9fef7..df23b5f8c 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1342,7 +1342,7 @@ }, "HostWorkflowAgentLaunchV7": { "additionalProperties": false, - "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``) or a known agent CLI a\nliteral ``run:`` starts with: ``claude`` with ``-p``/``--print``, or\n``codex exec``. ``form: read`` lists the documented permission inputs or\nflags the step declares in ``settings``, and the documented widening\nrules they meet in ``widening_rules``, omitted when none.\n``form: unresolved`` names why the step's settings were not\nread \u2014 a ``run:`` holding more than one command, a shell reserved word or\nquoting that does not balance, a shell expansion (an agent CLI inside a\ncommand substitution included), a ``${{ }}`` expression, or ``with:``\nthat is not a mapping \u2014 with no\nsettings, and records a non-blocking coverage issue. ``job_secrets`` names\nthe secrets the step's job references (``${{ secrets.NAME }}``) and the\nworkflow-level ``env`` passes: context for the row that names this step,\nnever compared. ``job`` and ``step`` are published labels (#802).", + "description": "A step that launches a known coding agent, read as text and never run (#823).\n\n``agent`` is a documented action reference's ``owner/repo`` (the step's\n``uses:`` at any ref; the Claude base action also as the ``base-action``\ndirectory of ``anthropics/claude-code-action``), or a known agent CLI a\n``run:`` launches when the whole ``run:`` is one line of plain words\n(letters, digits and ``_ . / : = , % + -``, separated by spaces or tabs),\nrun by ``bash``, ``sh`` or the runner's default shell, whose program,\nafter any ``NAME=value`` assignments, has the file name ``claude`` and\npasses ``-p``/``--print``, or ``codex`` followed by ``exec`` (``e``).\n``form: read`` lists the documented permission inputs or flags the step\ndeclares in ``settings``, and the documented widening rules they meet in\n``widening_rules``, omitted when none. ``form: unresolved`` is an agent\naction whose ``with:`` is not a mapping (``inputs_not_a_mapping``), with\nno settings, and records a non-blocking coverage issue. Any other\n``run:`` that mentions an agent CLI is not a launch: it is listed in\n``unread_agent_runs``. ``job_secrets`` names the secrets the step's job\nreferences (``${{ secrets.NAME }}``) and the workflow-level ``env``\npasses: context for the row that names this step, never compared.\n``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ @@ -1389,12 +1389,7 @@ "unresolved_reason": { "anyOf": [ { - "enum": [ - "compound_command", - "shell_expansion", - "expression", - "inputs_not_a_mapping" - ], + "const": "inputs_not_a_mapping", "type": "string" }, { @@ -1423,7 +1418,7 @@ }, "HostWorkflowAgentRuleV7": { "additionalProperties": false, - "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\nliteral text meets one: in a value holding ``${{ }}``, the words of\n``claude_args`` or ``codex-args`` before the first expression, less the\nword it touches and a quoted run still open at it, and the entries of a\nuser gate that hold none; a mode or ``settings`` input holding one meets\nnone. Claude Code settings written as JSON \u2014 the ``settings`` input, or a\n``--settings`` value in ``claude_args`` or on the CLI \u2014 meet\n``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``,\nread as the settings reader reads it; a path to a settings file is not\nread. ``setting`` is the input (``claude_args``, ``allowed_bots``,\n``sandbox``, ``permission-profile``, \u2026) or the CLI flag's primary\nspelling. One rule compares as one whatever setting meets it, except\n``open_gate``, which is one rule per gate input.", + "description": "One documented widening rule an agent launch meets, and the setting it was read from (#823).\n\nDecided when the workflow is read, from the declared text, before any of\nit is withheld for publication, so redaction never hides a rule. Only\ntext this reader reads exactly meets one: ``claude_args`` or\n``codex-args`` only when it is a plain list of words (never when it holds\na ``${{ }}`` expression), the entries of a user gate that hold no\nexpression, and a mode or ``settings`` input that holds none. Claude Code\nsettings written as JSON in the ``settings`` input meet\n``bypass_permissions`` when their ``defaultMode`` is\n``bypassPermissions``, read as the settings reader reads it; a path to a\nsettings file is not read. ``setting`` is the input (``claude_args``,\n``allowed_bots``, ``sandbox``, ``permission-profile``, \u2026) or the CLI\nflag's primary spelling. One rule compares as one whatever setting meets\nit, except ``open_gate``, which is one rule per gate input.", "properties": { "rule": { "enum": [ @@ -1450,7 +1445,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``. ``claude_args`` is the text the Claude\nactions parse, without the full-line ``#`` comments they drop. A\nstructured value \u2014 a JSON object (a ``settings`` or ``mcp_config`` value,\na ``--settings`` or ``--mcp-config`` value, any argument word), or a codex\n``--config`` table or array \u2014 publishes its shape and none of its free\ntext: key names, numbers, booleans and ``null``, with each string\nreplaced by ````, a short digest of what the host readers\ndigest for it, so an edit to it is still a change. ``env`` and\n``headers`` values, ``apiKeyHelper`` and every secret-named value are\n````, as the host readers redact them. The strings a host reader\npublishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a\ndocumented Claude Code setting's value such as ``defaultMode``, and an\nMCP server's command name and its URL's scheme and host, each followed by\nthe digest when it drops something the digest reads (a command's\narguments, a URL's query). So an MCP server's arguments and a hook's\ncommand publish nothing, as `.mcp.json` and `.claude/settings.json` do\nnot (#823 review). A codex ``--config`` override under ``env``,\n``headers`` or a secret-named key publishes ```` for its value,\nand a URL elsewhere publishes its scheme and host with\n```` for its path and query (#723). Each is withheld\nhowever it is attached to its flag: ``--settings={\u2026}`` and\n``-c`` as well as a separate word. Other argument text \u2014 a\nprompt, a flag's value, a codex ``--config`` override's scalar value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when the declared text holds a ``${{ }}``\nexpression, which GitHub substitutes before the action reads the input,\nand is omitted otherwise. A documented widening rule is then read only\nfrom the literal text the expression cannot reach, and a rule the launch\ngains in the same job afterwards is not claimed, because the substituted\ntext may already have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", "properties": { "holds_expression": { "default": false, @@ -1467,7 +1462,8 @@ "enum": [ "not_a_string", "redacted", - "unparsed_json" + "unparsed_json", + "unread_arguments" ], "type": "string" }, @@ -1548,7 +1544,7 @@ }, "HostWorkflowGrantV7": { "additionalProperties": false, - "description": "A v0.6 workflow grant plus the agent launches and checkout refs its steps declare.\n\nBoth lists are present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``access`` and ``risk`` still describe the workflow's token and\ntriggers alone.", + "description": "A v0.6 workflow grant plus the agent launches, unread agent steps and checkout refs its steps declare.\n\nEach list is present only when a step declares one. In a v0.7 grant an\nabsent list means the steps were read and declare none; the schema\nversion, not the key, separates that from a legacy grant that never read\nthem. ``unread_agent_runs`` is a named limit and is never compared.\n``access`` and ``risk`` still describe the workflow's token and triggers\nalone.", "properties": { "access": { "enum": [ @@ -1666,6 +1662,13 @@ "title": "Triggers", "type": "array" }, + "unread_agent_runs": { + "items": { + "$ref": "#/$defs/HostWorkflowUnreadAgentRunV7" + }, + "title": "Unread Agent Runs", + "type": "array" + }, "write_all": { "default": false, "title": "Write All", @@ -1792,6 +1795,35 @@ "title": "HostWorkflowStepActionV6", "type": "object" }, + "HostWorkflowUnreadAgentRunV7": { + "additionalProperties": false, + "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", + "properties": { + "agent": { + "enum": [ + "claude", + "codex" + ], + "title": "Agent", + "type": "string" + }, + "job": { + "title": "Job", + "type": "string" + }, + "step": { + "title": "Step", + "type": "string" + } + }, + "required": [ + "job", + "step", + "agent" + ], + "title": "HostWorkflowUnreadAgentRunV7", + "type": "object" + }, "InstructionStructureEvidence": { "additionalProperties": false, "properties": { diff --git a/docs/integrations.md b/docs/integrations.md index 30accb76c..e3e4c8c4a 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -225,12 +225,14 @@ stays quiet when no row widens what the agent can do. A workflow step moved to a different action reference, such as a pinned SHA to `@main`, is a non-widening row, so the hook stays quiet about it; `diff` and the PR comment still show it. The same holds for an agent launch in a workflow whose settings -change without gaining a documented widening rule, and for a checkout's ref. -One that gains a rule, such as `claude_args` gaining -`--dangerously-skip-permissions` on any of its lines, widens, and the hook -announces it (#823), unless the job launched that agent before only in a form -the audit does not read or with a `${{ }}` expression the rule is read from, -or the rule moved in from a job the launch left, as a renamed job's does. +change without gaining a documented widening rule, for a checkout's ref, and +for a `run:` or argument input the audit does not read, which it names as a +limit in `audit --host`. One that gains a rule, such as a plain `claude_args` +gaining `--dangerously-skip-permissions` on any of its lines, widens, and the +hook announces it (#823), unless the launch may be a step of that job the audit +did not read, rewritten, or the job's launch held before a `${{ }}` expression +or an unread argument input the rule is read from, or the rule moved in from a +job the launch left, as a renamed job's does. It names each widening row once, and repeats the announcement only when the change or its rows change. A missing base ref, an incomparable inventory or unparsed output is never diff --git a/llms-full.txt b/llms-full.txt index c3cbd29cd..925aa2e82 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1621,38 +1621,44 @@ comparison or a `scope` is refused. See Runtime contract v41, extended in place, also reads how a coding agent is launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and drift schemas move to `0.7`, and a workflow -grant adds `agent_launches[]` and `checkout_refs[]`, each omitted when empty. -An agent launch is a step whose `uses:` is a documented agent action -(`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, -`openai/codex-action`) with the permission inputs it declares, or a `run:` -that is one literal `claude -p` / `codex exec` command with its documented -permission flags; its `job`, `step`, `agent`, `form` (`read` or `unresolved` -with a reason), `settings[]` (`name`, `value`, `unresolved_reason`, -`holds_expression`), -`widening_rules[]` (`rule`, `setting`) and `job_secrets[]`. A checkout ref is -each `actions/checkout` step's `with.ref`, `null` for the default. Values are -compared as text and never executed; `claude_args` and `codex-args` are split -as each action splits them. A JSON object in a setting publishes its shape and -none of its free text — key names, with each string a `` digest -except those a host reader publishes (a permission rule, a documented -setting's value, an MCP server's command name and URL host) — so an MCP -server's arguments and a hook's command are compared but never published; a -URL publishes its scheme and host. Only a documented rule a job's launches -gain — bypassed permission checks (a flag, or JSON settings whose +grant adds `agent_launches[]`, `unread_agent_runs[]` and `checkout_refs[]`, +each omitted when empty. An agent launch is a step whose `uses:` is a +documented agent action (`anthropics/claude-code-action`, +`anthropics/claude-code-base-action`, `openai/codex-action`) with the +permission inputs it declares, or a `run:` that is one line of plain words +running `claude -p` / `codex exec` under `bash` or `sh`, with its documented +permission flags; its `job`, `step`, `agent`, `form` (`read`, or `unresolved` +with `inputs_not_a_mapping`), `settings[]` (`name`, `value`, +`unresolved_reason`, `holds_expression`), `widening_rules[]` (`rule`, +`setting`) and `job_secrets[]`. Shell is not parsed: any other `run:` that +mentions `claude` or `codex` is an `unread_agent_runs[]` entry (`job`, `step`, +`agent`), a named non-blocking limit that publishes none of its text, is never +compared and gives no row; and `claude_args` / `codex-args` are read only as a +plain list of words, any other value being `unread_arguments`, compared by a +digest and read for no rule. A checkout ref is each `actions/checkout` step's +`with.ref`, `null` for the default. Values are compared as text and never +executed. A JSON object in a `settings` or `mcp_config` input publishes its +shape and none of its free text — key names, with each string a +`` digest except those a host reader publishes (a permission rule, +a documented setting's value, an MCP server's command name and URL host) — so +an MCP server's arguments and a hook's command are compared but never +published; a URL publishes its scheme and host. Only a documented rule a job's +launches gain — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included, and in a `codex exec` step without `--sandbox` a `--config` override of `sandbox_mode` -or `default_permissions` that selects it), -`safety-strategy: unsafe`, or a user gate opened to `*` — raises `workflow_agent_widened_` and makes the -row `widened`. A rule is read only from literal text a `${{ }}` expression -cannot reach, and one a launch already met in a job it left, or where the job's -launch before was unread or held an expression the rule is read from, is named +or `default_permissions` that selects it), `safety-strategy: unsafe`, or a +user gate opened to `*` — raises `workflow_agent_widened_` and +makes the row `widened`. A rule is read only from text this audit reads +exactly, and one a launch already met in a job it left, one where an unread +step of the job became a read launch, or one where the job's launch before +held an expression or an unread argument input the rule is read from, is named and not claimed. Every other edit is `changed`, and a workflow row that runs an -agent ends its `why` with the job facts beside each agent step. A compound -`run:`, an expansion or an expression is `unresolved` and a named non-blocking -limit; a setting holding credential-shaped text, prose included, is published -redacted, compared as published and a named non-blocking limit, and a checkout -ref holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` +agent ends its `why` with the job facts beside each agent step. An action whose +`with:` is not a mapping is `unresolved` and a named non-blocking limit; a +setting holding credential-shaped text, prose included, is published redacted, +compared as published and a named non-blocking limit, and a checkout ref +holding it is a blocking limit, as a redacted step reference is. A `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`); one without a workflow stays comparable. It moves neither #821's verifier `0.21` nor its capability diff diff --git a/src/agents_shipgate/core/capability_diff_rows.py b/src/agents_shipgate/core/capability_diff_rows.py index 37fe2cbd8..49f008e33 100644 --- a/src/agents_shipgate/core/capability_diff_rows.py +++ b/src/agents_shipgate/core/capability_diff_rows.py @@ -292,8 +292,11 @@ def _agent_launch_value(item: dict[str, Any]) -> str: for setting in item.get("settings") or []: unread = setting.get("unresolved_reason") name = str(setting["name"]) + if unread == "unread_arguments": + # Compared by its digest alone, so the cell shows the digest (#823 review cycle 4). + parts.append(f"{name} (not read; digest {setting['value']})") # A redacted value is compared as published, so the cell shows it (#823 review F2). - if unread and not (unread == "redacted" and setting.get("value") is not None): + elif unread and not (unread == "redacted" and setting.get("value") is not None): parts.append(f"{name} (unresolved: {str(unread).replace('_', ' ')})") elif setting.get("value") is None: parts.append(name) @@ -326,10 +329,11 @@ def _agent_launch_reasons( Only a documented rule the engine claims — ``agent_rule_gains(...).claimed``, the rule behind ``workflow_agent_widened_*`` — is called a widening, and - the sentence says which rule and where. A rule gained where the job's - launch was unread before, or held a ``${{ }}`` expression the rule is read - from, or that moved in from another job, is named and not called a - widening, as the engine claims no expansion for it. Every other + the sentence says which rule and where. A rule gained where a step of the + job this audit does not read is gone, or whose launch held, in an + input the rule is read from, a ``${{ }}`` expression or an argument list + this audit does not read, or that moved in from another job, is named and + not called a widening, as the engine claims no expansion for it. Every other agent-launch or checkout edit is a change: its settings are compared as published text, and nothing here ranks one value against another. """ @@ -344,19 +348,24 @@ def where(item: dict[str, Any]) -> str: for _job, rule, detail, entry in gains.claimed: named.add(where(entry)) reasons.append(f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)})") - for _job, rule, detail, entry in gains.unread_before: + for (_job, rule, detail, entry), source in gains.unread_before: named.add(where(entry)) reasons.append( f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)}), which is not counted as a " - "widening: before, this job launched the agent in a form this audit does not read, which " - "may already have done the same" + f"widening: a step in this job that may launch the agent in a form this audit does not read " + f"is gone ({where(source)}), and this launch may be that step rewritten in a form this audit " + "reads, which may already have done the same" ) - for (_job, rule, detail, entry), setting in gains.expression_before: + for (_job, rule, detail, entry), setting, how in gains.setting_before: named.add(where(entry)) + held = ( + "held a `${{ }}` expression, whose substituted text this audit does not read" + if how == "expression" + else "was not a plain list of words this audit reads, so no rule was read from it" + ) reasons.append( f"an agent launch now {agent_rule_text(rule, detail)} ({where(entry)}), which is not counted as a " - f"widening: before, this job's {setting} held a " + "`${{ }}`" + " expression, whose " - "substituted text this audit does not read and which may already have done the same" + f"widening: before, this job's {setting} {held}, and it may already have done the same" ) for (_job, rule, detail, entry), source in gains.moved: named.update({where(entry), where(source)}) @@ -397,9 +406,9 @@ def where(item: dict[str, Any]) -> str: if "removed" in groups and after is not None: reasons.append( "a step that no longer declares one may still start an agent in a way this audit " - "does not read, such as an action outside its table, `npx`, a script, a path such as " - "`./node_modules/.bin/claude` or `codex` options before `exec`, so this row does not " - "say that it no longer starts one" + "does not read, such as an action outside its table, a script, or a `run:` this " + "audit does not read as a launch (more than one command, quoting, an expansion, `npx`, " + "`codex` options before `exec`), so this row does not say that it no longer starts one" ) # A rule is read only from literal text a `${{ }}` expression cannot # reach, so the row says where that leaves text unread (#823 review). @@ -416,6 +425,20 @@ def where(item: dict[str, Any]) -> str: "read only from the literal text the expression cannot reach, so this row does not " "say whether the text it reaches meets one" ) + # An argument input that is not a plain list of words is compared by its + # digest and read for no rule (#823 review cycle 4). + unread_arguments = list(dict.fromkeys( + f"{setting['name']} at {where(item)}" + for item in new + for setting in item.get("settings") or [] + if setting.get("unresolved_reason") == "unread_arguments" + )) + if unread_arguments: + reasons.append( + "an agent launch's argument input is not a plain list of words this audit reads (" + + ", ".join(unread_arguments) + "); none of its text is published and it is compared by " + "a digest only, so this row does not say whether it meets a documented widening rule" + ) unread = list(dict.fromkeys(where(item) for item in new if item.get("form") != "read")) if unread: reasons.append( diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 93fb029e8..d2ebfdbdc 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -8,14 +8,12 @@ from __future__ import annotations -import bisect import errno import hashlib import json import os import posixpath import re -import shlex import stat import sys import tomllib @@ -1610,11 +1608,16 @@ def step_action_key(entry: dict[str, Any]) -> tuple[str, str, str, str]: # How a coding agent is launched inside a job: a known agent action's permission # inputs, a literal agent CLI command in a `run:` step, and the ref each # `actions/checkout` step declares. Every value is compared as declared text; no -# action is fetched, no command is run and no expression is evaluated. A shape -# this reader does not read — a compound shell command, an expansion, a script, -# a composite or unknown action — is never guessed at: where it can be told -# apart it is listed as unresolved and named as a non-blocking limit, and -# otherwise it is one of the unread surfaces the support page names. +# action is fetched, no command is run and no expression is evaluated. +# +# Shell is not parsed (#823 review cycle 4). A `run:` is read only when it is one +# line of plain words — no quote, expansion, operator, redirection, comment or +# continuation, so every POSIX shell splits it at its blanks and nowhere else — +# whose program is a known agent CLI; `claude_args` and `codex-args` are read +# only when they are such a list of words, which each action's own splitter reads +# the same way. Every other form is never guessed at: a `run:` that mentions an +# agent CLI is named as an unread step, and an argument input as an unread +# setting, each a non-blocking limit that publishes none of its text. @dataclass(frozen=True) @@ -1624,8 +1627,8 @@ class _AgentAction: ``inputs`` are compared as text. ``gates`` are comma-separated user lists where a ``*`` entry opens the gate to every user (to every bot, for ``allowed_bots``). ``args`` names the input - that carries agent CLI arguments, read with the family's flag table for - the documented widening rules. ``modes`` are ``(input, value, rule)``: an + that carries agent CLI arguments, read, when it is a plain list of words, + for the documented widening rules. ``modes`` are ``(input, value, rule)``: an input whose value is itself a documented widening. ``settings`` names the input that holds Claude Code settings, JSON or a path; written as JSON, its ``defaultMode`` is read as the settings reader reads it (#823 review). @@ -1691,6 +1694,8 @@ class _AgentAction: #: Flag spelling -> (primary spelling, arity): ``0`` takes no value, ``1`` one, #: ``None`` every following word up to the next one starting with ``-``, as the #: CLI's variadic options read them. From Claude Code's CLI reference. +#: ``--settings`` and ``--mcp-config`` are not listed: a value passing one is +#: not read at all (:data:`_UNREAD_ARGUMENT_FLAGS`). _CLAUDE_FLAGS: dict[str, tuple[str, int | None]] = { "--permission-mode": ("--permission-mode", 1), "--dangerously-skip-permissions": ("--dangerously-skip-permissions", 0), @@ -1700,8 +1705,6 @@ class _AgentAction: "--disallowedTools": ("--disallowedTools", None), "--disallowed-tools": ("--disallowedTools", None), "--add-dir": ("--add-dir", None), - "--mcp-config": ("--mcp-config", None), - "--settings": ("--settings", 1), "--permission-prompt-tool": ("--permission-prompt-tool", 1), } @@ -1769,7 +1772,6 @@ def agent_rule_text(rule: str, detail: str) -> str: UNTRUSTED_INPUT_TRIGGERS = ("issue_comment", "issues", "pull_request_target", "workflow_run") _ASSIGNMENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*=") -_SHELL_OPERATOR_CHARS = ";&|()<>\n" _SECRET_REFERENCE_RE = re.compile(r"\bsecrets\.([A-Za-z_][A-Za-z0-9_]*)") _EXPRESSION_RE = re.compile(r"\$\{\{(.*?)\}\}", re.S) @@ -1780,437 +1782,119 @@ def pull_request_code_ref(ref: str | None) -> bool: return ref is not None and _PULL_REQUEST_CODE_REF_RE.fullmatch(ref.strip()) is not None -def _shell_words(text: str) -> list[str] | None: - """``text`` split into shell words and operator tokens, or ``None`` when unbalanced. - - Newlines, ``;``, ``&``, ``|``, parentheses and redirections outside quotes - are operator tokens of their own; a backslash-newline joins two lines, as - the shell reads it. - """ - - lexer = shlex.shlex( - re.sub(r"\\\r?\n", "", text), posix=True, punctuation_chars=_SHELL_OPERATOR_CHARS - ) - lexer.whitespace = " \t\r" - lexer.whitespace_split = True - lexer.commenters = "" - try: - return list(lexer) - except ValueError: +# --- the only forms read: plain lists of words (#823 review cycle 4) -------------------- +# +# Four review cycles each found a shell form a parser mis-read, so none is +# parsed. A word made only of these characters means the same to a POSIX shell, +# to shell-quote (which the Claude actions split `claude_args` with, after making +# `()|&;<>` literal) and to string-argv (which `openai/codex-action` splits +# `codex-args` with): no quote, `$`, backtick, backslash, `#`, `~`, glob, +# brace, bracket, `!`, `@`, `;`, `&`, `|`, `<`, `>` or non-ASCII character is +# among them, so the words are exactly the text split at its blanks. + +#: A word of a plain `run:` command. +_PLAIN_RUN_WORD_RE = re.compile(r"[A-Za-z0-9_./:=,%+-]+") +#: A word of a plain argument input: parentheses too, which neither action's +#: splitter reads as anything but a character (`Bash(git:status)`), while a +#: shell would. +_PLAIN_ARGUMENT_WORD_RE = re.compile(r"[A-Za-z0-9_./:=,%+()-]+") +#: Flags whose value the host readers withhold or read from a file: a value +#: passing one, in any spelling, is not read (#823 review cycle 4). +_UNREAD_ARGUMENT_FLAGS = frozenset({"--settings", "--mcp-config"}) +#: A known agent CLI's name as a word of its own: a `run:` that holds one +#: mentions that agent. +_AGENT_CLI_MENTION_RE = re.compile(r"(? list[str] | None: + """``words``, when each is made of plain characters and none is an unread flag, else ``None``.""" + + words = [word for word in words if word] + if not words or not all(pattern.fullmatch(word) for word in words): return None + if any(word.split("=", 1)[0] in _UNREAD_ARGUMENT_FLAGS for word in words): + return None + return words -def _is_operator(token: str) -> bool: - return bool(token) and all(char in _SHELL_OPERATOR_CHARS for char in token) - - -def _has_shell_expansion(text: str) -> bool: - """A ``$`` or backtick the shell would expand: anywhere outside single quotes.""" - - quote: str | None = None - escaped = False - for char in text: - if escaped: - escaped = False - elif char == "\\" and quote != "'": - escaped = True - elif quote == "'": - quote = None if char == "'" else quote - elif char in {"$", "`"}: - return True - elif char == quote: - quote = None - elif quote is None and char in {"'", '"'}: - quote = char - return False - - -def _has_shell_comment(text: str) -> bool: - """An unquoted ``#`` that starts a word: the shell reads the rest of the line as a comment. - - Read off the raw text because the word splitter keeps ``#`` as an - ordinary character, so a quoted ``"#123 review"`` is one word, not a comment. - """ - - quote: str | None = None - escaped = False - previous = " " - for char in text: - if escaped: - # An escaped character, a space included, is part of the word. - escaped = False - previous = "\\" - continue - if char == "\\" and quote != "'": - escaped = True - elif quote is not None: - quote = None if char == quote else quote - elif char in {"'", '"'}: - quote = char - elif char == "#" and (previous.isspace() or previous in _SHELL_OPERATOR_CHARS): - return True - previous = char - return False - - -@dataclass -class _SubstitutionLevel: - """One open level of :func:`_command_substitutions`: the top of the text, or one substitution.""" - - #: The character that ends it: `)` for `$(`, a backtick for a backtick, none at the top. - closer: str - #: Where its text starts. - start: int - #: ``$((…))`` is arithmetic, not a command. - arithmetic: bool = False - quoted: bool = False - #: Parentheses opened inside a `$(…)` and not yet closed. - depth: int = 0 - #: Its text so far, each nested substitution in it replaced by `_`. - parts: list[str] = field(default_factory=list) - #: Where its text not yet in ``parts`` starts. - cursor: int = 0 - - -def _command_substitutions(text: str, *, here_doc_body: bool = False) -> list[str]: - """The text of each ``$(…)`` and backtick command substitution the shell would run in ``text``. - - Found outside single quotes, inside double quotes too: the word splitter - reads a double-quoted ``"$(claude -p …)"`` as one word and a backtick as - no operator, so the command inside either never heads a command it splits - (#823 review). A nested substitution is listed on its own and reads as - ``_`` in the text around it, so each character is listed once and no - nesting depth costs more than the text's length. An escaped ``\\$`` or - backtick, a comment and the ``$((…))`` of arithmetic are not - substitutions, though one inside arithmetic still is, and one left open - runs to the end of the text. Only used to name an agent CLI as - unresolved: nothing found here is read or published. - - ``here_doc_body`` reads ``text`` as the body of a here-document whose - delimiter is unquoted (:func:`_here_documents`): the shell runs its - substitutions, but quotes and ``#`` in it are ordinary characters, so - ``'$(claude -p …)'`` there is a substitution. Inside a substitution the - usual quoting applies again. - """ - - bodies: list[str] = [] - levels = [_SubstitutionLevel(closer="", start=0)] - - def open_level(opener: int, closer: str, start: int, arithmetic: bool = False) -> None: - parent = levels[-1] - parent.parts += [text[parent.cursor:opener], "_"] - levels.append(_SubstitutionLevel(closer=closer, start=start, arithmetic=arithmetic, cursor=start)) - - def close_level(end: int) -> None: - level = levels.pop() - if not level.arithmetic: - bodies.append("".join([*level.parts, text[level.cursor:end]])) - levels[-1].cursor = min(end + 1, len(text)) - - index = 0 - while index < len(text): - level = levels[-1] - char = text[index] - if char == "\\": - index += 2 - continue - if char == "`": - if level.closer == "`": - close_level(index) - else: - open_level(index, "`", index + 1) - index += 1 - continue - if char == "$" and text.startswith("(", index + 1): - open_level(index, ")", index + 2, arithmetic=text.startswith("((", index + 1)) - index += 2 - continue - if here_doc_body and len(levels) == 1: - # Outside a substitution, a here-document body quotes nothing. - pass - elif level.quoted: - level.quoted = char != '"' - elif char == '"': - level.quoted = True - elif char == "'": - quote_end = text.find("'", index + 1) - index = len(text) if quote_end < 0 else quote_end + 1 - continue - elif char == "#" and ( - index == level.start or text[index - 1].isspace() or text[index - 1] in _SHELL_OPERATOR_CHARS - ): - # A comment runs to the end of its line, or inside backticks to - # the backtick that closes them. - ends = [text.find("\n", index), text.find("`", index) if level.closer == "`" else -1] - index = min((end for end in ends if end >= 0), default=len(text)) - continue - elif level.closer == ")" and char == "(": - level.depth += 1 - elif level.closer == ")" and char == ")": - if level.depth: - level.depth -= 1 - else: - close_level(index) - index += 1 - while len(levels) > 1: - close_level(len(text)) - return bodies - - -def _here_doc_delimiter(text: str, index: int) -> tuple[str, bool, int]: - """The here-document delimiter word at ``index``: its text after quote removal, whether any of it is quoted, and where it ends. +def _argument_words(text: str) -> list[str] | None: + """An agent action's argument input as its words, when it is a plain list of them. - An unclosed quote gives an empty delimiter, which is not read as one. + Blanks and newlines separate words, as both actions' splitters read + them. ``None`` — the input is not read — for anything else: a quote, a + ``${{ }}`` expression, ``$``, a backtick, a comment, a shell separator, + JSON, a ``--settings`` or ``--mcp-config`` flag, or any other character + outside the plain set. """ - word: list[str] = [] - quoted = False - while index < len(text) and not text[index].isspace() and text[index] not in _SHELL_OPERATOR_CHARS: - char = text[index] - if char == "\\": - quoted = True - word.append(text[index + 1:index + 2]) - index += 2 - elif char in {"'", '"'}: - end = text.find(char, index + 1) - if end < 0: - return "", quoted, len(text) - quoted = True - word.append(text[index + 1:end]) - index = end + 1 - else: - word.append(char) - index += 1 - return "".join(word), quoted, index + return _plain_words(re.split(r"[ \t\r\n]+", text), _PLAIN_ARGUMENT_WORD_RE) -@dataclass -class _LineStarts: - """Where each line of a text starts, by its content and by its content after leading tabs.""" - - exact: dict[str, list[int]] - tab_stripped: dict[str, list[int]] - - @classmethod - def of(cls, text: str) -> _LineStarts: - exact: dict[str, list[int]] = {} - tab_stripped: dict[str, list[int]] = {} - start = 0 - for line in text.split("\n"): - exact.setdefault(line, []).append(start) - tab_stripped.setdefault(line.lstrip("\t"), []).append(start) - start += len(line) + 1 - return cls(exact=exact, tab_stripped=tab_stripped) - - -def _here_doc_bodies_end( - text: str, lines: _LineStarts, start: int, pending: list[tuple[str, bool, bool]], -) -> tuple[int, list[str]] | None: - """Where the bodies of ``pending`` here-documents, read in order from ``start``, end, and the bodies the shell expands. - - Each body runs to the first line that is exactly its delimiter (after - leading tabs, for ``<<-``). ``None`` when one has no such line. The line - is looked up rather than searched for, so a text of many here-documents - with no closing line costs no more than its length times a logarithm. - """ +def _run_words(run: str) -> list[str] | None: + """A ``run:`` as the words of its one command, when it is one line of plain words. - position = start - expanded: list[str] = [] - for delimiter, quoted, strip_tabs in pending: - starts = (lines.tab_stripped if strip_tabs else lines.exact).get(delimiter, []) - found = bisect.bisect_left(starts, position) - if found == len(starts): - return None - closing = starts[found] - if not quoted: - expanded.append(text[position:closing]) - line_end = text.find("\n", closing) - position = len(text) if line_end < 0 else line_end + 1 - return position, expanded - - -def _here_documents(text: str) -> tuple[str, list[str]]: - """``text`` without the body and closing line of each here-document, and the bodies the shell expands. - - The shell never runs a here-document's lines as commands: it passes them - to the command as input. When any part of the delimiter is quoted - (``<<'EOF'``, ``<<"EOF"``, ``<<\\EOF``) the body is passed as written, so - ``claude -p`` in Markdown backticks there starts nothing (#823 review); - when it is unquoted (``<= 0), default=len(text)) - continue - elif char == "<" and text.startswith("<<", index) and not level.arithmetic: - if text.startswith("<<<", index): - index += 3 - continue - index += 2 - strip_tabs = text.startswith("-", index) - if strip_tabs: - index += 1 - while index < len(text) and text[index] in " \t": - index += 1 - delimiter, quoted, index = _here_doc_delimiter(text, index) - if delimiter: - pending.append((delimiter, quoted, strip_tabs)) - continue - elif level.closer == ")" and char == "(": - level.depth += 1 - elif level.closer == ")" and char == ")": - if level.depth: - level.depth -= 1 - else: - levels.pop() - index += 1 - kept.append(text[cursor:]) - return "".join(kept), expanded - - -def _commands(words: list[str]) -> list[list[str]]: - """``words`` grouped into the commands their operator tokens separate, empty ones included.""" - - commands: list[list[str]] = [[]] - for word in words: - if _is_operator(word): - commands.append([]) - else: - commands[-1].append(word) - return commands - - -def _substituted_agents(run: str, *, here_doc_body: bool = False) -> set[str]: - """The agent CLIs ``run`` starts at the head of a command inside a command substitution.""" - - agents: set[str] = set() - for body in _command_substitutions(run, here_doc_body=here_doc_body): - words = _shell_words(body) - if words is None: - agents.update(agent for line in body.splitlines() if (agent := _line_agent(line))) - continue - agents.update(launch[0] for command in _commands(words) if (launch := _agent_command(command))) - return agents - - -#: Shell reserved words that can come before a command's name: a pipeline's -#: ``!`` and ``time`` (with its ``-p``), and the words that open a list inside -#: a compound command, as in ``if …; then claude -p …; fi`` (#823 review). A -#: command after one is not a simple command, so an agent CLI there is named -#: as unresolved and never read. -_RESERVED_PREFIXES = frozenset({"!", "time", "{", "if", "then", "elif", "else", "while", "until", "do"}) + text = run.strip(" \t\n") + if "\n" in text: + return None + return _plain_words(re.split(r"[ \t]+", text), _PLAIN_RUN_WORD_RE) -def _reserved_prefix_length(words: list[str]) -> int: - """How many leading words of one command are shell reserved words, a ``function NAME`` included. +def _run_command(run: str) -> tuple[str, list[str]] | None: + """The agent CLI a plain one-command ``run:`` launches headless, and its arguments. - Reserved words are read only before a command's assignments and name, so - ``FOO=1 then`` runs a command named ``then``. + After any ``NAME=value`` assignments, the program's file name must be + ``claude`` with ``-p``/``--print`` among its arguments, or ``codex`` with + ``exec`` (or its alias ``e``) as its first argument. Any other command — + ``claude mcp add``, ``codex login``, ``npx …``, ``timeout 600 claude`` — + launches none this reader reads. """ + words = _run_words(run) + if words is None: + return None index = 0 - while index < len(words): - if words[index] in _RESERVED_PREFIXES: - index += 2 if words[index] == "time" and words[index + 1:index + 2] == ["-p"] else 1 - elif words[index] == "function": - # `function NAME { …; }`: the name is not a command. - index += 2 - else: - break - return min(index, len(words)) - - -def _agent_command(words: list[str]) -> tuple[str, list[str]] | None: - """The agent CLI one command launches headless, and its arguments. - - ``claude`` with ``-p``/``--print`` anywhere in its arguments, or ``codex`` - with ``exec`` (or its alias ``e``) as its first argument, after any leading - shell reserved words (``then``, ``do``, ``{``, ``!``, ``time``, a - ``function NAME``) and then any ``NAME=value`` assignments, the order the - shell reads them in. Any other command — ``claude mcp add``, - ``codex login``, an ``echo`` that mentions either — launches none. - """ - - index = _reserved_prefix_length(words) while index < len(words) and _ASSIGNMENT_RE.match(words[index]): index += 1 if index >= len(words): return None - command, arguments = words[index], words[index + 1:] - if command == "claude" and any(word in {"-p", "--print"} for word in arguments): + program, arguments = words[index].rsplit("/", 1)[-1], words[index + 1:] + if program == "claude" and any(word in {"-p", "--print"} for word in arguments): return "claude", arguments - if command == "codex" and arguments[:1] in (["exec"], ["e"]): + if program == "codex" and arguments[:1] in (["exec"], ["e"]): return "codex", arguments[1:] return None -def _line_agent(line: str) -> str | None: - """The agent CLI one line starts headless, read as plain words: a fallback only. +def _declared_shell(step: dict[Any, Any], job: dict[Any, Any], workflow: dict[Any, Any]) -> Any: + """The ``shell:`` a step runs its ``run:`` with: its own, else its job's, else its workflow's default. - Used when a ``run:``'s quoting cannot be split into commands, to name the - launch as unresolved rather than miss it; nothing it reads is published. + ``None`` when none is declared, which is the runner's default shell. """ - return (_agent_command(line.split()) or (None, None))[0] + if step.get("shell") is not None: + return step["shell"] + for container in (job, workflow): + defaults = container.get("defaults") + run = defaults.get("run") if isinstance(defaults, dict) else None + if isinstance(run, dict) and run.get("shell") is not None: + return run["shell"] + return None + + +def _read_shell(shell: Any) -> bool: + """Whether a ``run:`` under this declared shell is read: none declared, ``bash`` or ``sh``.""" + + if shell is None: + return True + if not isinstance(shell, str) or not shell.split(): + return False + return shell.split()[0].rsplit("/", 1)[-1] in _READ_SHELLS def _read_flags( @@ -2256,200 +1940,11 @@ def _read_flags( return flags -# --- how an agent action splits its argument input (#823) ----------------------------- -# -# `claude_args` and `codex-args` are not shell text: no shell reads them. Each -# action splits its input itself, and a widening rule is read from the words -# the action passes on, so these follow the actions' own parsers rather than -# the `run:` tokenizer above. - -#: One word of shell-quote's chunker once the Claude actions have made -#: ``()|&;<>`` literal: unquoted non-space characters (a backslash escaping a -#: quote or a blank), a double-quoted run or a single-quoted run, adjacent. -#: An unbalanced quote matches none of them, so shell-quote skips it. -_SHELL_QUOTE_CHUNK_RE = re.compile(r"""(?:(?:\\['" \t]|[^\s'"])+|"(?:\\"|[^"])*?"|'[^']*?')+""") - -#: string-argv's pattern, which `openai/codex-action` splits a shell-like -#: `codex-args` with: a word holding quotes keeps them, a quoted string alone -#: is its content, and whitespace, newlines included, separates the rest. -_STRING_ARGV_RE = re.compile( - r"""([^\s'"]([^\s'"]*(['"])([^\x03]*?)\3)+[^\s'"]*)|[^\s'"]+|(['"])([^\x03]*?)\5""" -) - - -def _shell_quote_variable(chunk: str, index: int) -> tuple[str, int]: - """shell-quote's ``parseEnvVar`` with no environment, at the ``$`` at ``index``. - - Returns the value, empty for any name and ``$`` for none, and the index of - the last character the variable took. Raises ``ValueError`` for the "Bad - substitution" shell-quote throws, which fails the action. - """ - - index += 1 - char = chunk[index:index + 1] - if char == "{": - index += 1 - if chunk[index:index + 1] == "}": - raise ValueError("bad substitution") - depth, end = 1, index - while depth > 0 and end < len(chunk): - if chunk[end] == "{" and chunk[end - 1] == "$": - depth += 1 - elif chunk[end] == "}": - depth -= 1 - end += 1 - if depth != 0: - raise ValueError("bad substitution") - name, index = chunk[index:end - 1], end - 1 - elif char and char in "*@#?$!_-": - # shell-quote steps past the name and then past one more character. - name, index = char, index + 1 - else: - match = re.search(r"[^A-Za-z0-9_]", chunk[index:]) - if match is None: - name, index = chunk[index:], len(chunk) - else: - name, index = chunk[index:index + match.start()], index + match.start() - 1 - return ("" if name else "$"), index - - -def _shell_quote_word(chunk: str, *, comments: bool) -> tuple[str, bool]: - """One chunk as shell-quote reads it: the word, and whether an unquoted ``#`` ended the input. - - Quotes and backslashes work as in a shell and ``$NAME`` reads as empty. - With ``comments``, an unquoted ``#`` ends the word and every word after - it, as it does for the action; without, ``#`` is an ordinary character. - """ - - out: list[str] = [] - quote = "" - escaped = False - index = 0 - while index < len(chunk): - char = chunk[index] - if escaped: - out.append(char) - escaped = False - elif quote: - if char == quote: - quote = "" - elif quote == "'": - out.append(char) - elif char == "\\": - index += 1 - following = chunk[index:index + 1] - out.append(following if following and following in "\"\\$" else "\\" + following) - elif char == "$": - value, index = _shell_quote_variable(chunk, index) - out.append(value) - else: - out.append(char) - elif char in {'"', "'"}: - quote = char - elif char == "#" and comments: - return "".join(out), True - elif char == "\\": - escaped = True - elif char == "$": - value, index = _shell_quote_variable(chunk, index) - out.append(value) - else: - out.append(char) - index += 1 - return "".join(out), False - - -@dataclass(frozen=True) -class _ArgumentInput: - """An agent action's argument input as the action splits it (#823). - - ``text`` is what the action parses. ``spans`` is every word with where it - sits in ``text``, for publication. ``words`` is what the action passes on, - or ``None`` when the action refuses the input and the agent does not run. - """ - - text: str - spans: tuple[tuple[str, int, int], ...] - words: tuple[str, ...] | None - - -def _claude_argument_input(value: str) -> _ArgumentInput: - """``claude_args`` split as ``base-action/src/parse-sdk-options.ts`` splits it. - - Each line whose first non-blank character is ``#`` is dropped, ``()|&;<>`` - are literal, and the rest is read by shell-quote with no environment: - whitespace, newlines included, separates words; quotes and backslashes work - as in a shell; ``$NAME`` reads as empty; and an unquoted ``#`` later in the - input ends it. - """ - - text = "\n".join( - line for line in value.split("\n") if not line.strip().startswith("#") - ).strip() - spans: list[tuple[str, int, int]] = [] - words: list[str] | None = [] - ended = False - for match in _SHELL_QUOTE_CHUNK_RE.finditer(text): - chunk = match.group() - if words is not None and not ended: - try: - word, ended = _shell_quote_word(chunk, comments=True) - except ValueError: - words = None - else: - if word or not ended: - words.append(word) - try: - shown = _shell_quote_word(chunk, comments=False)[0] - except ValueError: - shown = chunk - spans.append((shown, match.start(), match.end())) - return _ArgumentInput(text, tuple(spans), None if words is None else tuple(words)) - - -def _codex_argument_input(value: str) -> _ArgumentInput: - """``codex-args`` read as `openai/codex-action` reads it: a JSON array of strings, or string-argv. - - A value starting with ``[`` that is not a JSON array of strings makes the - action refuse it. The array form has no spans: it is published whole. - """ - - if value.startswith("["): - try: - loaded = json.loads(value) - except (ValueError, RecursionError): - return _ArgumentInput(value, (), None) - if isinstance(loaded, list) and all(isinstance(item, str) for item in loaded): - return _ArgumentInput(value, (), tuple(loaded)) - return _ArgumentInput(value, (), None) - spans = tuple( - ( - next(group for group in (match.group(1), match.group(6), match.group(0)) if group is not None), - match.start(), - match.end(), - ) - for match in _STRING_ARGV_RE.finditer(value) - ) - return _ArgumentInput(value, spans, tuple(word for word, _start, _end in spans)) - - -def _argument_input(family: str, value: str) -> _ArgumentInput: - return _claude_argument_input(value) if family == "claude" else _codex_argument_input(value) - - -# --- what an expression in an input leaves readable (#823 review) --------------------- -# -# GitHub substitutes a `${{ }}` expression into an input before the action reads -# it, and the substituted text may be anything: more words, a quote that closes -# one opened before it, a `#` that ends `claude_args`. So a rule is read only -# from literal text the expression cannot reach. - #: One ``${{ … }}`` expression, or an unterminated ``${{`` to the end of the text. _EXPRESSION_SPAN_RE = re.compile(r"\$\{\{.*?(?:\}\}|\Z)", re.S) -#: What an expression reads as while literal text around it is split: a -#: private-use character, which no splitter here reads as a blank, a quote or -#: an operator, so the word the expression touches holds it and is set aside. -_EXPRESSION_MARK = "\ue000" +#: What an expression reads as while a user gate's entries are split: a +#: private-use character, which is neither a comma nor a blank. +_EXPRESSION_MARK = "" def holds_expression(text: str) -> bool: @@ -2458,62 +1953,6 @@ def holds_expression(text: str) -> bool: return "${{" in text -def _skipped_quote(text: str, pattern: re.Pattern[str]) -> int | None: - """The first quote ``pattern`` leaves unmatched in ``text``, else ``None``. - - Both splitters skip a quote no word covers; before an expression, that is a - quoted run the substituted text may close, so nothing from it on is read. - """ - - covered = 0 - for match in (*pattern.finditer(text), None): - gap = text[covered:] if match is None else text[covered:match.start()] - quote = next((index for index, char in enumerate(gap) if char in "'\""), None) - if quote is not None: - return covered + quote - if match is not None: - covered = match.end() - return None - - -def _literal_argument_words(family: str, text: str) -> tuple[str, ...] | None: - """The words of an argument input no ``${{ }}`` expression in it can reach, as the action splits them. - - Without an expression, every word the action passes on. With one, the - words the action has finished reading before the first expression: the - word the expression touches, any quoted run still open at it, and - everything after it are not read. A JSON array ``codex-args`` gives the - elements before the one holding an expression. ``None`` when the action - refuses what is left. - """ - - start = text.find("${{") - if start == -1: - return _argument_input(family, text).words - if family == "codex" and text.startswith("["): - words = _argument_input(family, _EXPRESSION_SPAN_RE.sub(_EXPRESSION_MARK, text)).words - else: - prefix, pattern = text[:start], _STRING_ARGV_RE - if family == "claude": - # A line the action drops as a comment is dropped whatever the - # expression on it holds, and the lines before it are whole. - *whole, last = prefix.split("\n") - dropped = last.strip().startswith("#") - kept = [line for line in whole if not line.strip().startswith("#")] - prefix = "\n".join([*kept, ""] if dropped else [*kept, last]) - pattern = _SHELL_QUOTE_CHUNK_RE - quote = _skipped_quote(prefix, pattern) - words = _argument_input(family, prefix[:quote] + _EXPRESSION_MARK).words - if words is None: - return None - literal: list[str] = [] - for word in words: - if _EXPRESSION_MARK in word: - break - literal.append(word) - return tuple(literal) - - # --- what an agent setting publishes (#823, #802) -------------------------------------- #: A codex ``--config`` override under one of these keys carries values the @@ -2618,16 +2057,15 @@ def _json_shape(value: Any, path: tuple[str, ...] = ()) -> Any: return value -def _withheld_json(value: Any, path: tuple[str, ...] = ()) -> str | None: +def _withheld_json(value: Any) -> str | None: """A structured value as it may be published: its :func:`_json_shape`, as canonical JSON. - Canonical, so reformatting or reordering keys changes nothing. ``path`` is - the keys above ``value``, for a codex ``--config`` override's value. + Canonical, so reformatting or reordering keys changes nothing. """ try: return json.dumps( - _json_shape(value, path), sort_keys=True, separators=(",", ":"), + _json_shape(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False, default=str, ) except (RecursionError, TypeError, ValueError): @@ -2650,102 +2088,71 @@ def _withheld_word(word: str) -> str | None: return _withheld_json(loaded) -def _withheld_config(text: str) -> str | None: - """A codex ``--config key=value`` override, its secret-bearing value withheld. +#: The codex ``--config`` keys whose value is published as written: the +#: settings a documented widening rule reads (``sandbox_mode``, +#: ``default_permissions``), the approval policy and the model. Any other +#: key's value — an MCP server's command or URL, a shell environment value, a +#: provider — publishes only its digest (#823 review cycle 4). +_PUBLISHED_CONFIG_KEYS = frozenset({"sandbox_mode", "default_permissions", "approval_policy", "model"}) + - A key path through ``env``, ``headers`` or a secret-named key publishes - ```` for its value; a table or array value, parsed as TOML as - codex parses it, publishes its shape by :func:`_withheld_json`, read under - its key path, so ``mcp_servers.gh={command="gh", …}`` is a server whose - command name is kept. A value that starts like a table or array and does - not parse — string-argv keeps the quotes of ``--config='k={…}'`` — is - ``None``: what it holds cannot be told apart from its keys. Anything else, - a scalar value such as ``model="o3"``, is argument text and is kept. +def _withheld_config(text: str) -> str: + """A codex ``--config key=value`` override as it may be published. + + The key is published. Under ``env``, ``headers`` or a secret-named key the + value is ````, as the host readers redact such values, so + rotating it is quiet; under one of :data:`_PUBLISHED_CONFIG_KEYS` it is + published as written; under any other key it is ````, a + digest, so editing it is still a change while none of its text is + published. Only a plain word reaches here, so the value is never a TOML + table or array. """ key, equals, value = text.partition("=") if not equals: return text - segments = [segment.strip().strip("\"'") for segment in key.split(".")] + segments = [segment.strip() for segment in key.split(".")] if any(segment in _CONFIG_WITHHELD_KEYS or _is_secret_key(segment) for segment in segments): return f"{key}=" - try: - loaded = tomllib.loads(f"value = {value}").get("value") - except (tomllib.TOMLDecodeError, RecursionError): - return None if value.strip().strip("\"'").startswith(("{", "[")) else text - if isinstance(loaded, (dict, list)): - shown = _withheld_json(loaded, tuple(segments)) - return None if shown is None else f"{key}={shown}" - return text - - -def _withheld_attached(prefix: str, value: str, withhold: Callable[[str], str | None]) -> str | None: - """``prefix`` and a value written attached to its flag, the value withheld by ``withhold``.""" - - shown = withhold(value) - return None if shown is None else f"{prefix}{shown}" + if key.strip() in _PUBLISHED_CONFIG_KEYS: + return text + return f"{key}={_withheld_string(value)}" -def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> list[str] | None: - """Each argument word as it may be published, or ``None`` when one cannot be. +def _withheld_words(words: list[str] | tuple[str, ...], *, family: str) -> tuple[list[str], bool]: + """Each plain argument word as it may be published, and whether one was redacted as credential-shaped. - A value is withheld however it is attached to its flag (#823 review): a - separate word, ``--name=value`` (``--settings={…}``, ``--mcp-config={…}``, - ``--config=…``), and codex's ``-c`` and ``-c=``, which clap - reads as ``-c ``. + A codex ``--config`` override is withheld however it is attached to its + flag: a separate word after ``-c`` or ``--config``, ``--config=…``, and + ``-c`` or ``-c=``, which clap reads as ``-c ``. The + word after a secret-named word such as ``--token`` or ``password`` is + ````, as the host readers redact it in an MCP server's + arguments, and the value is then credential-shaped: an edit to that word + is not reported, so the caller names it as a limit. Every other word is + published as written, through the label redaction. """ shown: list[str] = [] config = False + secret = False + redacted = False for word in words: - if config: + if secret: + item = "" + redacted = True + elif config: item = _withheld_config(word) elif family == "codex" and word.startswith("--config="): - item = _withheld_attached("--config=", word.removeprefix("--config="), _withheld_config) + item = "--config=" + _withheld_config(word.removeprefix("--config=")) elif family == "codex" and word.startswith("-c") and len(word) > 2: prefix = "-c=" if word.startswith("-c=") else "-c" - item = _withheld_attached(prefix, word.removeprefix(prefix), _withheld_config) - elif word.startswith("--") and "=" in word: - name, _, value = word.partition("=") - item = _withheld_attached(f"{name}=", value, _withheld_word) + item = prefix + _withheld_config(word.removeprefix(prefix)) else: - item = _withheld_word(word) - if item is None: - return None + item = word shown.append(item) config = family == "codex" and word in {"-c", "--config"} - return shown - - -def _withheld_arguments(family: str, value: str) -> str | None: - """An argument input as it may be published: the text the action parses, JSON words withheld. - - A JSON word, or a codex ``--config`` override, is replaced, quoted, by its - shape (:func:`_withheld_json`); every other character stays as declared. - A JSON array ``codex-args`` publishes as its array of withheld words. - """ - - parsed = _argument_input(family, value) - if family == "codex" and value.startswith("["): - if parsed.words is None: - # The action refuses it; what it holds is still withheld as JSON. - try: - return _withheld_json(json.loads(value)) - except (ValueError, RecursionError): - return None - shown = _withheld_words(parsed.words, family=family) - return None if shown is None else json.dumps(shown, separators=(",", ":"), ensure_ascii=False) - shown = _withheld_words([word for word, _start, _end in parsed.spans], family=family) - if shown is None: - return None - pieces: list[str] = [] - cursor = 0 - for (word, start, end), published in zip(parsed.spans, shown, strict=True): - if published != word: - pieces.extend((parsed.text[cursor:start], shlex.quote(published))) - cursor = end - pieces.append(parsed.text[cursor:]) - return "".join(pieces) + secret = not secret and word.lower().lstrip("-").replace("-", "_") in _SECRET_KEY_MARKERS + return shown, redacted def _url_withheld(url: str) -> str: @@ -2789,9 +2196,8 @@ def _published_value(text: str) -> tuple[str, bool]: # No free character to stand for each expression: read the text as written. expressions = [] def url_withheld(match: re.Match[str]) -> str: - # A backslash the URL ends at is kept, so the `\"` a JSON-array - # `codex-args` element escapes a quote with stays an escape - # (#823 review cycle 3). + # A backslash the URL ends at is kept, so a JSON escape after it, as + # in a published permission rule, stays an escape (#823 review cycle 3). url = match.group(0) bare = url.rstrip("\\") return _url_withheld(bare) + url[len(bare):] @@ -2828,46 +2234,54 @@ def _setting_text(value: Any) -> str | None: return None -def _published_text(name: str, text: str | None) -> dict[str, Any]: - """A setting's withheld text as it may be published; ``None`` text is ``unparsed_json``.""" +def _published_text(name: str, text: str | None, *, redacted: bool = False) -> dict[str, Any]: + """A setting's withheld text as it may be published; ``None`` text is ``unparsed_json``. + + ``redacted`` says credential-shaped text was already withheld from it. + """ if text is None: return {"name": name, "value": None, "unresolved_reason": "unparsed_json"} - shown, redacted = _published_value(text) - return {"name": name, "value": shown, "unresolved_reason": "redacted" if redacted else None} + shown, rewritten = _published_value(text) + return {"name": name, "value": shown, "unresolved_reason": "redacted" if redacted or rewritten else None} def _published_setting(name: str, value: Any, *, arguments: str | None = None) -> dict[str, Any]: """One action input as it may be published: its text, or why it is not. ``arguments`` names the family whose action splits this input - (``claude_args``, ``codex-args``); every other input is one value, a JSON + (``claude_args``, ``codex-args``). Such an input is read only when it is a + plain list of words (:func:`_argument_words`), published as those words; + otherwise it is ``unread_arguments`` and publishes only ````, + a digest, so an edit to it is still a change while none of its text is + published (#823 review cycle 4). Every other input is one value, a JSON object published by :func:`_withheld_json`. """ text = _setting_text(value) if text is None: return {"name": name, "value": None, "unresolved_reason": "not_a_string"} - setting = _published_text( - name, _withheld_word(text) if arguments is None else _withheld_arguments(arguments, text) - ) + if arguments is not None: + words = _argument_words(text) + if words is None: + return {"name": name, "value": _withheld_string(text), "unresolved_reason": "unread_arguments"} + shown, redacted = _withheld_words(words, family=arguments) + return _published_text(name, " ".join(shown), redacted=redacted) + setting = _published_text(name, _withheld_word(text)) # Read off the declared text: redaction may rewrite the expression away. return {**setting, "holds_expression": True} if holds_expression(text) else setting def _published_flag(family: str, name: str, arity: int | None, values: list[str] | None) -> dict[str, Any]: - """One CLI flag as it may be published: its value words, each withheld as an input's are.""" + """One CLI flag of a plain ``run:`` as it may be published: its value words, a ``--config`` override withheld.""" if values is None: return {"name": name, "value": None, "unresolved_reason": None} if family == "codex" and name == "--config": - overrides = [_withheld_config(value) for value in values] - shown = None if None in overrides else [str(value) for value in overrides] + shown, redacted = [_withheld_config(value) for value in values], False else: - shown = _withheld_words(values, family=family) - if shown is None: - return _published_text(name, None) - return _published_text(name, shown[0] if arity == 1 else shlex.join(shown)) + shown, redacted = _withheld_words(values, family=family) + return _published_text(name, shown[0] if arity == 1 else " ".join(shown), redacted=redacted) def _setting_key(setting: dict[str, Any]) -> tuple[str, str, str]: @@ -2946,97 +2360,52 @@ def _action_launch(job: str, step_label: str, agent: str, step: dict[Any, Any]) ) -def _run_launches(job: str, step_label: str, run: str) -> list[dict[str, Any]]: - """The agent CLIs a ``run:`` step launches: read when literal, else unresolved. +def _run_launch(job: str, step_label: str, agent: str, arguments: list[str]) -> dict[str, Any]: + """A plain ``run:`` agent CLI command's launch: its documented permission flags, and the rules they meet.""" - Read only when the whole ``run:`` is one simple command that starts with - the agent CLI (after literal ``NAME=value`` assignments), holds no shell - expansion and no ``${{ }}`` expression: then its documented permission - flags are its settings. Otherwise an agent CLI found at the start of any - command in it — after any reserved words, and inside a ``$(…)`` or - backtick substitution outside single quotes — is one ``unresolved`` entry - with the reason, and none of its text is published. A here-document's - body is never a command, and only an unquoted delimiter's body has - substitutions the shell runs (:func:`_here_documents`). - """ - - run = run.strip() - # The commands the shell runs, without the here-document bodies it passes - # them as input; a step with one is never a simple command, since `<<` - # stays in `script`, so what is published is always read from `run`. - script, expanded = _here_documents(run) - here_doc_agents = {agent for body in expanded for agent in _substituted_agents(body, here_doc_body=True)} - words = _shell_words(script) - base = {"job": job, "step": step_label} - if words is None: - # Quoting that does not balance cannot be split into commands, so no - # word of it is read; a line that starts an agent CLI, or holds a - # substitution that does, is still named. - agents = sorted({ - agent for line in script.splitlines() - for agent in ({_line_agent(line)} | _substituted_agents(line)) - if agent is not None - } | here_doc_agents) - return [ - {**base, "agent": agent, "form": "unresolved", "unresolved_reason": "compound_command", "settings": []} - for agent in agents - ] - commands = _commands(words) - launches = [launch for command in commands if (launch := _agent_command(command))] - # An agent CLI inside a `$(…)` or backtick substitution heads no command - # the splitter sees when the substitution is double-quoted or a backtick - # (#823 review), and is named as unresolved. - substituted = _substituted_agents(script) | here_doc_agents - if not launches and not substituted: - return [] - # A comment is an unquoted `#` at the start of a word, read off the raw - # text: the splitter keeps `#` literal, so a quoted "#123 review" is one word. - # A command after a reserved word (`then`, `!`, `time`) is not a simple one. - simple = [command for command in commands if command] - single = ( - len(simple) == 1 - and not any(_is_operator(word) for word in words) - and not _has_shell_comment(script) - and not _reserved_prefix_length(simple[0]) - ) - if "${{" in run: - reason: str | None = "expression" - elif not single: - reason = "compound_command" - elif substituted or _has_shell_expansion(script): - reason = "shell_expansion" - else: - reason = None - if reason is not None: - agents = sorted({agent for agent, _arguments in launches} | substituted) - return [ - {**base, "agent": agent, "form": "unresolved", "unresolved_reason": reason, "settings": []} - for agent in agents - ] - (agent, arguments), = launches flags = _read_flags(arguments, _AGENT_FLAG_TABLES[agent]) settings = [_published_flag(agent, name, arity, values) for name, arity, values in flags] - return [_with_rules( + return _with_rules( { - **base, "agent": agent, "form": "read", "unresolved_reason": None, - "settings": sorted(settings, key=_setting_key), + "job": job, "step": step_label, "agent": agent, "form": "read", + "unresolved_reason": None, "settings": sorted(settings, key=_setting_key), }, _flag_rules(agent, flags), - )] + ) -def _step_agent_launches(job: str, step: dict[Any, Any], index: int) -> list[dict[str, Any]]: - """The agent launches one step declares: a known action, or a ``run:`` agent CLI.""" +def _step_agent_launches( + job: str, step: dict[Any, Any], index: int, shell: Any +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """The agent launches one step declares, and the unread ``run:`` agent steps it is. + + A known agent action is a launch. A ``run:`` is a launch only when it is + one line of plain words (:func:`_run_command`) whose program is a known + agent CLI, run by ``bash`` or ``sh`` or no declared ``shell:`` (``shell``). + Any other ``run:`` that mentions ``claude`` or ``codex`` as a word of its + own is one unread step per agent CLI it mentions: a non-blocking limit + that publishes none of its text, is never compared and so gives no row, + and never says that the step starts or does not start an agent + (#823 review cycle 4). + """ if "uses" in step: agent = _action_identity(step["uses"]) if agent is not None and agent in _AGENT_ACTIONS: - return [_action_launch(job, _step_label(step, index), agent, step)] - return [] + return [_action_launch(job, _step_label(step, index), agent, step)], [] + return [], [] run = step.get("run") - if isinstance(run, str) and ("claude" in run or "codex" in run): - return _run_launches(job, _step_label(step, index), run) - return [] + if not isinstance(run, str): + return [], [] + mentioned = sorted(set(_AGENT_CLI_MENTION_RE.findall(run))) + if not mentioned: + return [], [] + label = _step_label(step, index) + command = _run_command(run) if _read_shell(shell) else None + if command is None: + return [], [{"job": job, "step": label, "agent": agent} for agent in mentioned] + agent, arguments = command + return [_run_launch(job, label, agent, arguments)], [] def _checkout_ref(job: str, step: dict[Any, Any], index: int) -> dict[str, Any] | None: @@ -3112,15 +2481,13 @@ def _settings_bypass_permissions(text: str) -> bool: ) -def _claude_action_rules(words: tuple[str, ...]) -> set[str]: - """The widening rules the words a Claude action passes on meet. +def _claude_action_rules(words: list[str]) -> set[str]: + """The widening rules the plain words of a Claude action's ``claude_args`` meet. Read as ``parse-sdk-options.ts`` reads them: a word starting with ``--`` is always a flag and never another flag's value, so ``--dangerously-skip-permissions`` counts wherever it stands, and - ``--permission-mode`` and ``--settings`` take the next word unless that - starts with ``--``. A ``--settings`` value written as JSON meets the rule - its ``defaultMode`` sets. + ``--permission-mode`` takes the next word unless that starts with ``--``. """ rules: set[str] = set() @@ -3132,8 +2499,6 @@ def _claude_action_rules(words: tuple[str, ...]) -> set[str]: rules.add("bypass_permissions") elif name == "--permission-mode" and value == "bypassPermissions": rules.add("bypass_permissions") - elif name == "--settings" and _settings_bypass_permissions(value): - rules.add("bypass_permissions") return rules @@ -3213,7 +2578,6 @@ def _flag_rules( if family == "claude" and ( name == "--dangerously-skip-permissions" or (name == "--permission-mode" and value == "bypassPermissions") - or (name == "--settings" and value is not None and _settings_bypass_permissions(value)) ): rules.add(("bypass_permissions", name)) if family == "codex" and name == "--dangerously-bypass-approvals-and-sandbox": @@ -3227,12 +2591,13 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu """The documented widening rules an agent action's declared inputs meet, read from the raw text. Read here, before anything is withheld for publication, so redaction never - hides a rule. Only from literal text: GitHub substitutes a ``${{ }}`` - expression into the input before the action reads it, so a rule is read - only where the substituted text cannot reach — an argument input's words - before the first expression (:func:`_literal_argument_words`), a user - gate's entries that hold none — and a mode or settings input holding one - meets none. Each rule names the input it was read from. + hides a rule. Only from text this reader reads exactly: an argument input + only when it is a plain list of words (:func:`_argument_words`), which a + ``${{ }}`` expression never is; a user gate's entries that hold no + expression, since GitHub substitutes one before the action reads the + input and the substituted text may add entries but cannot remove a + literal one; and a mode or settings input only when it holds none. Each + rule names the input it was read from. """ rules: set[tuple[str, str]] = set() @@ -3241,7 +2606,6 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu if text is None: continue if name in spec.gates: - # The substituted text may add entries; it cannot remove a literal one. entries = _EXPRESSION_SPAN_RE.sub(_EXPRESSION_MARK, text).split(",") if "*" in {entry.strip() for entry in entries}: rules.add(("open_gate", name)) @@ -3251,7 +2615,7 @@ def _action_rules(spec: _AgentAction, declared: list[tuple[str, Any]]) -> set[tu if name == spec.settings and not holds_expression(text) and _settings_bypass_permissions(text): rules.add(("bypass_permissions", name)) if name == spec.args: - words = _literal_argument_words(spec.family, text) + words = _argument_words(text) if words is None: continue if spec.family == "claude": @@ -3312,17 +2676,23 @@ def agent_family(agent: str) -> str: } -def _expression_setting(entry: dict[str, Any], rule: str, detail: str) -> str | None: - """The input of ``entry`` holding a ``${{ }}`` expression whose substituted text may meet ``rule``.""" +def _unread_setting(entry: dict[str, Any], rule: str, detail: str) -> tuple[str, str] | None: + """The input of ``entry`` whose text this reader did not read for ``rule``, and why. + + ``expression``: it holds a ``${{ }}`` expression, whose substituted text + may meet the rule. ``unread``: it is an argument input that is not a + plain list of words (``unread_arguments``), so no rule was read from it. + """ names = frozenset({detail}) if rule == "open_gate" else _RULE_SETTINGS.get(rule, frozenset()) - return next( - ( - str(setting["name"]) for setting in entry.get("settings", []) - if setting.get("holds_expression") and setting["name"] in names - ), - None, - ) + for setting in entry.get("settings", []): + if setting["name"] not in names: + continue + if setting.get("unresolved_reason") == "unread_arguments": + return str(setting["name"]), "unread" + if setting.get("holds_expression"): + return str(setting["name"]), "expression" + return None @dataclass(frozen=True) @@ -3332,15 +2702,23 @@ class AgentRuleGains: Each widening is ``(job, rule, detail, entry)`` for the first launch at ``after`` in that job that meets it. Only ``claimed`` is a widening: - - ``unread_before``: the job launched that agent at ``before`` only in a - form this reader does not read, such as a compound ``run:`` that became a - literal one. That launch may have met the rule already, as a job whose + - ``unread_before``: the job has fewer steps at ``after`` than at + ``before`` that may launch that agent in a form this reader does not + read — a ``run:`` that mentions the agent CLI and is not one plain + command (``unread_agent_runs``), or an agent action whose ``with:`` is + not a mapping — and more launches of it this reader reads. The launch + that meets the rule may be one of those steps, rewritten in a form this + reader reads, and it may have met the rule already, as a job whose permissions were not explicit may already have held a write scope - (``unknown_before``). A job that also launched the agent in a form that - was read claims the gain. - - ``expression_before``: at ``before``, the job's launch of that agent held - a ``${{ }}`` expression in an input the rule is read from, and GitHub's - substituted text may already have met it. Each names that input. + (``unknown_before``). Each names a step that is gone. Such a step that + remains is a limit and nothing more, and one that goes while no read + launch is added takes no gain from a launch that was read before + (#823 review cycle 4). + - ``setting_before``: at ``before``, the job's launch of that agent held, + in an input the rule is read from, text this reader did not read for a + rule — a ``${{ }}`` expression (``expression``), whose substituted text + may already have met it, or an argument input that is not a plain list + of words (``unread``). Each names that input and which. - ``moved``: the rule left another job whose launch that met it left that job — the job no longer exists, or the same launch now runs here while each launch of that agent this job had still runs here or in that job — @@ -3354,8 +2732,8 @@ class AgentRuleGains: """ claimed: list[AgentWidening] - unread_before: list[AgentWidening] - expression_before: list[tuple[AgentWidening, str]] + unread_before: list[tuple[AgentWidening, dict[str, Any]]] + setting_before: list[tuple[AgentWidening, str, str]] moved: list[tuple[AgentWidening, dict[str, Any]]] @@ -3375,14 +2753,25 @@ def met(grant: dict[str, Any] | None) -> dict[_RuleKey, list[dict[str, Any]]]: found.setdefault(key, []).append(entry) return found - def launches(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[str, Any]]]: + def by_job(entries: list[dict[str, Any]]) -> dict[tuple[str, str], list[dict[str, Any]]]: found: dict[tuple[str, str], list[dict[str, Any]]] = {} - for entry in (grant or {}).get("agent_launches", []): + for entry in entries: found.setdefault((str(entry["job"]), agent_family(str(entry["agent"]))), []).append(entry) return found + def launches(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[str, Any]]]: + return by_job((grant or {}).get("agent_launches", [])) + + def unread_steps(grant: dict[str, Any] | None) -> dict[tuple[str, str], list[dict[str, Any]]]: + # The steps that may launch an agent in a form this reader does not read. + return by_job([ + *(entry for entry in (grant or {}).get("agent_launches", []) if entry.get("form") != "read"), + *(grant or {}).get("unread_agent_runs", []), + ]) + old, new = met(before), met(after) launched_before, launched_after = launches(before), launches(after) + unread_before, unread_after = unread_steps(before), unread_steps(after) lost = [key for key in old if key not in new] gained = [key for key in new if key not in old] @@ -3429,7 +2818,10 @@ def job_left(source: _RuleKey, _target: _RuleKey) -> bool: lost.remove(source) sources[key] = source - gains = AgentRuleGains(claimed=[], unread_before=[], expression_before=[], moved=[]) + def read(entries: list[dict[str, Any]]) -> int: + return sum(1 for entry in entries if entry.get("form") == "read") + + gains = AgentRuleGains(claimed=[], unread_before=[], setting_before=[], moved=[]) for key in gained: entries = new[key] job, family, rule, detail = key @@ -3438,14 +2830,20 @@ def job_left(source: _RuleKey, _target: _RuleKey) -> bool: gains.moved.append((widening, old[sources[key]][0])) continue before_launches = launched_before.get((job, family), []) - if before_launches and not any(entry.get("form") == "read" for entry in before_launches): - gains.unread_before.append(widening) + was_unread = unread_before.get((job, family), []) + still_unread = unread_after.get((job, family), []) + if len(was_unread) > len(still_unread) and read(launched_after.get((job, family), [])) > read( + before_launches + ): + remaining = {str(entry["step"]) for entry in still_unread} + gone = next((entry for entry in was_unread if str(entry["step"]) not in remaining), was_unread[0]) + gains.unread_before.append((widening, gone)) continue setting = next( - (name for entry in before_launches if (name := _expression_setting(entry, rule, detail))), None + (found for entry in before_launches if (found := _unread_setting(entry, rule, detail))), None ) if setting is not None: - gains.expression_before.append((widening, setting)) + gains.setting_before.append((widening, *setting)) continue gains.claimed.append(widening) return gains @@ -3457,27 +2855,18 @@ def gained_agent_widenings(before: dict[str, Any] | None, after: dict[str, Any] return agent_rule_gains(before, after).claimed -#: How an unresolved agent launch or checkout reads in the limit that names it. -_UNRESOLVED_AGENT_PHRASES = { - "compound_command": ( - "part of a `run:` that holds more than one command or a shell reserved word, " - "or quoting this audit cannot split" - ), - "shell_expansion": "a command holding, or inside, a shell expansion this static audit does not evaluate", - "expression": "a `run:` holding a `${{ }}` expression, which GitHub substitutes before the shell reads it", - "inputs_not_a_mapping": "a step whose `with:` is not a mapping", -} - - def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: - """One message per agent launch setting or checkout ref a workflow does not compare (#823). - - Not blocking, like an unread secret value (#693): the launch's job, agent, - form, reason and widening rules are still compared, so adding, removing or - re-forming one, or gaining a documented rule, is a row. Only an edit inside - what is named here is not reported. That includes a setting holding - credential-shaped text (#823 review): it is compared by its redacted text - and its rules, so only an edit inside what is redacted is not reported. A + """One message per agent launch setting, unread agent step or checkout ref a workflow does not compare (#823). + + Not blocking, like an unread secret value (#693). A launch's job, agent, + form and widening rules are still compared, so adding, removing or + re-forming one, or gaining a documented rule, is a row; only an edit + inside what is named here is not reported. A setting holding + credential-shaped text (#823 review) is compared by its redacted text and + its rules, so only an edit inside what is redacted is not reported. An + argument input that is not a plain list of words is compared by a digest + and read for no rule. An unread ``run:`` agent step is compared not at + all: it gives no row, whatever is edited (#823 review cycle 4). A redacted checkout ref is not named here: :func:`_uncompared_workflow_text` makes it a blocking limit, as a redacted step reference is (#767). """ @@ -3485,11 +2874,9 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: texts: list[str] = [] for entry in grant.get("agent_launches", []): where = f"{entry['job']}/{entry['step']} ({entry['agent']})" - reason = entry.get("unresolved_reason") - if reason: + if entry.get("unresolved_reason") == "inputs_not_a_mapping": texts.append( - f"the agent launch at {where} is " - f"{_UNRESOLVED_AGENT_PHRASES.get(str(reason), 'in an unsupported form')}; its " + f"the agent launch at {where} is a step whose `with:` is not a mapping; its " "settings are neither published nor compared, so an edit to them is not reported" ) for setting in entry.get("settings", []): @@ -3502,12 +2889,21 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: "is not reported" ) continue + if unread == "unread_arguments": + texts.append( + f"the {setting['name']} value of the agent launch at {where} is not a plain list " + "of words this audit reads: it holds a quote, a `${{ }}` expression, `$`, a " + "backtick, a comment, a shell operator, JSON, `--settings` or `--mcp-config`, or " + "another character outside the plain set; none of its text is published and no " + "documented widening rule is read from it, and it is compared only by a digest, " + "so an edit to it is a change, never a widening" + ) + continue what = { "not_a_string": "is not a string", "unparsed_json": ( - "holds text that starts like JSON, or a codex `--config` table or array, " - "and does not parse, so the values it may hold cannot be told apart from " - "its key names" + "holds text that starts like JSON and does not parse, so the values it may " + "hold cannot be told apart from its key names" ), }.get(str(unread)) if what: @@ -3516,6 +2912,13 @@ def uncompared_agent_launch_texts(grant: dict[str, Any]) -> list[str]: "neither published nor compared, so an edit to it that gains no documented " "widening rule is not reported" ) + for entry in grant.get("unread_agent_runs", []): + texts.append( + f"the `run:` at {entry['job']}/{entry['step']} mentions {entry['agent']} and is not read " + "as an agent launch: only a single-line command of plain words run by bash or sh, whose " + "program is `claude -p` or `codex exec`, is read, so this step may start an agent that is " + "neither published nor compared, and adding, removing or editing it gives no row" + ) for entry in grant.get("checkout_refs", []): unread = entry.get("unresolved_reason") what = { @@ -3761,8 +3164,9 @@ def _workflow_grant( Each label is published by :func:`published_workflow_label` before it is used anywhere, so the job in ``permission_contexts``, ``reusable_calls``, - ``step_actions``, ``agent_launches``, ``checkout_refs``, the - ``write_scopes`` and ``effective_write_scopes`` prefixes, every row built + ``step_actions``, ``agent_launches``, ``unread_agent_runs``, + ``checkout_refs``, the ``write_scopes`` and ``effective_write_scopes`` + prefixes, every row built from them, and ``config_sha256`` all hold the same label, and none holds the raw text. ``collided`` receives the kinds of label of which two distinct raw values publish alike. @@ -3783,6 +3187,7 @@ def _workflow_grant( reusable_calls: list[dict[str, Any]] = [] step_actions: list[dict[str, Any]] = [] agent_launches: list[dict[str, Any]] = [] + unread_agent_runs: list[dict[str, Any]] = [] checkout_refs: list[dict[str, Any]] = [] def collect(perms: Any, where: str) -> None: @@ -3832,7 +3237,11 @@ def collect(perms: Any, where: str) -> None: action = _step_action(label, step, index) if action is not None: step_actions.append(action) - job_launches.extend(_step_agent_launches(label, step, index)) + launches, unread = _step_agent_launches( + label, step, index, _declared_shell(step, job, data) + ) + job_launches.extend(launches) + unread_agent_runs.extend(unread) checkout = _checkout_ref(label, step, index) if checkout is not None: checkout_refs.append(checkout) @@ -3866,6 +3275,9 @@ def collect(perms: Any, where: str) -> None: # Omitted when empty for the same reason (#823). if agent_launches: projection["agent_launches"] = agent_launches + if unread_agent_runs: + # Named as a limit, never compared (#823 review cycle 4). + projection["unread_agent_runs"] = unread_agent_runs if checkout_refs: projection["checkout_refs"] = checkout_refs return { @@ -5758,7 +5170,12 @@ def _same_workflow_grant(before: dict | None, after: dict | None) -> bool: # Agent launches and checkout refs compare the same way (#823): as each # job's multiset of declared facts, never by step label, and a launch's # `job_secrets` is context for the row, never compared. - ignored = {"write_scopes", "config_sha256", "step_actions", "agent_launches", "checkout_refs"} + # An unread `run:` agent step is a named limit and is never compared, so + # adding, removing or editing one gives no row (#823 review cycle 4). + ignored = { + "write_scopes", "config_sha256", "step_actions", "agent_launches", "checkout_refs", + "unread_agent_runs", + } def comparable(grant: dict[str, Any]) -> dict[str, Any]: # Both sides are v0.4+ shapes here, and a drift payload refuses a diff --git a/src/agents_shipgate/schemas/contract.py b/src/agents_shipgate/schemas/contract.py index 8238c8cc1..a5b32be47 100644 --- a/src/agents_shipgate/schemas/contract.py +++ b/src/agents_shipgate/schemas/contract.py @@ -229,11 +229,14 @@ # (#823). Host-grants 0.6 shipped in 1.1.0, so this mints host-grants # inventory, baseline and drift 0.7: a workflow grant adds ``agent_launches[]`` # (a documented agent action's permission inputs, or the permission flags of a -# literal ``claude -p`` / ``codex exec`` run step, compared as text) and -# ``checkout_refs[]`` (each ``actions/checkout`` step's ``with.ref``), both -# omitted when empty. Only a documented rule gained by a job's agent launches -# widens; every other edit is a ``changed`` row, and a workflow row that runs -# an agent names the job facts beside it. A 0.4-0.6 baseline holding a +# ``run:`` that is one plain ``claude -p`` / ``codex exec`` command, compared as +# text), ``unread_agent_runs[]`` (any other ``run:`` that mentions an agent CLI: +# a named limit, never compared) and ``checkout_refs[]`` (each +# ``actions/checkout`` step's ``with.ref``), each omitted when empty. Shell is +# not parsed; an argument input that is not a plain list of words is compared by +# a digest and read for no rule. Only a documented rule gained by a job's agent +# launches widens; every other edit is a ``changed`` row, and a workflow row +# that runs an agent names the job facts beside it. A 0.4-0.6 baseline holding a # workflow grant is incomparable # (``baseline_workflow_agent_launches_unavailable``); one without a workflow # stays comparable. #823 moves neither verifier 0.21 nor capability diff 0.4: diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index 2b75205e9..c747e9755 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -627,29 +627,39 @@ class HostWorkflowAgentSettingV7(BaseModel): ``name`` is the documented input (``claude_args``, ``sandbox``, …) or the flag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too). ``value`` is the declared text, stripped, as it may be published; a flag - that takes no value has ``null``. ``claude_args`` is the text the Claude - actions parse, without the full-line ``#`` comments they drop. A - structured value — a JSON object (a ``settings`` or ``mcp_config`` value, - a ``--settings`` or ``--mcp-config`` value, any argument word), or a codex - ``--config`` table or array — publishes its shape and none of its free - text: key names, numbers, booleans and ``null``, with each string - replaced by ````, a short digest of what the host readers - digest for it, so an edit to it is still a change. ``env`` and - ``headers`` values, ``apiKeyHelper`` and every secret-named value are - ````, as the host readers redact them. The strings a host reader - publishes are kept: a ``permissions.allow``/``ask``/``deny`` rule and a - documented Claude Code setting's value such as ``defaultMode``, and an - MCP server's command name and its URL's scheme and host, each followed by - the digest when it drops something the digest reads (a command's - arguments, a URL's query). So an MCP server's arguments and a hook's - command publish nothing, as `.mcp.json` and `.claude/settings.json` do - not (#823 review). A codex ``--config`` override under ``env``, - ``headers`` or a secret-named key publishes ```` for its value, - and a URL elsewhere publishes its scheme and host with - ```` for its path and query (#723). Each is withheld - however it is attached to its flag: ``--settings={…}`` and - ``-c`` as well as a separate word. Other argument text — a - prompt, a flag's value, a codex ``--config`` override's scalar value — is + that takes no value has ``null``. + + ``claude_args`` and ``codex-args`` are read only when they are a plain + list of words: letters, digits and ``_ . / : = , % + - ( )``, separated by + blanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every + parser involved splits such text the same way, so it is published as + those words, one space apart. Any other value — holding a quote, a + ``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator, + JSON or another character — is ``unread_arguments``: ``value`` is + ````, a short digest, so an edit to it is still a change + while none of its text is published; no documented widening rule is read + from it; and it records a non-blocking coverage issue naming its + ``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps + its key; its value is ```` under ``env``, ``headers`` or a + secret-named key, as the host readers redact such values, published as + written for ``sandbox_mode``, ``default_permissions``, + ``approval_policy`` and ``model``, and ```` otherwise. + + Every other input is one value. A JSON object (a ``settings`` or + ``mcp_config`` value) publishes its shape and none of its free text: key + names, numbers, booleans and ``null``, with each string replaced by + ````, a short digest of what the host readers digest for it, + so an edit to it is still a change. ``env`` and ``headers`` values, + ``apiKeyHelper`` and every secret-named value are ````, as the + host readers redact them. The strings a host reader publishes are kept: + a ``permissions.allow``/``ask``/``deny`` rule and a documented Claude + Code setting's value such as ``defaultMode``, and an MCP server's command + name and its URL's scheme and host, each followed by the digest when it + drops something the digest reads (a command's arguments, a URL's query). + So an MCP server's arguments and a hook's command publish nothing, as + `.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in + other text publishes its scheme and host with ```` for its + path and query (#723). Other text — a prompt, a flag's value — is published through the workflow label redaction (#802). A value it rewrites is credential-shaped — a token, but also prose such as "never print bearer tokens" — and is published redacted with @@ -661,19 +671,22 @@ class HostWorkflowAgentSettingV7(BaseModel): is ``null`` and records a non-blocking coverage issue naming its ``job/step``: it is neither published nor compared. - ``holds_expression`` is ``true`` when the declared text holds a ``${{ }}`` - expression, which GitHub substitutes before the action reads the input, - and is omitted otherwise. A documented widening rule is then read only - from the literal text the expression cannot reach, and a rule the launch - gains in the same job afterwards is not claimed, because the substituted - text may already have met it. + ``holds_expression`` is ``true`` when an input other than an argument + input holds a ``${{ }}`` expression, which GitHub substitutes before the + action reads the input, and is omitted otherwise. A documented widening + rule is then read only from the entries of a user gate that hold none, + and from no mode or settings input, and a rule the launch gains in the + same job afterwards is not claimed, because the substituted text may + already have met it. """ model_config = ConfigDict(extra="forbid") name: str value: str | None - unresolved_reason: Literal["not_a_string", "redacted", "unparsed_json"] | None = None + unresolved_reason: Literal[ + "not_a_string", "redacted", "unparsed_json", "unread_arguments", + ] | None = None holds_expression: bool = Field(default=False, exclude_if=lambda value: not value) @@ -682,18 +695,17 @@ class HostWorkflowAgentRuleV7(BaseModel): Decided when the workflow is read, from the declared text, before any of it is withheld for publication, so redaction never hides a rule. Only - literal text meets one: in a value holding ``${{ }}``, the words of - ``claude_args`` or ``codex-args`` before the first expression, less the - word it touches and a quoted run still open at it, and the entries of a - user gate that hold none; a mode or ``settings`` input holding one meets - none. Claude Code settings written as JSON — the ``settings`` input, or a - ``--settings`` value in ``claude_args`` or on the CLI — meet - ``bypass_permissions`` when their ``defaultMode`` is ``bypassPermissions``, - read as the settings reader reads it; a path to a settings file is not - read. ``setting`` is the input (``claude_args``, ``allowed_bots``, - ``sandbox``, ``permission-profile``, …) or the CLI flag's primary - spelling. One rule compares as one whatever setting meets it, except - ``open_gate``, which is one rule per gate input. + text this reader reads exactly meets one: ``claude_args`` or + ``codex-args`` only when it is a plain list of words (never when it holds + a ``${{ }}`` expression), the entries of a user gate that hold no + expression, and a mode or ``settings`` input that holds none. Claude Code + settings written as JSON in the ``settings`` input meet + ``bypass_permissions`` when their ``defaultMode`` is + ``bypassPermissions``, read as the settings reader reads it; a path to a + settings file is not read. ``setting`` is the input (``claude_args``, + ``allowed_bots``, ``sandbox``, ``permission-profile``, …) or the CLI + flag's primary spelling. One rule compares as one whatever setting meets + it, except ``open_gate``, which is one rule per gate input. """ model_config = ConfigDict(extra="forbid") @@ -713,20 +725,22 @@ class HostWorkflowAgentLaunchV7(BaseModel): ``agent`` is a documented action reference's ``owner/repo`` (the step's ``uses:`` at any ref; the Claude base action also as the ``base-action`` - directory of ``anthropics/claude-code-action``) or a known agent CLI a - literal ``run:`` starts with: ``claude`` with ``-p``/``--print``, or - ``codex exec``. ``form: read`` lists the documented permission inputs or - flags the step declares in ``settings``, and the documented widening - rules they meet in ``widening_rules``, omitted when none. - ``form: unresolved`` names why the step's settings were not - read — a ``run:`` holding more than one command, a shell reserved word or - quoting that does not balance, a shell expansion (an agent CLI inside a - command substitution included), a ``${{ }}`` expression, or ``with:`` - that is not a mapping — with no - settings, and records a non-blocking coverage issue. ``job_secrets`` names - the secrets the step's job references (``${{ secrets.NAME }}``) and the - workflow-level ``env`` passes: context for the row that names this step, - never compared. ``job`` and ``step`` are published labels (#802). + directory of ``anthropics/claude-code-action``), or a known agent CLI a + ``run:`` launches when the whole ``run:`` is one line of plain words + (letters, digits and ``_ . / : = , % + -``, separated by spaces or tabs), + run by ``bash``, ``sh`` or the runner's default shell, whose program, + after any ``NAME=value`` assignments, has the file name ``claude`` and + passes ``-p``/``--print``, or ``codex`` followed by ``exec`` (``e``). + ``form: read`` lists the documented permission inputs or flags the step + declares in ``settings``, and the documented widening rules they meet in + ``widening_rules``, omitted when none. ``form: unresolved`` is an agent + action whose ``with:`` is not a mapping (``inputs_not_a_mapping``), with + no settings, and records a non-blocking coverage issue. Any other + ``run:`` that mentions an agent CLI is not a launch: it is listed in + ``unread_agent_runs``. ``job_secrets`` names the secrets the step's job + references (``${{ secrets.NAME }}``) and the workflow-level ``env`` + passes: context for the row that names this step, never compared. + ``job`` and ``step`` are published labels (#802). """ model_config = ConfigDict(extra="forbid") @@ -742,12 +756,7 @@ class HostWorkflowAgentLaunchV7(BaseModel): "codex", ] form: Literal["read", "unresolved"] - unresolved_reason: Literal[ - "compound_command", - "shell_expansion", - "expression", - "inputs_not_a_mapping", - ] | None = None + unresolved_reason: Literal["inputs_not_a_mapping"] | None = None settings: list[HostWorkflowAgentSettingV7] = Field(default_factory=list) widening_rules: list[HostWorkflowAgentRuleV7] = Field( default_factory=list, exclude_if=lambda value: not value, @@ -755,6 +764,28 @@ class HostWorkflowAgentLaunchV7(BaseModel): job_secrets: list[str] = Field(default_factory=list, exclude_if=lambda value: not value) +class HostWorkflowUnreadAgentRunV7(BaseModel): + """A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4). + + Any ``run:`` holding ``claude`` or ``codex`` as a word of its own that is + not an agent launch this reader reads — more than one line or command, a + quote, an expansion, a redirection, a comment, a continuation, a + ``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a + subcommand that is not a headless launch, or a declared ``shell:`` other + than ``bash`` or ``sh`` — once for each agent CLI it mentions. It is a + named, non-blocking limit and nothing more: none of the step's text is + published, it is never compared, so adding, removing or editing it gives + no row, and it never says that the step starts, or does not start, an + agent. ``job`` and ``step`` are published labels (#802). + """ + + model_config = ConfigDict(extra="forbid") + + job: str + step: str + agent: Literal["claude", "codex"] + + class HostWorkflowCheckoutRefV7(BaseModel): """One ``actions/checkout`` step and the ``with.ref`` it declares, as text (#823). @@ -777,18 +808,22 @@ class HostWorkflowCheckoutRefV7(BaseModel): class HostWorkflowGrantV7(HostWorkflowGrantV6): - """A v0.6 workflow grant plus the agent launches and checkout refs its steps declare. + """A v0.6 workflow grant plus the agent launches, unread agent steps and checkout refs its steps declare. - Both lists are present only when a step declares one. In a v0.7 grant an + Each list is present only when a step declares one. In a v0.7 grant an absent list means the steps were read and declare none; the schema version, not the key, separates that from a legacy grant that never read - them. ``access`` and ``risk`` still describe the workflow's token and - triggers alone. + them. ``unread_agent_runs`` is a named limit and is never compared. + ``access`` and ``risk`` still describe the workflow's token and triggers + alone. """ agent_launches: list[HostWorkflowAgentLaunchV7] = Field( default_factory=list, exclude_if=lambda value: not value, ) + unread_agent_runs: list[HostWorkflowUnreadAgentRunV7] = Field( + default_factory=list, exclude_if=lambda value: not value, + ) checkout_refs: list[HostWorkflowCheckoutRefV7] = Field( default_factory=list, exclude_if=lambda value: not value, ) diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 19ce03c46..a8ee873b8 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -231,7 +231,7 @@ def paths(self) -> list[Path]: # comes from the engine's # `workflow_agent_widened_*` expansion signal, itself read off the # `widening_rules` the engine published on each launch; the `why`'s - # moved, unread-before and expression sentences read the same + # moved, unread-before and setting-before sentences read the same # `agent_rule_gains` the signal is computed from, and the note reads the # triggers, write scopes, secrets and checkout refs the engine already # published on the grant, so it adds no claim; diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 7e1fbfab1..9f1380877 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -3,12 +3,17 @@ The workflow grant already read triggers, token permissions, reusable calls and step `uses:` references (#771), and nothing that says how an agent is started. It now lists each documented agent action's permission inputs, the permission -flags of a literal `claude -p` / `codex exec` run step, and each +flags of a `run:` that is one plain `claude -p` / `codex exec` command, and each `actions/checkout` step's `with.ref`. Nothing is executed, fetched or evaluated. Only a documented rule a job's launches gain widens; every other -edit is `changed`; a shape this reader does not read is a named limit, never a -row that claims an effect; and a workflow row that runs an agent names the job -facts beside it. +edit is `changed`; and a workflow row that runs an agent names the job facts +beside it. + +Shell is not parsed (#823 review cycle 4). A `run:` is read only as one line of +plain words whose program is a known agent CLI, and `claude_args` / +`codex-args` only as a plain list of words; every other form is a named, +non-blocking limit that publishes none of its text, and an unread `run:` step +is never compared, so it gives no row. """ from __future__ import annotations @@ -16,6 +21,7 @@ import hashlib import json import subprocess +import time from pathlib import Path import pytest @@ -25,10 +31,8 @@ from agents_shipgate.cli.main import app from agents_shipgate.core.capability_diff_rows import capability_diff_rows from agents_shipgate.core.host_grants import ( - _claude_argument_input, - _codex_argument_input, - _command_substitutions, - _literal_argument_words, + _argument_words, + _run_command, _uncompared_workflow_text, _workflow_grant, diff_host_grants, @@ -42,7 +46,7 @@ HEAD_SHA = "${{ github.event.pull_request.head.sha }}" -def _workflow(*steps, trigger="pull_request", permissions=None, jobs=None, env=None): +def _workflow(*steps, trigger="pull_request", permissions=None, jobs=None, env=None, defaults=None): data = { "on": trigger, "permissions": permissions if permissions is not None else {"contents": "read", "pull-requests": "read"}, @@ -50,15 +54,17 @@ def _workflow(*steps, trigger="pull_request", permissions=None, jobs=None, env=N } if env is not None: data["env"] = env + if defaults is not None: + data["defaults"] = defaults return data -def _agent(claude_args='--allowedTools "Read"', **extra): +def _agent(claude_args="--allowedTools Read", **extra): return {"uses": CLAUDE, "with": {"claude_args": claude_args, **extra}} -def _reproduction(trigger="pull_request", pr="read", claude_args='--allowedTools "Read"', run="echo done", ref=None): - """The workflow of #823's reproduction, as its `wf` shell function writes it.""" +def _reproduction(trigger="pull_request", pr="read", claude_args="--allowedTools Read", run="echo done", ref=None): + """The workflow of #823's reproduction, as its `wf` shell function writes it, with plain `claude_args`.""" checkout = {"uses": "actions/checkout@v4", **({"with": {"ref": ref}} if ref else {})} return _workflow( @@ -85,38 +91,74 @@ def _launches(value): return _grant(value).get("agent_launches", []) +def _unread(value): + return _grant(value).get("unread_agent_runs", []) + + +def _digest(text): + """What a withheld string publishes: a digest of what the host readers digest for it.""" + + from agents_shipgate.core.host_grants import redacted_config_sha256 + + return f"" + + # --- the four cases of the reproduction -------------------------------------------- def test_args_gaining_bypass_permissions_is_one_widened_row_naming_job_step_and_both_values(): row, = _rows( _reproduction(), - _reproduction(claude_args='--permission-mode bypassPermissions --allowedTools "Bash(*)"'), + _reproduction(claude_args="--permission-mode bypassPermissions --allowedTools Bash(git:status)"), ) assert row.subject == f"github {SOURCE}" assert (row.direction, row.expands) == ("widened", True) - assert 'review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools "Read"' in row.before + assert "review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools Read" in row.before assert ( - 'review/steps[1]: runs anthropics/claude-code-action with claude_args: ' - '--permission-mode bypassPermissions --allowedTools "Bash(*)"' + "review/steps[1]: runs anthropics/claude-code-action with claude_args: " + "--permission-mode bypassPermissions --allowedTools Bash(git:status)" ) in row.after assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[1])" in row.why +def test_the_quoted_args_of_the_reproduction_are_a_changed_row_that_publishes_only_digests(): + """#823 review cycle 4 scope: quoted `claude_args` is not read, only compared by a digest.""" + + before, after = '--allowedTools "Read"', '--permission-mode bypassPermissions --allowedTools "Bash(*)"' + assert host_grant_expansion_signals(_changes(_reproduction(claude_args=before), _reproduction(claude_args=after))) == [] + row, = _rows(_reproduction(claude_args=before), _reproduction(claude_args=after)) + + assert (row.direction, row.expands) == ("changed", False) + assert f"claude_args (not read; digest {_digest(before)})" in row.before + assert f"claude_args (not read; digest {_digest(after)})" in row.after + assert "an agent launch's declared settings changed (review/steps[1])" in row.why + assert ( + "an agent launch's argument input is not a plain list of words this audit reads (claude_args at " + "review/steps[1]); none of its text is published and it is compared by a digest only, so this row " + "does not say whether it meets a documented widening rule" + ) in row.why + for text in ("bypassPermissions", "Bash(*)", '"Read"'): + assert text not in row.before + row.after + limit, = uncompared_agent_launch_texts(_grant(_reproduction(claude_args=after))) + assert limit.startswith( + "the claude_args value of the agent launch at review/steps[1] (anthropics/claude-code-action) is not " + "a plain list of words this audit reads" + ) + + def test_a_literal_claude_run_step_is_one_changed_row_with_its_permission_flags(): row, = _rows( _reproduction(), - _reproduction(run='claude -p --permission-mode acceptEdits --allowedTools "Bash(*)" "Summarize this change"'), + _reproduction(run="claude -p --permission-mode acceptEdits --allowedTools Edit Summarize"), ) assert (row.direction, row.expands) == ("changed", False) assert "review/steps[2]" not in row.before # A variadic flag reads every following word up to the next flag, as the - # CLI reads it, so the trailing quoted word is part of --allowedTools. + # CLI reads it, so the trailing prompt word is part of --allowedTools. assert ( - "review/steps[2]: runs claude -p with --allowedTools 'Bash(*)' 'Summarize this change'; " - "--permission-mode acceptEdits" + "review/steps[2]: runs claude -p with --allowedTools Edit Summarize; --permission-mode acceptEdits" ) in row.after assert "a step now launches an agent (review/steps[2])" in row.why assert "not counted as a widening" in row.why @@ -213,23 +255,23 @@ def test_a_literal_claude_command_publishes_its_permission_flags_under_their_pri launch, = _launches(_workflow({ "name": "Review", "run": ( - "claude --print --allowed-tools=Read --disallowedTools 'Bash(rm *)' " - "--model sonnet --dangerously-skip-permissions --add-dir ../docs \"secret prompt text\"" + "claude --print --allowed-tools=Read --disallowedTools Bash " + "--model sonnet --dangerously-skip-permissions --add-dir ../docs secret-prompt-text" ), })) assert (launch["agent"], launch["step"], launch["form"]) == ("claude", "Review", "read") assert launch["settings"] == [ - {"name": "--add-dir", "value": "../docs 'secret prompt text'", "unresolved_reason": None}, + {"name": "--add-dir", "value": "../docs secret-prompt-text", "unresolved_reason": None}, {"name": "--allowedTools", "value": "Read", "unresolved_reason": None}, {"name": "--dangerously-skip-permissions", "value": None, "unresolved_reason": None}, - {"name": "--disallowedTools", "value": "'Bash(rm *)'", "unresolved_reason": None}, + {"name": "--disallowedTools", "value": "Bash", "unresolved_reason": None}, ] assert "sonnet" not in json.dumps(launch) def test_a_literal_codex_exec_command_publishes_its_permission_flags(): - launch, = _launches(_workflow({"run": "codex e -s danger-full-access --yolo -c model=o3 'fix it'"})) + launch, = _launches(_workflow({"run": "codex e -s danger-full-access --yolo -c model=o3 fix-it"})) assert (launch["agent"], launch["form"]) == ("codex", "read") assert launch["settings"] == [ @@ -240,100 +282,17 @@ def test_a_literal_codex_exec_command_publishes_its_permission_flags(): def test_literal_assignments_before_the_command_are_skipped_and_never_published(): - launch, = _launches(_workflow({"run": "CI=true claude -p --permission-mode plan 'go'"})) + launch, = _launches(_workflow({"run": "CI=true ANTHROPIC_API_KEY=sk-canary claude -p --permission-mode plan go"})) assert launch["form"] == "read" assert launch["settings"] == [{"name": "--permission-mode", "value": "plan", "unresolved_reason": None}] + assert "sk-canary" not in json.dumps(_grant(_workflow({"run": "CI=true ANTHROPIC_API_KEY=sk-canary claude -p go"}))) -@pytest.mark.parametrize( - "run", - [ - "claude mcp add github -- npx server", - "claude --version", - "codex login --api-key sk-test", - 'echo "claude -p --dangerously-skip-permissions"', - "echo claude -p done", - "npm test", - # a substitution the shell never runs, or one that runs another command - 'echo "\\$(claude -p --dangerously-skip-permissions)"', - "echo '$(claude -p --dangerously-skip-permissions)'", - "echo '`claude -p --dangerously-skip-permissions`'", - 'echo "$(date) claude -p --dangerously-skip-permissions"', - # a reserved word is read only before a command's assignments - "CI=true then claude -p 'go'", - ], - ids=["claude-mcp", "claude-version", "codex-login", "echo-quoted", "echo-bare", "unrelated", - "escaped-substitution", "single-quoted-substitution", "single-quoted-backtick", - "substitution-of-another-command", "reserved-word-after-assignment"], -) -def test_a_command_that_launches_no_headless_agent_is_not_listed(run): - assert _launches(_workflow({"run": run})) == [] - - -@pytest.mark.parametrize( - ("run", "reason"), - [ - ("npm ci && claude -p --dangerously-skip-permissions 'go'", "compound_command"), - ("npm ci\nclaude -p 'go'", "compound_command"), - ("claude -p 'go' | tee review.md", "compound_command"), - ("cat < prompt.md\nIt's broken\nEOF\nclaude -p --dangerously-skip-permissions 'go'", "compound_command"), - ("claude -p --dangerously-skip-permissions \"go", "compound_command"), - # an agent CLI inside a double-quoted or backtick substitution (#823 review) - ('gh pr comment "$PR" --body "$(claude -p --dangerously-skip-permissions \'go\')"', "shell_expansion"), - ('REVIEW="$(claude -p --dangerously-skip-permissions \'go\')"', "shell_expansion"), - ("REVIEW=`claude -p --dangerously-skip-permissions 'go'`", "shell_expansion"), - ('echo "$(echo "$(claude -p --dangerously-skip-permissions \'go\')")"', "compound_command"), - # a here-document body is not read, so an apostrophe in it no longer unbalances the rest - ("cat < prompt.md\nIt's broken\nEOF\nREVIEW=\"$(claude -p --dangerously-skip-permissions 'go')\"", - "compound_command"), - # quoting that does not balance is read a line at a time, substitutions included - ("echo 'broken\nclaude -p --dangerously-skip-permissions go", "compound_command"), - ("echo 'broken\nREVIEW=\"$(claude -p --dangerously-skip-permissions go)\"", "compound_command"), - # an agent CLI after a shell reserved word - ("if true; then claude -p --dangerously-skip-permissions 'go'; fi", "compound_command"), - ("for f in a b; do claude -p --dangerously-skip-permissions 'go'; done", "compound_command"), - ("{ claude -p --dangerously-skip-permissions 'go'; }", "compound_command"), - ("! claude -p --dangerously-skip-permissions 'go'", "compound_command"), - ("time -p claude -p --dangerously-skip-permissions 'go'", "compound_command"), - ], - ids=["and", "lines", "pipe", "heredoc", "variable", "substitution", "expression", "after-heredoc", - "unbalanced", "quoted-substitution-argument", "quoted-substitution-assignment", "backtick", - "nested-substitution", "substitution-after-heredoc", "unbalanced-lines", "unbalanced-substitution", - "if-then", "for-do", "brace-group", "negated", "timed"], -) -def test_a_shape_this_reader_does_not_read_is_unresolved_and_publishes_no_text(run, reason): - launch, = _launches(_workflow({"run": run})) - - assert (launch["agent"], launch["form"], launch["unresolved_reason"]) == ("claude", "unresolved", reason) - assert launch["settings"] == [] - assert "dangerously" not in json.dumps(launch) and "go" not in json.dumps(launch["settings"]) - limit, = uncompared_agent_launch_texts(_grant(_workflow({"run": run}))) - assert limit.startswith("the agent launch at review/steps[0] (claude) is ") - assert "not reported" in limit - - -def test_deeply_nested_substitutions_are_read_in_one_pass(): - """#823 review: a substitution is found without recursion, and each character is split once.""" - - depth = 20000 - run = 'REVIEW="' + "$(" * depth + "claude -p --dangerously-skip-permissions 'go'" + ")" * depth + '"' - launch, = _launches(_workflow({"run": run})) - assert (launch["agent"], launch["form"], launch["unresolved_reason"]) == ("claude", "unresolved", "shell_expansion") - # Each nested substitution reads as `_` in the one around it. - assert _command_substitutions('echo "$(echo "$(codex exec x)")"') == ["codex exec x", 'echo "_"'] - - -def test_a_quoted_word_that_starts_with_a_hash_is_not_a_comment(): - launch, = _launches(_workflow({"run": 'claude -p --allowedTools Read "#123 review"'})) - assert launch["form"] == "read" - assert launch["settings"] == [{"name": "--allowedTools", "value": "Read '#123 review'", "unresolved_reason": None}] - - commented, = _launches(_workflow({"run": "claude -p --allowedTools Read # review"})) - assert (commented["form"], commented["unresolved_reason"]) == ("unresolved", "compound_command") +def test_an_agent_cli_named_by_its_path_is_read_by_its_file_name(): + for run in ("./node_modules/.bin/claude -p --dangerously-skip-permissions go", "/usr/local/bin/codex exec --yolo go"): + launch, = _launches(_workflow({"run": run})) + assert launch["form"] == "read" and "widening_rules" in launch + assert run.split()[0] not in json.dumps(launch) def test_the_base_action_directory_of_the_claude_action_is_read_as_the_base_action(): @@ -350,17 +309,13 @@ def test_the_base_action_directory_of_the_claude_action_is_read_as_the_base_acti assert launch["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] -def test_single_quoted_dollars_are_literal_and_do_not_stop_the_read(): - launch, = _launches(_workflow({"run": "claude -p --allowedTools 'Bash(echo $HOME)' 'go'"})) - assert launch["form"] == "read" - assert launch["settings"][0]["value"] == "'Bash(echo $HOME)' go" - - def test_inputs_that_are_not_a_mapping_are_unresolved(): launch, = _launches(_workflow({"uses": CLAUDE, "with": ["claude_args"]})) assert (launch["form"], launch["unresolved_reason"], launch["settings"]) == ( "unresolved", "inputs_not_a_mapping", [], ) + limit, = uncompared_agent_launch_texts(_grant(_workflow({"uses": CLAUDE, "with": ["claude_args"]}))) + assert "a step whose `with:` is not a mapping" in limit def test_every_checkout_step_records_its_declared_ref(): @@ -405,7 +360,7 @@ def test_pull_request_code_is_the_documented_head_refs_only(ref, pull_request_co def test_a_workflow_without_agents_or_checkouts_keeps_its_v0_6_shape(): grant = _grant(_workflow({"run": "make test"}, {"uses": "actions/setup-python@v5"})) - assert "agent_launches" not in grant and "checkout_refs" not in grant + assert "agent_launches" not in grant and "checkout_refs" not in grant and "unread_agent_runs" not in grant def test_job_secrets_name_what_the_agent_job_and_the_workflow_env_reference(): @@ -423,382 +378,441 @@ def test_job_secrets_name_what_the_agent_job_and_the_workflow_env_reference(): assert launch["job_secrets"] == ["ANTHROPIC_API_KEY", "DEPLOY_KEY", "REVIEW_TOKEN", "WORKFLOW_ENV"] -# --- how an agent action splits its argument input (#823 review cycle 1) ----------------- +# --- the only forms read: plain lists of words (#823 review cycle 4) -------------------- # -# `claude_args` and `codex-args` are not shell text. The Claude actions split -# `claude_args` with shell-quote after dropping full `#` lines and making -# `()|&;<>` literal (base-action/src/parse-sdk-options.ts); `openai/codex-action` -# reads `codex-args` as a JSON array of strings or with string-argv. A widening -# rule is read from the words the action passes on. +# Four review cycles each found a shell form the `run:` reader mis-read, so no +# shell is parsed. A `run:` is read only as one line of plain words — letters, +# digits and `_ . / : = , % + -` — whose program is a known agent CLI; +# `claude_args` and `codex-args` only as such words (parentheses too) across +# blanks and newlines, with no `--settings` or `--mcp-config` flag. @pytest.mark.parametrize( - ("before", "after"), + ("value", "words"), [ - ("--max-turns 5\n--allowedTools Read", "--max-turns 5\n--dangerously-skip-permissions"), - ("--allowedTools Bash(git:*)", "--allowedTools Bash(git:*) --dangerously-skip-permissions"), - ("# review agent\n--max-turns 5", "# review agent\n--dangerously-skip-permissions"), - ("--max-turns 5", "--max-turns 5\n--permission-mode\nbypassPermissions"), - # A word starting with `--` is always a flag to the action, never a value. - ("--allowedTools Read", "--settings --dangerously-skip-permissions"), + ("--max-turns 5\n--allowedTools Read", ["--max-turns", "5", "--allowedTools", "Read"]), + (" --allowedTools\tBash(git:status),Read ", ["--allowedTools", "Bash(git:status),Read"]), + ("--permission-mode=bypassPermissions --add-dir ../docs", ["--permission-mode=bypassPermissions", "--add-dir", "../docs"]), + ("-c model=o3 -csandbox_mode=read-only --json", ["-c", "model=o3", "-csandbox_mode=read-only", "--json"]), ], - ids=["several-lines", "unquoted-parentheses", "comment-line", "mode-over-lines", "never-a-value"], + ids=["lines", "blanks-and-parentheses", "attached-values", "codex-config"], ) -def test_claude_args_are_split_as_the_claude_actions_split_them(before, after): - changes = _changes(_workflow(_agent(before)), _workflow(_agent(after))) - assert host_grant_expansion_signals(changes) == [f"workflow_agent_widened_changed: {SOURCE}"] - row, = _rows(_workflow(_agent(before)), _workflow(_agent(after))) - - assert (row.direction, row.expands) == ("widened", True) - assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[0])" in row.why - assert "not counted as a widening" not in row.why +def test_a_plain_argument_input_is_its_words(value, words): + assert _argument_words(value) == words @pytest.mark.parametrize( - "after", + "value", [ - "# --dangerously-skip-permissions\n--max-turns 5", - " # an indented comment line is dropped too\n--max-turns 5", - # An unquoted `#` later in the input ends it, as shell-quote reads it. + '--allowedTools "Read"', + "--allowedTools 'Read'", + "--model ${{ vars.CLAUDE_MODEL }}", + "--append-system-prompt $PROMPT", + "--append-system-prompt `cat prompt.md`", + "--append-system-prompt $(cat prompt.md)", + "# reviewer: alice\n--max-turns 5", "--max-turns 5 # --dangerously-skip-permissions", - "--max-turns 5 notes#--dangerously-skip-permissions", + "--max-turns 5 notes#x", + "a\\ b", + "--allowedTools Read;Edit", + "--allowedTools Read|Edit", + "--allowedTools Read&Edit", + "--x out.md\n" + "# We'll post the result below\ngh pr comment \"$PR\" --body-file out.md", + "# it's gated\nif [ -n \"$X\" ]; then claude -p --dangerously-skip-permissions go; fi", + "# we don't pipe secrets\ngit diff | claude -p 'Review'", + "npm ci && claude -p --dangerously-skip-permissions go # it's fine", + "# can't\nset -e; codex exec --yolo 'review'", + "# To reproduce locally: npm ci; claude -p \"review this change\"\nnpm test", + "cat < review.md", + "claude -p go 2>&1", + "claude -p \\\n --dangerously-skip-permissions go", + "claude -p $CLAUDE_FLAGS go", + "claude -p \"$(cat prompt.md)\"", + "claude -p 'go'", + 'claude -p "Fix ${{ github.event.issue.title }}"', + 'gh pr comment "$PR" --body "$(claude -p --dangerously-skip-permissions \'go\')"', + "REVIEW=`claude -p --dangerously-skip-permissions go`", + "cat <<'EOF'\nReproduce locally with `claude -p \"review this change\"`.\nEOF", + "cat < comment.md <<'EOF'\nReproduce locally with `claude -p \"review this change\"`.\nEOF\n", + "npm install -g @anthropic-ai/claude-code", + ): + step = {"run": unread} + before = _workflow(step, permissions={"contents": "read", "pull-requests": "write"}) + after = _workflow( + step, {"run": "claude -p --dangerously-skip-permissions Review"}, + permissions={"contents": "read", "pull-requests": "write"}, + ) + assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[1])" in row.why + assert "not counted as a widening" not in row.why.split("; an agent runs at")[0] + assert "review/steps[0]" not in row.why + row.before + row.after -def test_an_edit_inside_an_unresolved_command_is_quiet_and_named_as_a_limit(): - before = _workflow({"run": "npm ci && claude -p --allowedTools Read 'go'"}) - after = _workflow({"run": "npm ci && claude -p --dangerously-skip-permissions 'go'"}) - assert _rows(before, after) == [] - assert uncompared_agent_launch_texts(_grant(after)) +def test_an_unread_step_rewritten_as_a_read_launch_does_not_claim_the_rule_it_may_already_have_met(): + before = _workflow({"run": 'npm ci && claude -p --dangerously-skip-permissions "Review"'}) + after = _workflow({"run": "npm ci"}, {"run": "claude -p --dangerously-skip-permissions Review"}) + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + assert ( + "an agent launch now skips permission checks (bypassPermissions) (review/steps[1]), which is not " + "counted as a widening: a step in this job that may launch the agent in a form this audit does not " + "read is gone (review/steps[0]), and this launch may be that step rewritten in a form this audit " + "reads, which may already have done the same" + ) in row.why -def test_adding_an_unresolved_launch_is_a_row_that_claims_no_effect(): - row, = _rows(_workflow({"run": "npm test"}), _workflow({"run": "npm ci && claude -p --yolo 'go'"})) - assert (row.direction, row.expands) == ("changed", False) - assert "review/steps[0]: runs claude -p (unresolved: compound command)" in row.after - assert "a step launches an agent in a form this audit does not read (review/steps[0])" in row.why - assert "does not say what that agent may do" in row.why +def test_an_unread_step_that_goes_while_a_read_launch_gains_a_rule_still_widens(): + """The read launch was read on both sides, so the gain is its own.""" + before = _workflow({"run": "npm ci && claude -p go"}, _agent()) + after = _workflow(_agent("--dangerously-skip-permissions")) -SUBSTITUTED = 'gh pr comment "$PR" --body "$(claude -p {flags} \'Review this change\')"' + assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] @pytest.mark.parametrize( - "run", + ("before", "after"), [ - SUBSTITUTED.format(flags="--dangerously-skip-permissions"), - "if true; then claude -p --dangerously-skip-permissions 'Review this change'; fi", + ("claude -p --allowedTools Read Review", + "npx @anthropic-ai/claude-code -p --dangerously-skip-permissions Review"), + ("claude -p --allowedTools Read Review", 'claude -p --dangerously-skip-permissions "Review"'), + ("codex exec -s workspace-write review", "codex -c sandbox_mode=danger-full-access exec review"), ], - ids=["quoted-substitution", "if-then"], + ids=["npx", "quoted", "codex-root-options"], ) -def test_an_agent_cli_the_splitter_does_not_head_is_a_row_and_a_named_limit(run): - """#823 review: a launch inside a quoted `$(…)`, or after `then`, is named, never missed.""" +def test_a_read_launch_that_becomes_a_form_this_audit_does_not_read_is_not_called_gone(before, after): + """#823 review: what was established is that no launch this audit reads is declared.""" - row, = _rows(_reproduction(), _reproduction(run=run)) + row, = _rows(_workflow({"run": before}), _workflow({"run": after})) assert (row.direction, row.expands) == ("changed", False) - assert "a step now launches an agent (review/steps[2])" in row.why - assert "a step launches an agent in a form this audit does not read (review/steps[2])" in row.why - assert "dangerously" not in row.after - - limit, = uncompared_agent_launch_texts(_grant(_reproduction(run=run))) - assert limit.startswith("the agent launch at review/steps[2] (claude) is ") - - -def test_an_edit_inside_a_quoted_substitution_is_quiet_and_named_as_a_limit(): - before = _workflow({"run": SUBSTITUTED.format(flags="--allowedTools Read")}) - after = _workflow({"run": SUBSTITUTED.format(flags="--dangerously-skip-permissions")}) - - assert _rows(before, after) == [] - limit, = uncompared_agent_launch_texts(_grant(after)) - assert "a command holding, or inside, a shell expansion this static audit does not evaluate" in limit - + assert "a step no longer declares an agent launch this audit reads (review/steps[0])" in row.why + assert ( + "a step that no longer declares one may still start an agent in a way this audit does not read, " + "such as an action outside its table, a script, or a `run:` this audit does not read as a launch " + "(more than one command, quoting, an expansion, `npx`, `codex` options before `exec`), so this row " + "does not say that it no longer starts one" + ) in row.why + assert "no longer launches an agent" not in row.why + assert "dangerously" not in row.after and "danger-full-access" not in row.after -# --- here-documents (#823 review cycle 3, C3-F1) ------------------------------------- -#: The step of the review: a PR comment drafted in a quoted here-document. -HERE_DOC_COMMENT = "cat > comment.md <<'EOF'\nReproduce locally with `claude -p \"review this change\"`.\nEOF\n" -COMMENT_PERMISSIONS = {"contents": "read", "pull-requests": "write"} +# --- an agent action's argument input (#823 review cycles 1 and 4) ------------------------ +# +# `claude_args` and `codex-args` are not shell text: each action splits its own +# input. A plain list of words is split alike by all of them, so only that is +# read; a widening rule is read from those words. @pytest.mark.parametrize( - ("opener", "closer"), - [("<<'EOF'", "EOF"), ('<<"EOF"', "EOF"), ("<<\\EOF", "EOF"), ("<<-'EOF'", "\t\tEOF"), ("<< 'EOF'", "EOF"), - ("< comment.md {opener}\n{body}\n{closer}\n"} - assert _launches(_workflow(step, permissions=COMMENT_PERMISSIONS)) == [] - assert _rows( - _workflow({"run": "echo done"}, permissions=COMMENT_PERMISSIONS), - _workflow({"run": "echo done"}, step, permissions=COMMENT_PERMISSIONS), - ) == [] - - -def test_beside_a_quoted_here_doc_a_new_bypass_step_is_a_widening(): - """The here-document is no launch the job had before, so the gain is claimed.""" - - here_doc = {"run": HERE_DOC_COMMENT} - before = _workflow(here_doc, permissions=COMMENT_PERMISSIONS) - after = _workflow( - here_doc, {"run": 'claude -p --dangerously-skip-permissions "Review"'}, permissions=COMMENT_PERMISSIONS, - ) +def test_plain_claude_args_gaining_a_bypass_is_a_widening(before, after): + changes = _changes(_workflow(_agent(before)), _workflow(_agent(after))) + assert host_grant_expansion_signals(changes) == [f"workflow_agent_widened_changed: {SOURCE}"] + row, = _rows(_workflow(_agent(before)), _workflow(_agent(after))) - assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] - row, = _rows(before, after) assert (row.direction, row.expands) == ("widened", True) - assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[1])" in row.why - assert "a form this audit does not read" not in row.why - assert "review/steps[0]" not in row.why + row.before + row.after + assert "an agent launch now skips permission checks (bypassPermissions) (review/steps[0])" in row.why + assert "not counted as a widening" not in row.why @pytest.mark.parametrize( - ("body", "agent"), + "after", [ - ("claude -p --dangerously-skip-permissions 'review'", None), - ("Summary: $(claude -p 'review')", "claude"), - ("Summary: `codex exec 'review'`", "codex"), - # quotes and `#` are ordinary characters in the body - ("'$(claude -p review)'", "claude"), - ("# $(claude -p review)", "claude"), - ("\\$(claude -p review)", None), + "# --dangerously-skip-permissions\n--max-turns 5", + "--max-turns 5 # --dangerously-skip-permissions", + '--dangerously-skip-permissions --append-system-prompt "Review"', + "--dangerously-skip-permissions --model ${{ vars.CLAUDE_MODEL }}", + "--dangerously-skip-permissions --settings ./ci/settings.json", + "--dangerously-skip-permissions --mcp-config '{\"mcpServers\":{}}'", + "--dangerously-skip-permissions --allowedTools Bash(*)", ], - ids=["line-head", "substitution", "backtick", "single-quoted", "hash", "escaped"], + ids=["comment-line", "inline-comment", "quoted-prompt", "expression", "settings", "mcp-config", "glob"], ) -def test_an_unquoted_here_doc_runs_its_substitutions_and_none_of_its_lines(body, agent): - launches = _launches(_workflow({"run": f"cat > comment.md < prompt.md <<'EOF'\nreview\nEOF\nclaude -p --dangerously-skip-permissions 'go'", - # two on one line, their bodies in order - "cat <<'A' <" - - def test_a_json_value_publishes_its_shape_and_none_of_its_free_text(): """#823 review C2-F1: every string a host reader does not publish is withheld, and still compared. The same server in `.mcp.json` publishes `remote (command name npx)`, and the same hook in `.claude/settings.json` publishes `Stop`; neither - publishes an argument or a command. + publishes an argument or a command. The same JSON passed through + `claude_args` or a `run:` is not read at all (#823 review cycle 4). """ + args = f"--allowedTools Read --mcp-config '{REMOTE_MCP_JSON}'" grant = _grant(_workflow( - _agent(f"--allowedTools Read --mcp-config '{REMOTE_MCP_JSON}'", settings=HOOK_JSON), + _agent(args, settings=HOOK_JSON, mcp_config=REMOTE_MCP_JSON), {"run": f"codex exec -c '{REMOTE_CODEX_CONFIG}' 'go'"}, )) - action, cli = grant["agent_launches"] - args = ["mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY0123456789abcdef"] + action, = grant["agent_launches"] + server_args = ["mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY0123456789abcdef"] server = json.dumps( - {"mcpServers": {"remote": {"args": [_digest(arg) for arg in args], "command": "npx"}}}, + {"mcpServers": {"remote": {"args": [_digest(arg) for arg in server_args], "command": "npx"}}}, separators=(",", ":"), ) hook = ( @@ -1398,21 +1298,25 @@ def test_a_json_value_publishes_its_shape_and_none_of_its_free_text(): + '","type":"' + _digest("command") + '"}]}]}}' ) assert {item["name"]: item["value"] for item in action["settings"]} == { - "claude_args": f"--allowedTools Read --mcp-config '{server}'", + "claude_args": _digest(args), + "mcp_config": server, "settings": hook, } - codex_args = ["mcp-remote", "https://mcp.example.com/sse", "--header", "Authorization: Bearer tokCANARY-codex-0123456789"] - assert cli["settings"] == [{ - "name": "--config", - "value": 'mcp_servers.remote={"args":[' + ",".join(f'"{_digest(arg)}"' for arg in codex_args) - + '],"command":"npx"}', - "unresolved_reason": None, - }] + assert grant["unread_agent_runs"] == [{"job": "review", "step": "steps[1]", "agent": "codex"}] text = json.dumps(grant) for canary in SHAPE_CANARIES: assert canary not in text - # Nothing was redacted, so no limit is named: nothing of the text is published to redact. - assert uncompared_agent_launch_texts(grant) == [] + limits = uncompared_agent_launch_texts(grant) + assert [limit.split(";")[0] for limit in limits] == [ + "the claude_args value of the agent launch at review/steps[0] (anthropics/claude-code-action) is not a " + "plain list of words this audit reads: it holds a quote, a `${{ }}` expression, `$`, a backtick, a " + "comment, a shell operator, JSON, `--settings` or `--mcp-config`, or another character outside the " + "plain set", + "the `run:` at review/steps[1] mentions codex and is not read as an agent launch: only a " + "single-line command of plain words run by bash or sh, whose program is `claude -p` or `codex exec`, " + "is read, so this step may start an agent that is neither published nor compared, and adding, " + "removing or editing it gives no row", + ] # A withheld string is still compared: a new argument or command is a row. edited = HOOK_JSON.replace("curl -H", "wget --header") @@ -1453,53 +1357,62 @@ def test_a_json_value_publishes_only_what_the_host_readers_publish(): _agent(f"--mcp-config '{MCP_JSON}' --allowedTools Read", settings=SETTINGS_JSON, mcp_config=MCP_JSON), {"run": f"claude -p --mcp-config '{MCP_JSON}' --settings '{SETTINGS_JSON}' 'go'"}, )) - action, cli = grant["agent_launches"] + action, = grant["agent_launches"] assert {item["name"]: item["value"] for item in action["settings"]} == { - "claude_args": f"--mcp-config '{MCP_PUBLISHED}' --allowedTools Read", + "claude_args": _digest(f"--mcp-config '{MCP_JSON}' --allowedTools Read"), "mcp_config": MCP_PUBLISHED, "settings": SETTINGS_PUBLISHED, } - assert {item["name"]: item["value"] for item in cli["settings"]} == { - "--mcp-config": f"'{MCP_PUBLISHED}'", - "--settings": SETTINGS_PUBLISHED, - } + assert grant["unread_agent_runs"] == [{"job": "review", "step": "steps[1]", "agent": "claude"}] text = json.dumps(grant) for canary in JSON_CANARIES: assert canary not in text assert hashlib.sha256(canary.encode()).hexdigest() not in text - assert uncompared_agent_launch_texts(grant) == [] and _uncompared_workflow_text(grant) is None + assert _uncompared_workflow_text(grant) is None + + +@pytest.mark.parametrize( + "value", + [ + f"--allowedTools Read --settings='{SETTINGS_JSON}'", + f"--allowedTools Read --mcp-config='{MCP_JSON}'", + "--allowedTools Read --settings=./ci/settings.json", + "--allowedTools Read --mcp-config .mcp.json", + ], + ids=["settings-attached-json", "mcp-config-attached-json", "settings-path", "mcp-config-path"], +) +def test_a_settings_or_mcp_config_flag_leaves_claude_args_unread(value): + """#823 review cycle 4 scope: such a value is never published, however it is attached.""" + + launch, = _launches(_workflow(_agent(value))) + assert launch["settings"] == [{"name": "claude_args", "value": _digest(value), "unresolved_reason": "unread_arguments"}] + for canary in JSON_CANARIES: + assert canary not in json.dumps(launch) -def test_a_value_attached_to_its_flag_is_withheld_as_a_separate_word_is(): - """#823 review F1: `--settings=…`, `--mcp-config=…` and codex's `-c` published verbatim.""" +def test_the_word_after_a_secret_named_argument_is_withheld(): + """As the host readers withhold it among an MCP server's arguments.""" grant = _grant(_workflow( - _agent(f"--allowedTools Read --settings='{SETTINGS_JSON}' --mcp-config='{MCP_JSON}'"), - {"uses": "openai/codex-action@v1", "with": { - "codex-args": '-cmcp_servers.db.env.REGION="canary-short" -c=mcp_servers.x.env.T=canary-eq --json', - }}, + _agent("--allowedTools Read --token canary-arg-1 --max-turns 5"), + {"run": "claude -p --allowedTools Read password canary-arg-2 go"}, )) - claude, codex = grant["agent_launches"] - - assert claude["settings"] == [{ - "name": "claude_args", - "value": f"--allowedTools Read '--settings={SETTINGS_PUBLISHED}' '--mcp-config={MCP_PUBLISHED}'", - "unresolved_reason": None, - }] - assert codex["settings"] == [{ - "name": "codex-args", - "value": "'-cmcp_servers.db.env.REGION=' '-c=mcp_servers.x.env.T=' --json", - "unresolved_reason": None, + action, run = grant["agent_launches"] + assert action["settings"] == [{ + "name": "claude_args", "value": "--allowedTools Read --token --max-turns 5", + "unresolved_reason": "redacted", }] - text = json.dumps(grant) - for canary in (*JSON_CANARIES, "canary-short", "canary-eq"): - assert canary not in text - # Each spelling compares as the host readers compare it: rotating an env value is quiet. - rotated = SETTINGS_JSON.replace("hunter2-canary", "rotated") - assert _rows( - _workflow(_agent(f"--settings='{SETTINGS_JSON}'")), _workflow(_agent(f"--settings='{rotated}'")) - ) == [] + assert run["settings"] == [ + {"name": "--allowedTools", "value": "Read password go", "unresolved_reason": "redacted"}, + ] + assert "canary-arg" not in json.dumps(grant) + # An edit inside what is redacted is not reported, so it is named as a limit. + assert [limit.split(";")[0] for limit in uncompared_agent_launch_texts(grant)] == [ + "the claude_args value of the agent launch at review/steps[0] (anthropics/claude-code-action) contains " + "credential-shaped text", + "the --allowedTools value of the agent launch at review/steps[1] (claude) contains credential-shaped text", + ] def test_a_withheld_json_value_compares_as_the_host_readers_compare_it(): @@ -1513,12 +1426,14 @@ def settings(env): assert "three" not in row.after -def test_a_codex_config_override_withholds_env_header_and_secret_values(): +def test_a_codex_config_override_publishes_its_key_and_withholds_its_value(): + """#823 review cycle 4: only a rule-bearing key's value, the approval policy and the model are published.""" + run = ( - "codex exec -c 'mcp_servers.db.env.TOKEN=\"canary-cfg\"' " - "-c 'mcp_servers.gh={command=\"gh\", env={GH_TOKEN=\"canary-inline\"}}' -c model=o3 'go'" + "codex exec -c mcp_servers.db.env.TOKEN=canary-cfg -c mcp_servers.gh.command=/opt/canary-bin/gh " + "-c shell_environment_policy.set.LEVEL=canary-env -c model=o3 go" ) - action_args = '["-c", "mcp_servers.db.env.TOKEN=\\"canary-array\\"", "--yolo"]' + action_args = "-cmcp_servers.db.env.REGION=canary-short -c=mcp_servers.x.url=https://canary.example/p --yolo" grant = _grant(_workflow( {"run": run}, {"uses": "openai/codex-action@v1", "with": {"codex-args": action_args}}, )) @@ -1526,58 +1441,50 @@ def test_a_codex_config_override_withholds_env_header_and_secret_values(): assert cli["settings"] == [ {"name": "--config", "value": "mcp_servers.db.env.TOKEN=", "unresolved_reason": None}, - {"name": "--config", "value": 'mcp_servers.gh={"command":"gh","env":{"GH_TOKEN":""}}', + {"name": "--config", "value": f"mcp_servers.gh.command={_digest('/opt/canary-bin/gh')}", "unresolved_reason": None}, {"name": "--config", "value": "model=o3", "unresolved_reason": None}, + {"name": "--config", "value": f"shell_environment_policy.set.LEVEL={_digest('canary-env')}", + "unresolved_reason": None}, ] assert action["settings"] == [{ - "name": "codex-args", "value": '["-c","mcp_servers.db.env.TOKEN=","--yolo"]', + "name": "codex-args", + "value": ( + "-cmcp_servers.db.env.REGION= " + f"-c=mcp_servers.x.url={_digest('https://canary.example/p')} --yolo" + ), "unresolved_reason": None, }] assert action["widening_rules"] == [{"rule": "bypass_approvals_and_sandbox", "setting": "codex-args"}] assert "canary" not in json.dumps(grant) - - -def test_a_codex_config_table_that_does_not_parse_is_withheld_and_named(): - """#823 review (P3): string-argv keeps the quotes of `--config='k={…}'`, so it does not parse as TOML.""" - - codex_args = "--config='mcp_servers.db={command=\"x\", env={T=\"canary-quoted\"}}' --json" - grant = _grant(_workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}})) - launch, = grant["agent_launches"] - - assert launch["settings"] == [{"name": "codex-args", "value": None, "unresolved_reason": "unparsed_json"}] - assert "canary-quoted" not in json.dumps(grant) - limit, = uncompared_agent_launch_texts(grant) - assert "a codex `--config` table or array, and does not parse" in limit + # Rotating a redacted value is quiet; editing a withheld one is a change. + assert _rows(_workflow({"run": run}), _workflow({"run": run.replace("canary-cfg", "rotated")})) == [] + row, = _rows(_workflow({"run": run}), _workflow({"run": run.replace("canary-env", "edited")})) + assert row.direction == "changed" def test_text_that_starts_like_json_and_does_not_parse_is_withheld_and_named(): - # shell-quote strips the double quotes of an unquoted JSON word, so the - # action reads it as a path; its values cannot be told from its keys. - value = '--mcp-config {"mcpServers":{"db":{"env":{"T":"canary-unquoted"}}}} --dangerously-skip-permissions' - grant = _grant(_workflow(_agent(value))) + value = '{"env": {"T": "canary-unparsed"}' + grant = _grant(_workflow(_agent(settings=value))) launch, = grant["agent_launches"] - assert launch["settings"] == [{"name": "claude_args", "value": None, "unresolved_reason": "unparsed_json"}] - # The rule is read from the declared text, so withholding it hides no rule. - assert launch["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] - assert "canary-unquoted" not in json.dumps(grant) + setting = next(item for item in launch["settings"] if item["name"] == "settings") + assert setting == {"name": "settings", "value": None, "unresolved_reason": "unparsed_json"} + assert "canary-unparsed" not in json.dumps(grant) limit, = uncompared_agent_launch_texts(grant) - assert limit.startswith("the claude_args value of the agent launch at review/steps[0] (anthropics/claude-code-action)") - assert "starts like JSON, or a codex `--config` table or array, and does not parse" in limit + assert limit.startswith("the settings value of the agent launch at review/steps[0] (anthropics/claude-code-action)") + assert "holds text that starts like JSON and does not parse" in limit assert _uncompared_workflow_text(grant) is None def test_a_url_path_is_withheld_while_the_rest_of_the_setting_and_a_rule_beside_it_are_read(): - before = _workflow(_agent("--append-system-prompt 'Follow https://example.com/style-guide' --allowedTools Read")) - after = _workflow(_agent( - "--append-system-prompt 'Follow https://example.com/style-guide' --dangerously-skip-permissions" - )) + before = _workflow(_agent("--append-system-prompt Follow https://example.com/style-guide --allowedTools Read")) + after = _workflow(_agent("--append-system-prompt Follow https://example.com/style-guide --dangerously-skip-permissions")) row, = _rows(before, after) assert (row.direction, row.expands) == ("widened", True) assert ( - "claude_args: --append-system-prompt 'Follow https://example.com/' " + "claude_args: --append-system-prompt Follow https://example.com/ " "--dangerously-skip-permissions" ) in row.after assert "style-guide" not in row.before + row.after @@ -1585,20 +1492,6 @@ def test_a_url_path_is_withheld_while_the_rest_of_the_setting_and_a_rule_beside_ assert _uncompared_workflow_text(_grant(after)) is None -def test_a_quoted_url_in_a_json_array_codex_args_element_keeps_the_array_valid(): - """#823 review cycle 3: withholding the URL keeps the element's escaped closing quote.""" - - codex_args = json.dumps(["-c", 'mcp_servers.x.url="https://h2.example.com/p?token=canary-q"', "--json"]) - launch, = _launches(_workflow({"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}})) - setting, = launch["settings"] - - assert json.loads(setting["value"]) == [ - "-c", 'mcp_servers.x.url="https://h2.example.com/"', "--json", - ] - assert setting["unresolved_reason"] is None - assert "canary-q" not in json.dumps(launch) - - def test_a_marketplace_url_compares_by_scheme_and_host_as_an_mcp_server_url_does(): def marketplace(url): return _workflow(_agent(plugin_marketplaces=url)) @@ -1622,23 +1515,23 @@ def test_credential_shaped_text_in_a_setting_is_published_redacted_and_named_whi """ grant = _grant(_workflow( - _agent("--append-system-prompt 'use token=ARGCANARY' --dangerously-skip-permissions", + _agent("--append-system-prompt use token=ARGCANARY --dangerously-skip-permissions", plugin_marketplaces="https://robot:PWCANARY@github.com/org/repo.git"), {"uses": "actions/checkout@v4", "with": {"ref": "token=REFCANARY"}}, - {"run": "claude -p --settings ghp_" + "A" * 36 + " 'review token=SECRETCANARY'"}, + {"run": "claude -p --permission-prompt-tool ghp_" + "A" * 36 + " review token=SECRETCANARY"}, )) action, cli = grant["agent_launches"] assert action["settings"] == [ {"name": "claude_args", - "value": "--append-system-prompt 'use token=' --dangerously-skip-permissions", + "value": "--append-system-prompt use token= --dangerously-skip-permissions", "unresolved_reason": "redacted"}, {"name": "plugin_marketplaces", "value": "https://github.com/", "unresolved_reason": "redacted"}, ] # The rule is read from the declared text, so redaction does not hide it. assert action["widening_rules"] == [{"rule": "bypass_permissions", "setting": "claude_args"}] assert cli["settings"] == [ - {"name": "--settings", "value": "[REDACTED:github_token]", "unresolved_reason": "redacted"}, + {"name": "--permission-prompt-tool", "value": "[REDACTED:github_token]", "unresolved_reason": "redacted"}, ] assert grant["checkout_refs"] == [ {"job": "review", "step": "steps[1]", "ref": "token=", "unresolved_reason": "redacted"}, @@ -1653,7 +1546,7 @@ def test_credential_shaped_text_in_a_setting_is_published_redacted_and_named_whi for name, where in ( ("claude_args", "review/steps[0] (anthropics/claude-code-action)"), ("plugin_marketplaces", "review/steps[0] (anthropics/claude-code-action)"), - ("--settings", "review/steps[2] (claude)"), + ("--permission-prompt-tool", "review/steps[2] (claude)"), ) ] assert _uncompared_workflow_text(grant) == ( @@ -1664,18 +1557,16 @@ def test_credential_shaped_text_in_a_setting_is_published_redacted_and_named_whi #: Prose the #802 label redaction rewrites, as security-review prompts write it: #: ``claude_args``, and a ``run:`` whose prompt is a variadic flag's value. PROSE = [ - ('--append-system-prompt "Never print bearer tokens in review comments" --allowedTools Read', - "claude -p --allowedTools Read 'Never print bearer tokens in review comments'"), - ('--append-system-prompt "Flag Authorization: headers logged in plain text" --allowedTools Read', - "claude -p --allowedTools Read 'Flag Authorization: headers logged in plain text'"), - ('--allowedTools "Bash(curl -H Authorization:*)"', - "claude -p --allowedTools 'Bash(curl -H Authorization:*)' -- 'go'"), - ('--append-system-prompt "check the secret=... assignment" --allowedTools Read', - "claude -p --allowedTools Read 'check the secret=... assignment'"), + ("--append-system-prompt Never print bearer tokens in review comments --allowedTools Read", + "claude -p --allowedTools Read Never print bearer tokens in review comments"), + ("--append-system-prompt Flag Authorization: headers logged in plain text --allowedTools Read", + "claude -p --allowedTools Read Flag Authorization: headers logged in plain text"), + ("--append-system-prompt check the secret=... assignment --allowedTools Read", + "claude -p --allowedTools Read check the secret=... assignment"), ] -@pytest.mark.parametrize(("prose", "run"), PROSE, ids=["bearer", "authorization", "tool-rule", "assignment"]) +@pytest.mark.parametrize(("prose", "run"), PROSE, ids=["bearer", "authorization", "assignment"]) def test_prose_the_label_redaction_rewrites_is_a_named_limit_that_refuses_nothing(prose, run): """#823 review F2: such prose used to make the whole workflow a blocking limit.""" @@ -1716,13 +1607,16 @@ def test_token_shaped_job_and_step_labels_are_redacted_in_every_entry(): job = "ghp_" + "B" * 36 grant = _grant(_workflow(jobs={job: {"steps": [ {"name": "Pull docker://ci:hunter2@gcr.io/x", "uses": "actions/checkout@v4"}, - {"name": "Run ghp_" + "C" * 36, "run": "claude -p 'go'"}, + {"name": "Run ghp_" + "C" * 36, "run": "claude -p go"}, + {"name": "Unread ghp_" + "E" * 36, "run": "npm ci && claude -p go"}, ]}})) - text = json.dumps({key: grant[key] for key in ("agent_launches", "checkout_refs")}) + text = json.dumps({key: grant[key] for key in ("agent_launches", "checkout_refs", "unread_agent_runs")}) assert "ghp_" not in text and "hunter2" not in text assert grant["checkout_refs"][0]["step"] == "Pull docker://@gcr.io/x" assert grant["agent_launches"][0]["job"] == "[REDACTED:github_token]" + assert grant["unread_agent_runs"][0]["job"] == "[REDACTED:github_token]" + assert "ghp_" not in " ".join(uncompared_agent_launch_texts(grant)) # --- saved baselines --------------------------------------------------------------- @@ -1766,6 +1660,7 @@ def _v06_baseline(workspace: Path): legacy["host_grants_schema_version"] = "0.6" for grant in legacy["inventory"]["grants"]: grant.pop("agent_launches", None) + grant.pop("unread_agent_runs", None) grant.pop("checkout_refs", None) legacy["inventory_sha256"] = host_grants_sha256(legacy["inventory"]) return current, legacy @@ -1862,6 +1757,8 @@ def test_a_current_baseline_compares_agent_launches_and_validates_against_the_sc inventory = host_audit_inventory(tmp_path) baseline = build_host_grants_baseline(inventory) assert baseline["host_grants_schema_version"] == "0.7" + workflow, = [grant for grant in baseline["inventory"]["grants"] if grant["kind"] == "workflow"] + assert workflow["unread_agent_runs"] == [{"job": "review", "step": "steps[2]", "agent": "claude"}] for name, payload in (("inventory", inventory), ("baseline", baseline)): schema = json.loads((ROOT / f"docs/host-grants-{name}-schema.v0.7.json").read_text()) Draft202012Validator(schema).validate(payload) @@ -1874,7 +1771,7 @@ def test_a_current_baseline_compares_agent_launches_and_validates_against_the_sc Draft202012Validator(schema).validate(drift) -def test_an_unresolved_launch_is_a_non_blocking_limit_that_leaves_coverage_complete(tmp_path): +def test_an_unread_run_is_a_non_blocking_limit_that_leaves_coverage_complete(tmp_path): from agents_shipgate.cli.host_audit import host_audit_inventory _write(tmp_path, {SOURCE: _yaml(_workflow({"run": "npm ci && claude -p 'go'"}))}) @@ -1884,35 +1781,37 @@ def test_an_unresolved_launch_is_a_non_blocking_limit_that_leaves_coverage_compl assert github["status"] == "complete" issue, = [item for item in inventory["issues"] if item["host"] == "github"] assert (issue["kind"], issue["blocking"]) == ("unsupported", False) - assert "review/steps[0] (claude)" in issue["message"] + assert issue["message"].startswith( + "the `run:` at review/steps[0] mentions claude and is not read as an agent launch" + ) -def test_the_quoted_substitution_of_the_review_is_a_row_in_diff_and_a_coverage_issue(tmp_path): - """#823 review: `--body "$(claude -p …)"` gave no row, no launch and no coverage issue.""" +@pytest.mark.parametrize("run", UNREAD_RUNS[:8], ids=[f"review-cycle-4-{index}" for index in range(8)]) +def test_the_run_forms_of_review_cycle_4_give_no_row_and_are_a_named_coverage_issue(tmp_path, run): + """#823 review cycle 4 (a) end to end: named in `audit --host`, never a row, none of the text published.""" from agents_shipgate.cli.host_audit import host_audit_inventory - repo = _repo(tmp_path, {SOURCE: _yaml(_reproduction())}) + permissions = {"contents": "write", "pull-requests": "write"} + base = _workflow({"uses": "actions/checkout@v4"}, trigger="issue_comment", permissions=permissions) + head = _workflow({"uses": "actions/checkout@v4"}, {"run": run}, trigger="issue_comment", permissions=permissions) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) _git(repo, "checkout", "-qb", "change") - _write(repo, {SOURCE: _yaml(_reproduction(run=SUBSTITUTED.format(flags="--dangerously-skip-permissions")))}) - _git(repo, "commit", "-qam", "comment with a review") - - row, = _diff(repo)["rows"] - assert (row["direction"], row["expands"]) == ("changed", False) - assert "a step launches an agent in a form this audit does not read (review/steps[2])" in row["why"] - text = CliRunner().invoke(app, ["diff", "--workspace", str(repo), "--base", "main"]) - assert text.exit_code == 0, text.output - assert "No static host-grant changes detected" not in text.output - assert "review/steps[2]" in text.output and "dangerously" not in text.output + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "an agent step") + assert _diff(repo)["rows"] == [] inventory = host_audit_inventory(repo) workflow, = [grant for grant in inventory["grants"] if grant.get("kind") == "workflow"] - assert [(item["step"], item["form"], item["unresolved_reason"]) for item in workflow["agent_launches"]] == [ - ("steps[1]", "read", None), ("steps[2]", "unresolved", "shell_expansion"), - ] - issue, = [item for item in inventory["issues"] if item["host"] == "github"] - assert (issue["kind"], issue["blocking"]) == ("unsupported", False) - assert "review/steps[2] (claude)" in issue["message"] + assert "agent_launches" not in workflow + issues = [item for item in inventory["issues"] if item["host"] == "github"] + assert issues and all(not item["blocking"] for item in issues) + assert all("review/steps[1] mentions" in item["message"] for item in issues) + audit = CliRunner().invoke(app, ["audit", "--host", "--workspace", str(repo)]) + assert "review/steps[1] mentions" in audit.output + joined = json.dumps(inventory) + audit.output + for text in ("dangerously", "yolo", "Review this PR", "review this change"): + assert text not in joined # --- the same row on every route --------------------------------------------------- @@ -1922,7 +1821,7 @@ def test_the_quoted_substitution_of_the_review_is_a_row_in_diff_and_a_coverage_i def pr(tmp_path): repo = _repo(tmp_path, {SOURCE: _yaml(_reproduction())}) _git(repo, "checkout", "-qb", "change") - _write(repo, {SOURCE: _yaml(_reproduction(claude_args='--permission-mode bypassPermissions --allowedTools "Bash(*)"'))}) + _write(repo, {SOURCE: _yaml(_reproduction(claude_args="--permission-mode bypassPermissions --allowedTools Bash"))}) _git(repo, "add", ".") _git(repo, "commit", "-qm", "bypass permissions") return repo @@ -1930,7 +1829,7 @@ def pr(tmp_path): def _assert_the_row(row: dict) -> None: assert row["subject"] == f"github {SOURCE}" - assert 'review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools "Read"' in row["before"] + assert "review/steps[1]: runs anthropics/claude-code-action with claude_args: --allowedTools Read" in row["before"] assert "--permission-mode bypassPermissions" in row["after"] assert (row["direction"], row["expands"]) == ("widened", True) assert "skips permission checks" in row["why"] @@ -2024,7 +1923,7 @@ def _assert_absent(joined: str, canaries) -> None: def test_no_canary_reaches_any_published_output(tmp_path): - """The #802 sweep for agent launches, JSON-shaped canaries included (#823 review F3).""" + """The #802 sweep for agent launches, JSON-shaped canaries included (#823 review F3, cycle 4).""" canary = "sk-ant-api03-" + "Z" * 40 job = "ghp_" + "D" * 36 @@ -2033,8 +1932,8 @@ def test_no_canary_reaches_any_published_output(tmp_path): "headers": {"X-API-Key": "canary-hdr-456", "Authorization": f"Bearer {canary}"}, }}}) codex_args = ( - "-c 'mcp_servers.db.env.TOKEN=\"canary-cfg-789\"' -cmcp_servers.db.env.REGION=canary-short-c " - "-c=mcp_servers.db.env.ZONE=canary-eq-c --full-auto" + "-c mcp_servers.db.env.TOKEN=canary-cfg-789 -cmcp_servers.db.env.REGION=canary-short-c " + "-c=mcp_servers.db.env.ZONE=canary-eq-c -c mcp_servers.db.command=/opt/canary-cmd-c --full-auto" ) # The `=` spellings of #823 review F1, beside the separate-word ones. equals = json.dumps({"env": {"DB_PASSWORD": "hunter2-eqcanary"}, "apiKeyHelper": "echo helper-eqcanary"}) @@ -2045,16 +1944,21 @@ def test_no_canary_reaches_any_published_output(tmp_path): head = _workflow(jobs={job: {"steps": [ {"name": "Pull docker://ci:" + "p4ssCANARY" + "@gcr.io/x", "uses": "actions/checkout@v4"}, _agent( - f"--mcp-config '{mcp}' --dangerously-skip-permissions", + "--allowedTools Read --dangerously-skip-permissions", settings=SETTINGS_JSON, mcp_config=mcp, plugin_marketplaces="https://github.com/canary-org/canary-repo.git", ), + # #823 review cycle 4: `claude_args` or a `run:` holding JSON, `--settings` + # or `--mcp-config` is not read, and publishes only a digest or nothing. + _agent(f"--mcp-config '{mcp}' --dangerously-skip-permissions"), _agent(f"--allowedTools Read --settings='{equals}' --mcp-config='{equals_mcp}'"), {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read --mcp-config '{mcp}' 'go'"}, + {"run": f"ANTHROPIC_API_KEY={canary} claude -p --allowedTools Read go"}, {"uses": "openai/codex-action@v1", "with": {"codex-args": codex_args}}, # #823 review C2-F1: an MCP server's arguments and a hook's command, # which the host readers never publish, in every spelling. - _agent(f"--allowedTools Read\n--mcp-config '{REMOTE_MCP_JSON}'", settings=HOOK_JSON), + _agent("--allowedTools Read", mcp_config=REMOTE_MCP_JSON, settings=HOOK_JSON), + _agent(f"--allowedTools Read\n--mcp-config '{REMOTE_MCP_JSON}'"), {"run": f"claude -p --settings '{HOOK_JSON}' --mcp-config '{REMOTE_MCP_JSON}' 'go'"}, {"run": f"codex exec -c '{REMOTE_CODEX_CONFIG}' 'go'"}, ]}}) @@ -2074,11 +1978,12 @@ def test_no_canary_reaches_any_published_output(tmp_path): _assert_absent(joined, ( canary, "p4ssCANARY", job, "canary-org", "canary-repo", "canary-cfg-789", *JSON_CANARIES, "hunter2-eqcanary", "helper-eqcanary", "tok-eqcanary", "hdr-eqcanary", "canary-short-c", "canary-eq-c", - *SHAPE_CANARIES, + "canary-cmd-c", *SHAPE_CANARIES, )) assert "runs claude -p with --allowedTools Read" in joined assert "mcp_servers.db.env.TOKEN=" in joined assert '"command":"npx"' in joined and '"Stop":[{"hooks":[{"command":"' --dangerously-skip-permissions" in row["after"] + assert "--append-system-prompt use token= --dangerously-skip-permissions" in row["after"] audit = json.loads(CliRunner().invoke(app, ["audit", "--host", "--workspace", str(repo), "--json"]).stdout) issue, = [item for item in audit["issues"] if item["host"] == "github"] assert (issue["kind"], issue["blocking"]) == ("unsupported", False) From 321d17231c5f8706980709806d2ed725848192c9 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 11:00:23 -0700 Subject: [PATCH 12/14] Address review cycle 5 on agent launches in CI (#823) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit C5-F1: the move rule took a launch that only stopped being read to have left its job. With job a running `claude -p --dangerously-skip-permissions Review` in the base, quoting its prompt (an unread step) or running it through `npx` while job b added the same plain launch gave a `changed` row saying the launch "moved between jobs (a → b)" and "already met that rule in the job it left", with no widening signal. Job a still runs it. - `same_launch` now refuses a move while the losing job may still run the launch in a form this reader does not read: an unread step of that agent stands at a step label the lost launch held, the job has more unread steps of that agent than before (the launch may have moved to another index), or one of its read launches of that agent holds an expression or an unread argument input in an input the rule is read from. The review's M5 and M6 are now `widened` and name "a step no longer declares an agent launch this audit reads (a/steps[0])"; a real move, a rename, a swap, and a move beside an unread step the job already had stay moves. C5-F2: docs/integrations.md said an unread `run:` is a non-widening row that diff and the PR comment show. It gives no row; only an unread argument input is a row. The sentence is split accordingly. Nonblocking items fixed: - `_job_secrets` walks each container once, without recursion. A job `env` holding itself through a YAML alias raised RecursionError, and a ten-way fan-out eight levels deep did not finish; both now read in well under a second. - A `settings` or `mcp_config` value that is neither a JSON object nor a plain file path (path characters, and a `${{ }}` only as a plain context reference) publishes only a `` digest. A comment line before JSON published the `env` value that JSON held. - The last of a repeated `--permission-mode` counts, in `claude_args` and in a plain `run:`, as parse-sdk-options and the CLI keep it. - The docs/distribution-surfaces.md capability_diff row no longer says an unread argument input or an unresolved launch is named only by the inventory and audit --host: a row reporting its launch names it too. The support page, STABILITY (Direction and What is withheld) and the CHANGELOG entry say the same. Tests: tests/test_workflow_agent_launches.py grows to 290 cases. The five new move-rule widening cases, the alias test, the two json-or-path cases and the four --permission-mode cases fail on the previous head; the moved-beside-an-unread-step case and an end-to-end diff/audit test of M5 and M6 are added beside them. --- CHANGELOG.md | 2 +- STABILITY.md | 4 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 18 +++- docs/integrations.md | 7 +- src/agents_shipgate/core/host_grants.py | 108 +++++++++++++++----- tests/test_workflow_agent_launches.py | 129 +++++++++++++++++++++++- 7 files changed, 229 insertions(+), 41 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ceaae396c..ca005ae53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, and one the job's unread step or unread input may already have met is named, not claimed, while an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, and one the job's unread step or unread input may already have met is named, not claimed, while an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index b2247d879..ffcc152e0 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -365,8 +365,8 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin - **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets. A `run:` is an agent launch only when it is one line of plain words (letters, digits and `_ . / : = , % + -`, separated by spaces or tabs), run by `bash`, `sh` or no declared `shell:`, whose program, after any `NAME=value` assignments (skipped, never published), has the file name `claude` and passes `-p`/`--print`, or is `codex` followed by `exec` (`codex e`); it lists its documented permission flags under their primary spelling, and a `--settings` or `--mcp-config` flag leaves it unread. Any other `run:` that mentions `claude` or `codex` as a word of its own is one `unread_agent_runs` entry per agent CLI it mentions, holding only `job`, `step` and `agent`. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). - **Argument inputs.** `claude_args` and `codex-args` are read only when they are a plain list of words (the characters above and parentheses, separated by blanks or newlines, with no `--settings` or `--mcp-config` flag), and published as those words one space apart, so reformatting one is quiet. Any other value — a quote, a `${{ }}` expression, `$`, a backtick, a backslash, a `#` comment, a shell separator or redirection, a glob, JSON, a `--settings` or `--mcp-config` flag, any other character — is `unresolved_reason: unread_arguments`: its `value` is ``, a short digest, so an edit is a `changed` row showing the digest, and no documented widening rule is read from it. So a quoted `claude_args` such as the #823 reproduction's `--allowedTools "Read"` is compared only by its digest. - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. `unread_agent_runs` is never compared: adding, removing or editing an unread step gives no row. -- **What is withheld.** A JSON object in a `settings` or `mcp_config` input publishes, as canonical JSON, its key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so never published. A codex `--config` override in a plain list of words (`-c`, `--config=`, `-c`, `-c=`) publishes its key, and its value as `` under `env`, `headers` or a secret-named key, as written for `sandbox_mode`, `default_permissions`, `approval_policy` and `model`, and as a `` digest under any other key, such as an MCP server's `command` or `url`. The word after a secret-named word such as `--token` or `password` is ``, as the host readers redact it among an MCP server's arguments, and the value is then published redacted (`redacted`, below). Other argument text — a prompt word, a flag's value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path, as an MCP server's URL does (#723), so a change only to such a URL's path is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text in a `settings` or `mcp_config` input that starts like JSON and does not parse is withheld whole (`unparsed_json`). -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **What is withheld.** A JSON object in a `settings` or `mcp_config` input publishes, as canonical JSON, its key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so never published. A codex `--config` override in a plain list of words (`-c`, `--config=`, `-c`, `-c=`) publishes its key, and its value as `` under `env`, `headers` or a secret-named key, as written for `sandbox_mode`, `default_permissions`, `approval_policy` and `model`, and as a `` digest under any other key, such as an MCP server's `command` or `url`. The word after a secret-named word such as `--token` or `password` is ``, as the host readers redact it among an MCP server's arguments, and the value is then published redacted (`redacted`, below). Other argument text — a prompt word, a flag's value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path, as an MCP server's URL does (#723), so a change only to such a URL's path is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text in a `settings` or `mcp_config` input that starts like JSON and does not parse is withheld whole (`unparsed_json`); one that is neither a JSON object nor a plain file path (path characters, and a `${{ }}` expression only as a plain context reference) publishes only a `` digest. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, the last of a repeated `--permission-mode` counting, as the action and the CLI keep it, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it, nor has one whose job still has an unread step of that agent where it stood, more unread steps of it than before, or a launch of it holding an expression or an unread argument input the rule is read from), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. An unread step is not an agent step here. A removed workflow gets none. - **Unread and unreadable values are a named limit, not a blocking one.** An unread `run:` step, an argument input that is not a plain list of words (`unread_arguments`), an agent action whose `with:` is not a mapping (`form: unresolved`, `inputs_not_a_mapping`), a setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), a setting holding credential-shaped text (`redacted`), and a ref that is not a string each record a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`. An unread `run:` step is never compared, so it gives no row whatever is edited, and it never says the step starts or does not start an agent. For the others, adding, removing or re-forming the entry, or its gaining a rule, is still a row; only an edit inside it that gains no rule is not reported (an unread argument input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them: a change that only adds an unread step prints `No static host-grant changes detected`, and `audit --host` names the step. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index 2302b3011..8e43ed4d4 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unread argument input, an unresolved launch, an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, an unread argument input or an unresolved launch is named there and in the `why` of a row reporting its launch, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 42da440f1..f3c5b6cf2 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -146,7 +146,7 @@ never guessed at. Four things are listed on the workflow grant, each naming its | Action | Inputs compared as text | Documented widening | |---|---|---| - | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | a plain `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions`; `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | + | `anthropics/claude-code-action` | `additional_permissions`, `allowed_bots`, `allowed_non_write_users`, `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, and the earlier `allowed_tools`, `disallowed_tools`, `mcp_config` | a plain `claude_args` gains `--dangerously-skip-permissions` or `--permission-mode bypassPermissions` (the last `--permission-mode` counting); `settings`, written as JSON, gains `defaultMode: bypassPermissions` (under `permissions`, else at the top, as the settings reader reads `.claude/settings.json`); `allowed_bots` (any bot) or `allowed_non_write_users` (any user) gains a `*` entry | | `anthropics/claude-code-base-action`, also published as `anthropics/claude-code-action/base-action` | `claude_args`, `plugin_marketplaces`, `plugins`, `settings`, `allowed_tools`, `disallowed_tools`, `mcp_config` | `claude_args` and `settings` as above | | `openai/codex-action` | `allow-bot-users`, `allow-bots`, `allow-users`, `codex-args`, `permission-profile`, `safety-strategy`, `sandbox` | `sandbox` becomes `danger-full-access`; `permission-profile` becomes `:danger-full-access`, Codex's reserved name for its built-in full-access profile; `safety-strategy` becomes `unsafe`; a plain `codex-args` gains `--dangerously-bypass-approvals-and-sandbox` (`--yolo`) or `--sandbox danger-full-access` (`-s`, attached or not); `allow-users` gains a `*` entry. A sandbox `--config` override in `codex-args` meets none: after `codex-args` the action appends its own `--sandbox`, or its own `default_permissions` override for a `permission-profile`, which takes precedence | @@ -190,7 +190,8 @@ never guessed at. Four things are listed on the workflow grant, each naming its `--allowedTools`/`--allowed-tools`, `--disallowedTools`/`--disallowed-tools`, `--add-dir` and `--permission-prompt-tool`; gaining `--dangerously-skip-permissions` or `--permission-mode bypassPermissions` - (one rule, so moving between the spellings is not a widening) widens, and a + (one rule, so moving between the spellings is not a widening; of a repeated + `--permission-mode`, the last counts, as the CLI keeps it) widens, and a `--settings` or `--mcp-config` flag makes the step unread. For `codex exec`: `--sandbox`/`-s`, `--dangerously-bypass-approvals-and-sandbox`/`--yolo`, `--approve-for-me`/`--not-so-yolo`, `--dangerously-bypass-hook-trust`, @@ -281,7 +282,12 @@ names the rule and step. Three gains are named in the `why` and not claimed: gaining it while the first job still exists, is claimed: that job may still run its launch in a form this audit does not read (`npx`, quoting), so a launch that only stops being read has not left it, and a launch edited in - place into the one that job had gains the rule. + place into the one that job had gains the rule. A launch has not left a job + that still has an unread step of that agent where the launch stood, more + unread steps of it than before, or a launch of it holding an expression or + an unread argument input the rule is read from — so quoting the prompt of + `claude -p --dangerously-skip-permissions Review` in one job while another + job adds that plain step is a widening. Any other edit — `--allowedTools Read` to `--allowedTools Bash`, `acceptEdits`, a new plugin, an argument input this audit does not read, a @@ -326,6 +332,12 @@ in canonical JSON: for `npx -y some-server`, a URL's digest for its query. A URL's path is neither published nor compared, as an MCP server's is not (#723). +A `settings` or `mcp_config` value that is not a JSON object is published as +written only when it is a plain file path: path characters, and any `${{ }}` +expression in it a plain context reference. Any other value, such as a comment +line before the JSON, publishes only ``, a digest, so an edit to it +is still a `changed` row and none of its text is published. + JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so it is never published: the argument input or step is not read at all. A codex `--config` override in a plain list of words — `-c key=value`, diff --git a/docs/integrations.md b/docs/integrations.md index e3e4c8c4a..099b2c586 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -226,8 +226,11 @@ a different action reference, such as a pinned SHA to `@main`, is a non-widening row, so the hook stays quiet about it; `diff` and the PR comment still show it. The same holds for an agent launch in a workflow whose settings change without gaining a documented widening rule, for a checkout's ref, and -for a `run:` or argument input the audit does not read, which it names as a -limit in `audit --host`. One that gains a rule, such as a plain `claude_args` +for an edit to an argument input the audit does not read, which is compared by +a digest, named in the row and named as a limit in `audit --host`. A `run:` +that mentions an agent CLI and that the audit does not read is different: it +gives no row in `diff` or the PR comment, whatever is edited, and is named only +as a limit in `audit --host`. One that gains a rule, such as a plain `claude_args` gaining `--dangerously-skip-permissions` on any of its lines, widens, and the hook announces it (#823), unless the launch may be a step of that job the audit did not read, rewritten, or the job's launch held before a `${{ }}` expression diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index d2ebfdbdc..1a87f6274 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -2088,6 +2088,28 @@ def _withheld_word(word: str) -> str | None: return _withheld_json(loaded) +#: The action inputs that hold JSON or a file path (#823 review). +_JSON_OR_PATH_INPUTS = frozenset({"mcp_config", "settings"}) +#: A file path as it may be published: plain path characters, and any +#: ``${{ }}`` expression in it a plain context reference. +_PLAIN_PATH_RE = re.compile(r"(?:[A-Za-z0-9_./~@+,:=%-]|\$\{\{[ A-Za-z0-9_.\-\[\]*]*\}\})*") + + +def _withheld_json_or_path(text: str) -> str | None: + """A ``settings`` or ``mcp_config`` value as it may be published. + + A JSON object by :func:`_withheld_word`, and a plain file path as + written. Any other text is neither, so the action rejects it, but it may + still hold what a host reader withholds, such as an ``env`` value after a + comment line: it publishes only ````, a digest, so an edit to + it is still a change (#823 review cycle 5). + """ + + if text.lstrip().startswith("{") or _PLAIN_PATH_RE.fullmatch(text): + return _withheld_word(text) + return _withheld_string(text) + + #: The codex ``--config`` keys whose value is published as written: the #: settings a documented widening rule reads (``sandbox_mode``, #: ``default_permissions``), the approval policy and the model. Any other @@ -2255,7 +2277,9 @@ def _published_setting(name: str, value: Any, *, arguments: str | None = None) - otherwise it is ``unread_arguments`` and publishes only ````, a digest, so an edit to it is still a change while none of its text is published (#823 review cycle 4). Every other input is one value, a JSON - object published by :func:`_withheld_json`. + object published by :func:`_withheld_json`; a ``settings`` or + ``mcp_config`` value that is neither JSON nor a plain path publishes only + a digest (:func:`_withheld_json_or_path`). """ text = _setting_text(value) @@ -2267,7 +2291,9 @@ def _published_setting(name: str, value: Any, *, arguments: str | None = None) - return {"name": name, "value": _withheld_string(text), "unresolved_reason": "unread_arguments"} shown, redacted = _withheld_words(words, family=arguments) return _published_text(name, " ".join(shown), redacted=redacted) - setting = _published_text(name, _withheld_word(text)) + setting = _published_text( + name, _withheld_json_or_path(text) if name in _JSON_OR_PATH_INPUTS else _withheld_word(text) + ) # Read off the declared text: redaction may rewrite the expression away. return {**setting, "holds_expression": True} if holds_expression(text) else setting @@ -2300,20 +2326,18 @@ def _job_secrets(job: dict[Any, Any], workflow_env: Any) -> list[str]: """ names: set[str] = set() - - def walk(value: Any) -> None: + # Walked once per container, without recursion: a YAML alias can make a + # mapping hold itself, or fan one out many times over (#823 review cycle 5). + seen: set[int] = set() + pending: list[Any] = [job, workflow_env] + while pending: + value = pending.pop() if isinstance(value, str): for expression in _EXPRESSION_RE.findall(value): names.update(_SECRET_REFERENCE_RE.findall(expression)) - elif isinstance(value, dict): - for item in value.values(): - walk(item) - elif isinstance(value, list): - for item in value: - walk(item) - - walk(job) - walk(workflow_env) + elif isinstance(value, (dict, list)) and id(value) not in seen: + seen.add(id(value)) + pending.extend(value.values() if isinstance(value, dict) else value) return sorted({published_workflow_label(name) for name in names}) @@ -2487,18 +2511,23 @@ def _claude_action_rules(words: list[str]) -> set[str]: Read as ``parse-sdk-options.ts`` reads them: a word starting with ``--`` is always a flag and never another flag's value, so ``--dangerously-skip-permissions`` counts wherever it stands, and - ``--permission-mode`` takes the next word unless that starts with ``--``. + ``--permission-mode`` takes the next word unless that starts with ``--``, + the last one counting, as it and the CLI keep the last value of a + repeated option (#823 review cycle 5). """ rules: set[str] = set() + mode: str | None = None for index, word in enumerate(words): name, equals, attached = word.partition("=") following = words[index + 1] if index + 1 < len(words) else "" value = attached if equals else ("" if following.startswith("--") else following) if name == "--dangerously-skip-permissions": rules.add("bypass_permissions") - elif name == "--permission-mode" and value == "bypassPermissions": - rules.add("bypass_permissions") + elif name == "--permission-mode": + mode = value + if mode == "bypassPermissions": + rules.add("bypass_permissions") return rules @@ -2573,12 +2602,13 @@ def _flag_rules( rules: set[tuple[str, str]] = set() if family == "codex" and config_selects_sandbox and _codex_config_full_access(flags): rules.add(("danger_full_access", "--config")) + # The CLI keeps the last value of a repeated `--permission-mode` (#823 review cycle 5). + modes = [values[0] if values else None for name, _arity, values in flags if name == "--permission-mode"] + if family == "claude" and modes and modes[-1] == "bypassPermissions": + rules.add(("bypass_permissions", "--permission-mode")) for name, _arity, values in flags: value = values[0] if values else None - if family == "claude" and ( - name == "--dangerously-skip-permissions" - or (name == "--permission-mode" and value == "bypassPermissions") - ): + if family == "claude" and name == "--dangerously-skip-permissions": rules.add(("bypass_permissions", name)) if family == "codex" and name == "--dangerously-bypass-approvals-and-sandbox": rules.add(("bypass_approvals_and_sandbox", name)) @@ -2726,9 +2756,12 @@ class AgentRuleGains: jobs swap launches (#823 review). A job that remains may still run its launch in a form this reader does not read, so a launch that only stops being read has not left it, and one edited in place into the launch - that job had gains the rule (#823 review cycle 3). Each names the - launch it left, the way a step reference moved between jobs adds no - scope (#771). + that job had gains the rule (#823 review cycle 3). The launch has not + left while the job it met the rule in has an unread step of that agent + at a step the launch held, more such steps than before, or a read + launch of that agent whose input the rule is read from it did not read + (#823 review cycle 5). Each names the launch it left, the way a step + reference moved between jobs adds no scope (#771). """ claimed: list[AgentWidening] @@ -2779,15 +2812,34 @@ def compared(entries: list[dict[str, Any]]) -> set[tuple[Any, ...]]: # A launch as it is compared, less its job. return {agent_launch_key(entry)[1:] for entry in entries} + def may_still_meet(source: _RuleKey) -> bool: + # The losing job may still run the launch that met the rule, in a + # form this reader does not read: an unread step of that agent now + # stands at a step the launch held, or the job has more of them than + # before, or one of its read launches of that agent holds text in an + # input the rule is read from that this reader did not read. Such a + # launch has not left the job (#823 review cycle 5). + job, family, rule, detail = source + still_unread = unread_after.get((job, family), []) + held = {str(entry["step"]) for entry in old[source]} + if any(str(entry["step"]) in held for entry in still_unread): + return True + if len(still_unread) > len(unread_before.get((job, family), [])): + return True + return any(_unread_setting(entry, rule, detail) for entry in launched_after.get((job, family), [])) + def same_launch(source: _RuleKey, target: _RuleKey) -> bool: - # The launch that met the rule in the losing job now runs here, and - # none of this job's own launches of that agent changed in place: - # each still runs here or now runs in the losing job, as when two - # jobs swap. A job whose launch was edited into the one the losing - # job had, while the losing job still exists, gained it (#823 + # The launch that met the rule in the losing job now runs here, the + # losing job does not still run it in a form this reader does not + # read, and none of this job's own launches of that agent changed in + # place: each still runs here or now runs in the losing job, as when + # two jobs swap. A job whose launch was edited into the one the + # losing job had, while the losing job still exists, gained it (#823 # review cycle 3). if not compared(old[source]) & compared(new[target]): return False + if may_still_meet(source): + return False family = target[1] remaining = compared([ entry for job in (target[0], source[0]) for entry in launched_after.get((job, family), []) diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 9f1380877..0853843d4 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -378,6 +378,37 @@ def test_job_secrets_name_what_the_agent_job_and_the_workflow_env_reference(): assert launch["job_secrets"] == ["ANTHROPIC_API_KEY", "DEPLOY_KEY", "REVIEW_TOKEN", "WORKFLOW_ENV"] +def _job_env_workflow(env_lines: list[str]) -> str: + return "\n".join([ + "on: pull_request", + "permissions: {contents: read}", + "jobs:", + " review:", + " runs-on: ubuntu-latest", + " env:", + *(f" {line}" for line in env_lines), + " steps:", + f" - run: claude -p {BYPASS} Review", + ]) + "\n" + + +def test_job_secrets_read_a_yaml_alias_that_holds_itself_or_fans_out_once(): + """#823 review cycle 5 (P3): a self-referential alias raised RecursionError, a fan-out took minutes.""" + + holds_itself = yaml.safe_load(_job_env_workflow(["&env", "A: ${{ secrets.TOKEN }}", "B: *env"])) + launch, = _launches(holds_itself) + assert launch["job_secrets"] == ["TOKEN"] + + # Ten references at each of eight levels: 10**8 leaves if each is walked. + levels = ["l0: &l0 '${{ secrets.DEEP }}'"] + [ + f"l{level}: &l{level} [{', '.join([f'*l{level - 1}'] * 10)}]" for level in range(1, 9) + ] + started = time.monotonic() + launch, = _launches(yaml.safe_load(_job_env_workflow(levels))) + assert launch["job_secrets"] == ["DEEP"] + assert time.monotonic() - started < 5 + + # --- the only forms read: plain lists of words (#823 review cycle 4) -------------------- # # Four review cycles each found a shell form the `run:` reader mis-read, so no @@ -869,10 +900,17 @@ def test_renaming_or_moving_an_agent_step_within_its_job_is_quiet(): (_workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":workspace"}}), _workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":danger-full-access"}}), "runs without a sandbox (danger-full-access)"), + # the last of a repeated `--permission-mode` counts (#823 review cycle 5) + (_workflow(_agent("--permission-mode bypassPermissions --permission-mode default")), + _workflow(_agent("--permission-mode default --permission-mode bypassPermissions")), + "skips permission checks (bypassPermissions)"), + (_workflow({"run": "claude -p --permission-mode bypassPermissions --permission-mode default x"}), + _workflow({"run": "claude -p --permission-mode default --permission-mode=bypassPermissions x"}), + "skips permission checks (bypassPermissions)"), ], ids=["skip-flag", "mode-flag", "gate", "bots", "codex-sandbox", "codex-unsafe", "codex-args", "codex-users", "codex-cli", "gate-entry-beside-an-expression", "settings-default-mode", - "settings-top-level-default-mode", "codex-permission-profile"], + "settings-top-level-default-mode", "codex-permission-profile", "last-mode-args", "last-mode-cli"], ) def test_a_documented_rule_gained_is_a_widening(before, after, rule): changes = _changes(before, after) @@ -914,10 +952,16 @@ def test_a_documented_rule_gained_is_a_widening(before, after, rule): _workflow(_agent(settings=json.dumps({"permissions": {"defaultMode": "acceptEdits"}})))), (_workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":read-only"}}), _workflow({"uses": "openai/codex-action@v1", "with": {"permission-profile": ":workspace"}})), + # a `--permission-mode` a later one replaces meets no rule (#823 review cycle 5) + (_workflow(_agent("--permission-mode default")), + _workflow(_agent("--permission-mode bypassPermissions --permission-mode default"))), + (_workflow({"run": "claude -p --permission-mode default x"}), + _workflow({"run": "claude -p --permission-mode=bypassPermissions --permission-mode default x"})), ], ids=["respelled", "moved-to-action", "narrowed", "gate-closed", "after-an-expression", "before-an-expression", "gate-entry-holding-an-expression", "mode-expression", "tool-rule", "accept-edits", "flag-to-settings", - "settings-expression", "settings-path", "settings-accept-edits", "codex-workspace-profile"], + "settings-expression", "settings-path", "settings-accept-edits", "codex-workspace-profile", + "replaced-mode-args", "replaced-mode-cli"], ) def test_any_other_edit_is_changed(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] @@ -1053,8 +1097,13 @@ def test_renaming_a_job_that_launches_a_bypassing_agent_is_not_a_widening(): # a gate the launch opened, moved with it (_jobs(review=[{"name": "agent", **_agent(allowed_bots="*")}]), _jobs(triage=[{"name": "agent", **_agent(allowed_bots="*")}])), + # the job it left keeps an unread step it already had, at another step (#823 review cycle 5) + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}, {"name": "mcp", "run": "claude mcp add x"}], + review=[{"run": "make"}]), + _jobs(lint=[{"name": "mcp", "run": "claude mcp add x"}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), ], - ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved"], + ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved", "moved-beside-an-unread-step-that-stays"], ) def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] @@ -1085,9 +1134,32 @@ def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): # a step moved and edited while the job it left remains (_jobs(lint=[_named(BYPASS)], review=[{"run": "make"}]), _jobs(lint=[{"run": "make"}], review=[_named(f"{BYPASS} --max-turns 5")])), + # #823 review cycle 5 (M5): the job it met it in still runs it, now + # quoted and so unread, while the other job adds the same plain launch + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), + _jobs(lint=[{"name": "agent", "run": f'claude -p {BYPASS} "Review"'}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # (M6) the same, the job it met it in now running it through `npx` + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), + _jobs(lint=[{"name": "agent", "run": f"npx @anthropic-ai/claude-code -p {BYPASS} Review"}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # the same, the unread step now standing at another step label + (_jobs(lint=[{"run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), + _jobs(lint=[{"run": "echo hi"}, {"run": f'claude -p {BYPASS} "Review"'}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # the job it met it in still runs the action, with an argument input + # it no longer reads, or a settings input holding an expression + (_jobs(lint=[_named(BYPASS)], review=[{"run": "make"}]), + _jobs(lint=[_named(f'{BYPASS} --append-system-prompt "Review"')], review=[_named(BYPASS)])), + (_jobs(lint=[{"name": "agent", **_agent(settings='{"permissions":{"defaultMode":"bypassPermissions"}}')}], + review=[{"run": "make"}]), + _jobs(lint=[{"name": "agent", **_agent(settings='{"permissions":{"defaultMode":"${{ vars.MODE }}"}}')}], + review=[{"name": "agent", **_agent(settings='{"permissions":{"defaultMode":"bypassPermissions"}}')}])), ], ids=["second-job", "narrowed-there-widened-here", "unread-in-the-job-it-met-it", "edited-into-the-same-launch", - "moved-and-edited"], + "moved-and-edited", "quoted-in-the-job-it-met-it", "npx-in-the-job-it-met-it", + "unread-at-another-step-in-the-job-it-met-it", "arguments-unread-in-the-job-it-met-it", + "setting-expression-in-the-job-it-met-it"], ) def test_a_rule_another_job_gains_while_no_launch_left_is_a_widening(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] @@ -1426,6 +1498,25 @@ def settings(env): assert "three" not in row.after +@pytest.mark.parametrize("name", ["settings", "mcp_config"]) +def test_a_json_or_path_input_that_is_neither_publishes_only_a_digest(name): + """#823 review cycle 5 (P3): text before the JSON, such as a comment line, published the env value it holds.""" + + value = '// ci\n{"env": {"DB_PASSWORD_PLAIN": "canary-env-value"}}' + launch, = _launches(_workflow(_agent(**{name: value}))) + setting = next(item for item in launch["settings"] if item["name"] == name) + assert setting == {"name": name, "value": _digest(value), "unresolved_reason": None} + assert "canary-env-value" not in json.dumps(launch) + # Still compared: an edit to it is a row, and one showing neither text. + row, = _rows(_workflow(_agent(**{name: value})), _workflow(_agent(**{name: value.replace("canary", "other")}))) + assert (row.direction, row.expands) == ("changed", False) + assert "canary" not in row.before + row.after and "other-env" not in row.after + # A plain path, an expression naming one included, is published as written. + for path in (".github/claude-settings.json", "${{ github.workspace }}/ci/settings.json"): + launch, = _launches(_workflow(_agent(**{name: path}))) + assert next(item for item in launch["settings"] if item["name"] == name)["value"] == path + + def test_a_codex_config_override_publishes_its_key_and_withholds_its_value(): """#823 review cycle 4: only a rule-bearing key's value, the approval policy and the model are published.""" @@ -1814,6 +1905,36 @@ def test_the_run_forms_of_review_cycle_4_give_no_row_and_are_a_named_coverage_is assert text not in joined +@pytest.mark.parametrize( + "still_there", + [f'claude -p {BYPASS} "Review"', f"npx @anthropic-ai/claude-code -p {BYPASS} Review"], + ids=["quoted", "npx"], +) +def test_a_launch_the_job_still_runs_unread_has_not_moved_to_the_job_that_adds_it(tmp_path, still_there): + """#823 review cycle 5 (C5-F1, M5 and M6) end to end: `diff` widens, `audit --host` names the unread step.""" + + from agents_shipgate.cli.host_audit import host_audit_inventory + + permissions = {"contents": "write"} + plain = f"claude -p {BYPASS} Review" + base = _workflow(jobs={"a": {"steps": [{"run": plain}]}, "b": {"steps": [{"run": "echo hi"}]}}, + permissions=permissions) + head = _workflow(jobs={"a": {"steps": [{"run": still_there}]}, "b": {"steps": [{"run": plain}]}}, + permissions=permissions) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "an agent step in b") + + row, = _diff(repo)["rows"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (b/steps[0])" in row["why"] + assert "a step no longer declares an agent launch this audit reads (a/steps[0])" in row["why"] + assert "moved between jobs" not in row["why"] + workflow, = [grant for grant in host_audit_inventory(repo)["grants"] if grant.get("kind") == "workflow"] + assert workflow["unread_agent_runs"] == [{"job": "a", "step": "steps[0]", "agent": "claude"}] + + # --- the same row on every route --------------------------------------------------- From 70e7dd5317862098aa6341d9ac39887124c18f18 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 11:42:49 -0700 Subject: [PATCH 13/14] Address review cycle 6 on agent launches in CI (#823) C6-F1: the cycle 5 move guard only refused a move when an unread step of that agent stood at a step label the lost launch held, or the job had more unread steps than before. Merging job a's `npm i -g @anthropic-ai/claude-code` step into `claude -p --dangerously-skip-permissions Review` while job b added that plain step (V1), or removing an unread `echo "claude"` step while the launch became a quoted, unread step at another index (V2), kept the unread count and missed every held label, so the row said the bypass "moved between jobs (a/steps[1] -> b/steps[0])" and gave no widening signal. Job a may still run the launch. An unread step carries no text that tells which launch it is, so `may_still_meet` now refuses the move whenever the losing job has any unread step of that agent, wherever it stands and whether or not it was there before (option (a) of the review). V1 and V2 are now `widened`, with the "may still start an agent" caveat, and name "a step no longer declares an agent launch this audit reads". The cycle 5 guard that kept a move beside an unread step the job already had (`claude mcp add x`) now widens, in the safe direction; it moves to the widening cases, and a move beside a step that names no agent stays a move. C6-F2: `_codex_override` caught only `TOMLDecodeError` and `RecursionError`, and `tomllib` raises a plain `ValueError` for an integer past Python's 4300-digit limit, so `run: codex exec -c sandbox_mode=<5000 digits> Review` crashed `diff`, `audit --host` and `check` (exit 1) and `verify` (exit 4, internal_error). It now catches `ValueError`, which `TOMLDecodeError` subclasses; such a value is read as text and selects no sandbox, so the row is `changed`. `-c default_permissions=<4400 digits>` and `codex e --config=sandbox_mode=<4301 digits>` are covered too. Wording: the support page, STABILITY (Direction and What is not read) and the CHANGELOG say a launch has not left a job that keeps any named unread step of that agent or a launch of it with an unread input the rule is read from. The support page and STABILITY also name the one case where a launch that stops being read is worded as moved rather than as no longer declaring a launch: another job adds the same launch while the job it left keeps none of those, as when the launch became a script in the same change. Nonblocking items fixed: - docs/integrations.md: "One that gains a rule" read as the unread `run:` of the sentence before it; it now says "An agent launch that gains a rule". - The HostWorkflowAgentSettingV7 docstring, published as the 0.7 schema's description, states that a `settings` or `mcp_config` value that neither starts like a JSON object nor is a plain file path publishes only a `` digest. The 0.7 inventory and baseline schema files are regenerated. --- CHANGELOG.md | 2 +- STABILITY.md | 4 +- docs/host-boundary-support.md | 25 ++++-- docs/host-grants-baseline-schema.v0.7.json | 2 +- docs/host-grants-inventory-schema.v0.7.json | 2 +- docs/integrations.md | 2 +- src/agents_shipgate/core/host_grants.py | 30 +++---- src/agents_shipgate/schemas/host_grants.py | 6 +- tests/test_workflow_agent_launches.py | 99 +++++++++++++++++++-- 9 files changed, 137 insertions(+), 35 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ca005ae53..207e68e56 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, and one the job's unread step or unread input may already have met is named, not claimed, while an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, though never while that job keeps any unread step of that agent or a launch of it with an unread input the rule is read from; one the job's unread step or unread input may already have met is named, not claimed; and an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index ffcc152e0..a8ec3a9e8 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -366,11 +366,11 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin - **Argument inputs.** `claude_args` and `codex-args` are read only when they are a plain list of words (the characters above and parentheses, separated by blanks or newlines, with no `--settings` or `--mcp-config` flag), and published as those words one space apart, so reformatting one is quiet. Any other value — a quote, a `${{ }}` expression, `$`, a backtick, a backslash, a `#` comment, a shell separator or redirection, a glob, JSON, a `--settings` or `--mcp-config` flag, any other character — is `unresolved_reason: unread_arguments`: its `value` is ``, a short digest, so an edit is a `changed` row showing the digest, and no documented widening rule is read from it. So a quoted `claude_args` such as the #823 reproduction's `--allowedTools "Read"` is compared only by its digest. - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. `unread_agent_runs` is never compared: adding, removing or editing an unread step gives no row. - **What is withheld.** A JSON object in a `settings` or `mcp_config` input publishes, as canonical JSON, its key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so never published. A codex `--config` override in a plain list of words (`-c`, `--config=`, `-c`, `-c=`) publishes its key, and its value as `` under `env`, `headers` or a secret-named key, as written for `sandbox_mode`, `default_permissions`, `approval_policy` and `model`, and as a `` digest under any other key, such as an MCP server's `command` or `url`. The word after a secret-named word such as `--token` or `password` is ``, as the host readers redact it among an MCP server's arguments, and the value is then published redacted (`redacted`, below). Other argument text — a prompt word, a flag's value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path, as an MCP server's URL does (#723), so a change only to such a URL's path is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text in a `settings` or `mcp_config` input that starts like JSON and does not parse is withheld whole (`unparsed_json`); one that is neither a JSON object nor a plain file path (path characters, and a `${{ }}` expression only as a plain context reference) publishes only a `` digest. -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, the last of a repeated `--permission-mode` counting, as the action and the CLI keep it, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only stops being read has not left it, nor has one whose job still has an unread step of that agent where it stood, more unread steps of it than before, or a launch of it holding an expression or an unread argument input the rule is read from), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, the last of a repeated `--permission-mode` counting, as the action and the CLI keep it, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only becomes a named unread step has not left it, nor has one whose job still has any named unread step of that agent, wherever it stands and whether or not it was there before, since an unread step carries no text that tells which launch it is, or a launch of it holding an expression or an unread argument input the rule is read from), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. An unread step is not an agent step here. A removed workflow gets none. - **Unread and unreadable values are a named limit, not a blocking one.** An unread `run:` step, an argument input that is not a plain list of words (`unread_arguments`), an agent action whose `with:` is not a mapping (`form: unresolved`, `inputs_not_a_mapping`), a setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), a setting holding credential-shaped text (`redacted`), and a ref that is not a string each record a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`. An unread `run:` step is never compared, so it gives no row whatever is edited, and it never says the step starts or does not start an agent. For the others, adding, removing or re-forming the entry, or its gaining a rule, is still a row; only an edit inside it that gains no rule is not reported (an unread argument input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them: a change that only adds an unread step prints `No static host-grant changes detected`, and `audit --host` names the step. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. -- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through a variable or a function, and a step's `env:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them, or an unread `run:`, is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. +- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through a variable or a function, and a step's `env:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them, or an unread `run:`, is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. The one exception is a rule that moved between jobs (Direction, above): when another job adds the same launch while this job keeps no named unread step of that agent and no launch of it whose input the rule is read from this audit did not read, the row says the launch moved there, as a step reference moved between jobs does, so a launch that became a script or an action outside the table in the same change reads as moved; while the job keeps such a step, it never does. **Compatibility.** - **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index f3c5b6cf2..4fc99d373 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -77,7 +77,13 @@ the changed inputs the candidate rules at the end of this section name (#821): gives no row. The rest give no row and name no limit. A launch this audit read that becomes one of these is a row saying the step no longer declares an agent launch this audit reads, and that it may still start one this way; - it never says the step no longer starts an agent. + it never says the step no longer starts an agent. The one exception is a + rule that moved between jobs (below): when another job adds the same launch + while this job keeps no named unread step of that agent and no launch of it + whose input the rule is read from this audit did not read, the row says the + launch moved there, as a step reference moved between jobs does, so a + launch that became a script or an action outside the table in the same + change reads as moved; while the job keeps such a step, it never does. Review changes to those files and fields as you would a change to the workflow, hook or server entry that holds them. @@ -281,13 +287,16 @@ names the rule and step. Three gains are named in the `why` and not claimed: A second job gaining a rule a first job keeps, or a different launch gaining it while the first job still exists, is claimed: that job may still run its launch in a form this audit does not read (`npx`, quoting), so a - launch that only stops being read has not left it, and a launch edited in - place into the one that job had gains the rule. A launch has not left a job - that still has an unread step of that agent where the launch stood, more - unread steps of it than before, or a launch of it holding an expression or - an unread argument input the rule is read from — so quoting the prompt of - `claude -p --dangerously-skip-permissions Review` in one job while another - job adds that plain step is a widening. + launch that only becomes a named unread step has not left it, and a launch + edited in place into the one that job had gains the rule. A launch has not + left a job that still has any named unread step of that agent — wherever it + stands and whether or not it was there before, because an unread step + carries no text that tells which launch it is — or a launch of it holding + an expression or an unread argument input the rule is read from. So quoting + the prompt of `claude -p --dangerously-skip-permissions Review` in one job, + or merging that job's `npm i -g @anthropic-ai/claude-code` step into it, + while another job adds that plain step is a widening, and so is moving that + step to another job while the job it left keeps a `claude mcp add` step. Any other edit — `--allowedTools Read` to `--allowedTools Bash`, `acceptEdits`, a new plugin, an argument input this audit does not read, a diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 1672919e6..5024eee85 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1387,7 +1387,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A\n``settings`` or ``mcp_config`` value that neither starts like a JSON\nobject nor is a plain file path (path characters, and a ``${{ }}``\nexpression only as a plain context reference) is ````, a\ndigest and none of its text (#823 review cycle 5). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", "properties": { "holds_expression": { "default": false, diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index df23b5f8c..53cb666cc 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1445,7 +1445,7 @@ }, "HostWorkflowAgentSettingV7": { "additionalProperties": false, - "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", + "description": "One permission input or flag an agent launch declares, compared as text (#823).\n\n``name`` is the documented input (``claude_args``, ``sandbox``, \u2026) or the\nflag's primary spelling (``--allowedTools`` for ``--allowed-tools`` too).\n``value`` is the declared text, stripped, as it may be published; a flag\nthat takes no value has ``null``.\n\n``claude_args`` and ``codex-args`` are read only when they are a plain\nlist of words: letters, digits and ``_ . / : = , % + - ( )``, separated by\nblanks or newlines, with no ``--settings`` or ``--mcp-config`` flag. Every\nparser involved splits such text the same way, so it is published as\nthose words, one space apart. Any other value \u2014 holding a quote, a\n``${{ }}`` expression, ``$``, a backtick, a comment, a shell operator,\nJSON or another character \u2014 is ``unread_arguments``: ``value`` is\n````, a short digest, so an edit to it is still a change\nwhile none of its text is published; no documented widening rule is read\nfrom it; and it records a non-blocking coverage issue naming its\n``job/step`` (#823 review cycle 4). A codex ``--config`` override keeps\nits key; its value is ```` under ``env``, ``headers`` or a\nsecret-named key, as the host readers redact such values, published as\nwritten for ``sandbox_mode``, ``default_permissions``,\n``approval_policy`` and ``model``, and ```` otherwise.\n\nEvery other input is one value. A JSON object (a ``settings`` or\n``mcp_config`` value) publishes its shape and none of its free text: key\nnames, numbers, booleans and ``null``, with each string replaced by\n````, a short digest of what the host readers digest for it,\nso an edit to it is still a change. ``env`` and ``headers`` values,\n``apiKeyHelper`` and every secret-named value are ````, as the\nhost readers redact them. The strings a host reader publishes are kept:\na ``permissions.allow``/``ask``/``deny`` rule and a documented Claude\nCode setting's value such as ``defaultMode``, and an MCP server's command\nname and its URL's scheme and host, each followed by the digest when it\ndrops something the digest reads (a command's arguments, a URL's query).\nSo an MCP server's arguments and a hook's command publish nothing, as\n`.mcp.json` and `.claude/settings.json` do not (#823 review). A\n``settings`` or ``mcp_config`` value that neither starts like a JSON\nobject nor is a plain file path (path characters, and a ``${{ }}``\nexpression only as a plain context reference) is ````, a\ndigest and none of its text (#823 review cycle 5). A URL in\nother text publishes its scheme and host with ```` for its\npath and query (#723). Other text \u2014 a prompt, a flag's value \u2014 is\npublished through the workflow label redaction (#802). A value it\nrewrites is credential-shaped \u2014 a token, but also prose such as \"never\nprint bearer tokens\" \u2014 and is published redacted with\n``unresolved_reason: redacted``: it is compared as published, beside the\nrules read from its declared text, and records a non-blocking coverage\nissue naming its ``job/step``, because an edit inside what is redacted is\nnot reported. A value that is not a string (``not_a_string``), or one\nholding text that starts like JSON and does not parse (``unparsed_json``),\nis ``null`` and records a non-blocking coverage issue naming its\n``job/step``: it is neither published nor compared.\n\n``holds_expression`` is ``true`` when an input other than an argument\ninput holds a ``${{ }}`` expression, which GitHub substitutes before the\naction reads the input, and is omitted otherwise. A documented widening\nrule is then read only from the entries of a user gate that hold none,\nand from no mode or settings input, and a rule the launch gains in the\nsame job afterwards is not claimed, because the substituted text may\nalready have met it.", "properties": { "holds_expression": { "default": false, diff --git a/docs/integrations.md b/docs/integrations.md index 099b2c586..e3eebc811 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -230,7 +230,7 @@ for an edit to an argument input the audit does not read, which is compared by a digest, named in the row and named as a limit in `audit --host`. A `run:` that mentions an agent CLI and that the audit does not read is different: it gives no row in `diff` or the PR comment, whatever is edited, and is named only -as a limit in `audit --host`. One that gains a rule, such as a plain `claude_args` +as a limit in `audit --host`. An agent launch that gains a rule, such as a plain `claude_args` gaining `--dangerously-skip-permissions` on any of its lines, widens, and the hook announces it (#823), unless the launch may be a step of that job the audit did not read, rewritten, or the job's launch held before a `${{ }}` expression diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 1a87f6274..4725a6729 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -2546,7 +2546,10 @@ def _codex_override(text: str) -> tuple[str, Any] | None: raw = value.strip() try: loaded = tomllib.loads(f"value = {raw}").get("value") - except (tomllib.TOMLDecodeError, RecursionError): + except (ValueError, RecursionError): + # ``TOMLDecodeError`` is a ``ValueError``, and so is the error + # ``int()`` raises for an integer longer than Python's digit limit + # (#823 review cycle 6). loaded = raw.strip("\"'") return key.strip(), loaded @@ -2757,11 +2760,11 @@ class AgentRuleGains: launch in a form this reader does not read, so a launch that only stops being read has not left it, and one edited in place into the launch that job had gains the rule (#823 review cycle 3). The launch has not - left while the job it met the rule in has an unread step of that agent - at a step the launch held, more such steps than before, or a read - launch of that agent whose input the rule is read from it did not read - (#823 review cycle 5). Each names the launch it left, the way a step - reference moved between jobs adds no scope (#771). + left while the job it met the rule in has any unread step of that + agent, wherever it stands, or a read launch of that agent whose input + the rule is read from it did not read (#823 review cycles 5 and 6). + Each names the launch it left, the way a step reference moved between + jobs adds no scope (#771). """ claimed: list[AgentWidening] @@ -2814,17 +2817,14 @@ def compared(entries: list[dict[str, Any]]) -> set[tuple[Any, ...]]: def may_still_meet(source: _RuleKey) -> bool: # The losing job may still run the launch that met the rule, in a - # form this reader does not read: an unread step of that agent now - # stands at a step the launch held, or the job has more of them than - # before, or one of its read launches of that agent holds text in an + # form this reader does not read: it has any unread step of that + # agent, or one of its read launches of that agent holds text in an # input the rule is read from that this reader did not read. Such a - # launch has not left the job (#823 review cycle 5). + # launch has not left the job. An unread step carries no text to + # tell which launch it is, so any one counts, wherever it stands and + # whether or not it was there before (#823 review cycles 5 and 6). job, family, rule, detail = source - still_unread = unread_after.get((job, family), []) - held = {str(entry["step"]) for entry in old[source]} - if any(str(entry["step"]) in held for entry in still_unread): - return True - if len(still_unread) > len(unread_before.get((job, family), [])): + if unread_after.get((job, family)): return True return any(_unread_setting(entry, rule, detail) for entry in launched_after.get((job, family), [])) diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index c747e9755..44cad7cae 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -657,7 +657,11 @@ class HostWorkflowAgentSettingV7(BaseModel): name and its URL's scheme and host, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So an MCP server's arguments and a hook's command publish nothing, as - `.mcp.json` and `.claude/settings.json` do not (#823 review). A URL in + `.mcp.json` and `.claude/settings.json` do not (#823 review). A + ``settings`` or ``mcp_config`` value that neither starts like a JSON + object nor is a plain file path (path characters, and a ``${{ }}`` + expression only as a plain context reference) is ````, a + digest and none of its text (#823 review cycle 5). A URL in other text publishes its scheme and host with ```` for its path and query (#723). Other text — a prompt, a flag's value — is published through the workflow label redaction (#802). A value it diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 0853843d4..1aac61c1e 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -1035,6 +1035,27 @@ def test_a_codex_exec_sandbox_the_cli_does_not_select_is_changed(run): assert (row.direction, row.expands) == ("changed", False) +@pytest.mark.parametrize( + "run", + [ + "codex exec -c sandbox_mode=" + "1" * 5000 + " Review", + "codex exec -c default_permissions=" + "1" * 4400 + " Review", + "codex e --config=sandbox_mode=" + "1" * 4301 + " Review", + ], + ids=["sandbox-mode", "default-permissions", "attached-config"], +) +def test_a_codex_config_integer_past_the_digit_limit_is_read_as_text_and_selects_nothing(run): + """#823 review cycle 6 (C6-F2): `tomllib` raises a plain `ValueError` for such an integer.""" + + before, after = _workflow({"run": "echo hi"}), _workflow({"run": run}) + + launch, = _launches(after) + assert not launch.get("widening_rules") + assert host_grant_expansion_signals(_changes(before, after)) == [] + row, = _rows(before, after) + assert (row.direction, row.expands) == ("changed", False) + + def test_attached_short_values_publish_under_the_primary_spelling(): launch, = _launches(_workflow({"run": "codex exec -sdanger-full-access -c=model=o3 -pci review"})) @@ -1097,13 +1118,13 @@ def test_renaming_a_job_that_launches_a_bypassing_agent_is_not_a_widening(): # a gate the launch opened, moved with it (_jobs(review=[{"name": "agent", **_agent(allowed_bots="*")}]), _jobs(triage=[{"name": "agent", **_agent(allowed_bots="*")}])), - # the job it left keeps an unread step it already had, at another step (#823 review cycle 5) - (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}, {"name": "mcp", "run": "claude mcp add x"}], + # the job it left keeps a step that does not mention the agent + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}, {"run": "make"}], review=[{"run": "make"}]), - _jobs(lint=[{"name": "mcp", "run": "claude mcp add x"}], + _jobs(lint=[{"run": "make"}], review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), ], - ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved", "moved-beside-an-unread-step-that-stays"], + ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved", "moved-beside-a-step-that-is-no-launch"], ) def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] @@ -1147,6 +1168,24 @@ def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): (_jobs(lint=[{"run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), _jobs(lint=[{"run": "echo hi"}, {"run": f'claude -p {BYPASS} "Review"'}], review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # #823 review cycle 6 (V1): the install step merged into the launch, + # so the job it met it in has as many unread steps as before, at a + # step the launch did not hold + (_jobs(lint=[{"run": "npm i -g @anthropic-ai/claude-code"}, {"run": f"claude -p {BYPASS} Review"}], + review=[{"run": "echo hi"}]), + _jobs(lint=[{"run": f"npm i -g @anthropic-ai/claude-code && claude -p {BYPASS} Review"}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # (V2) an unread step removed while the launch becomes unread at another step + (_jobs(lint=[{"run": f"claude -p {BYPASS} Review"}, {"run": 'echo "claude"'}], review=[{"run": "echo hi"}]), + _jobs(lint=[{"run": "echo hi"}, {"run": f'claude -p {BYPASS} "Review"'}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # the job it left keeps an unread step it already had, at another + # step: that step carries no text to tell it is not the launch (#823 + # review cycle 6 flips the cycle 5 guard, in the safe direction) + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}, {"name": "mcp", "run": "claude mcp add x"}], + review=[{"run": "make"}]), + _jobs(lint=[{"name": "mcp", "run": "claude mcp add x"}], + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), # the job it met it in still runs the action, with an argument input # it no longer reads, or a settings input holding an expression (_jobs(lint=[_named(BYPASS)], review=[{"run": "make"}]), @@ -1158,7 +1197,9 @@ def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): ], ids=["second-job", "narrowed-there-widened-here", "unread-in-the-job-it-met-it", "edited-into-the-same-launch", "moved-and-edited", "quoted-in-the-job-it-met-it", "npx-in-the-job-it-met-it", - "unread-at-another-step-in-the-job-it-met-it", "arguments-unread-in-the-job-it-met-it", + "unread-at-another-step-in-the-job-it-met-it", "install-merged-into-the-launch", + "unread-step-removed-while-the-launch-becomes-unread", "beside-an-unread-step-that-stays", + "arguments-unread-in-the-job-it-met-it", "setting-expression-in-the-job-it-met-it"], ) def test_a_rule_another_job_gains_while_no_launch_left_is_a_widening(before, after): @@ -1935,6 +1976,54 @@ def test_a_launch_the_job_still_runs_unread_has_not_moved_to_the_job_that_adds_i assert workflow["unread_agent_runs"] == [{"job": "a", "step": "steps[0]", "agent": "claude"}] +def test_a_launch_merged_into_an_unread_step_it_did_not_hold_has_not_moved(tmp_path): + """#823 review cycle 6 (C6-F1, V1) end to end: the job keeps as many unread steps, at another step.""" + + from agents_shipgate.cli.host_audit import host_audit_inventory + + permissions = {"contents": "write"} + plain = f"claude -p {BYPASS} Review" + install = "npm i -g @anthropic-ai/claude-code" + base = _workflow(jobs={"a": {"steps": [{"run": install}, {"run": plain}]}, "b": {"steps": [{"run": "echo hi"}]}}, + permissions=permissions) + head = _workflow(jobs={"a": {"steps": [{"run": f"{install} && {plain}"}]}, "b": {"steps": [{"run": plain}]}}, + permissions=permissions) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "merge the install into the launch, and launch in b") + + payload = _diff(repo) + row, = payload["rows"] + assert (row["direction"], row["expands"]) == ("widened", True) + assert "an agent launch now skips permission checks (bypassPermissions) (b/steps[0])" in row["why"] + assert "a step no longer declares an agent launch this audit reads (a/steps[1])" in row["why"] + assert "may still start an agent in a way this audit does not read" in row["why"] + assert "moved between jobs" not in row["why"] + workflow, = [grant for grant in host_audit_inventory(repo)["grants"] if grant.get("kind") == "workflow"] + assert workflow["unread_agent_runs"] == [{"job": "a", "step": "steps[0]", "agent": "claude"}] + + +def test_a_codex_config_integer_past_the_digit_limit_crashes_no_route(tmp_path): + """#823 review cycle 6 (C6-F2): `diff`, `audit --host` and `check` exit 0, and `verify` is no internal error.""" + + repo = _repo(tmp_path, {SOURCE: _yaml(_workflow({"run": "echo hi"}))}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(_workflow({"run": "codex exec -c sandbox_mode=" + "1" * 5000 + " Review"}))}) + _git(repo, "commit", "-qam", "a codex step") + + for args in ( + ["diff", "--workspace", str(repo), "--base", "main", "--json"], + ["audit", "--host", "--workspace", str(repo), "--json"], + ["check", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "agent-boundary-json"], + ["verify", "--workspace", str(repo), "--base", "main", "--head", "HEAD", "--format", "text"], + ): + result = CliRunner().invoke(app, args) + assert result.exit_code == 0, (args[0], result.output, result.exception) + row, = _diff(repo)["rows"] + assert (row["direction"], row["expands"]) == ("changed", False) + + # --- the same row on every route --------------------------------------------------- From 3986c93da637a284295da23b980c8c5c581b02f6 Mon Sep 17 00:00:00 2001 From: pengfei-threemoonslab Date: Wed, 23 Sep 2026 12:43:33 -0700 Subject: [PATCH 14/14] Address review cycle 7 on agent launches in CI (#823) C7-F1: `may_still_meet` looked for unread steps only under the losing job's name, which holds nothing once that job is renamed or removed, and `job_left` accepted any job that no longer exists. So renaming job a to a2 while quoting its `claude -p --dangerously-skip-permissions Review` (R6), or running it through `npx` (R7), or removing a while an existing job c gained the quoted launch (R3), as job b added the plain launch, was one `low changed` row saying the bypass "moved between jobs (a/steps[0] -> b/steps[0])", with no widening signal. The launch may still run, unread, under a2 or in c. `may_still_meet` now also refuses the move while any job other than the losing and the receiving one has more unread steps of that agent, or read launches of it holding unread text in an input the rule is read from, than it had at the base; a job new at the head counts from zero. `job_left` applies the same check, so both move paths refuse alike. R6, R7, R3, the same while the losing job remains, and a rename into an unread argument input are widenings; a plain rename, a rename with the install step the job keeps beside the launch, and a rename beside another job that keeps its unread step stay moves. Two jobs renamed at once, one of them holding an unread step, cannot be told apart from R6 and now claim the gain, in the safe direction; the support page, STABILITY, the CHANGELOG and the `AgentRuleGains` docstring say so. C7-F2: `_EXPRESSION_RE` (`\$\{\{(.*?)\}\}`) scanned to the end of the text from every unterminated `${{`, so a job `env:` or an agent action's `settings` input holding a few hundred KB of them stalled `diff`, `audit --host` and `check` (92 s at 300 KB, 1070 s at 1 MB). The pattern now ends at `}}` or the end of the text, as `_EXPRESSION_SPAN_RE` does, and only a closed expression names a secret, so what `job_secrets` publishes is unchanged. 300 KB and 1 MB now take 0.8 s and 1.0 s end to end; a new case bounds about 360 KB of them, and the settings input, to 5 s. Nonblocking, fixed: - A declared `shell:` was read whenever its first word was `bash` or `sh`, so `bash -c 'claude -p --dangerously-skip-permissions Review' {0}` beside `run: claude -p Review` read the step and missed the template. A template is now read only as `bash`/`sh` running the script alone: `set` flags, `-l`, `-i`, `-r`, `-o`/`-O` with a name other than `noexec`, and `--noprofile`, `--norc`, `--posix`, `--login`, `--restricted`, `--noediting`, `--verbose`, then `{0}` last. Anything else (`-c`, `-s`, `-n`, `--rcfile`, words after `{0}`) leaves the step an unread, named limit. The schema description of `unread_agent_runs` says so, and the 0.7 inventory and baseline schema files are regenerated. - The coverage line for a workflow that changed with no row said "redacted values such as env values and apiKeyHelper are not compared", which named nothing a workflow holds. A workflow's line now says "text this entry does not read, such as a step's env or an unread agent step, is not compared; audit --host names each unread agent step". Other files keep their note. The capability_diff row in docs/distribution-surfaces.md and its parity comment are updated. --- CHANGELOG.md | 2 +- STABILITY.md | 8 +- docs/distribution-surfaces.md | 2 +- docs/host-boundary-support.md | 38 +++++- docs/host-grants-baseline-schema.v0.7.json | 2 +- docs/host-grants-inventory-schema.v0.7.json | 2 +- src/agents_shipgate/core/host_grants.py | 124 ++++++++++++++---- src/agents_shipgate/report/host_comparison.py | 23 +++- src/agents_shipgate/schemas/host_grants.py | 3 +- tests/test_distribution_surface_parity.py | 6 +- tests/test_workflow_agent_launches.py | 112 +++++++++++++++- 11 files changed, 270 insertions(+), 52 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 207e68e56..ae33df1a0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; this change moves no host-grants schema (the unreleased host-grants `0.7` is #823's), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh`, and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, though never while that job keeps any unread step of that agent or a launch of it with an unread input the rule is read from; one the job's unread step or unread input may already have met is named, not claimed; and an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh` (a `shell:` template only when it runs the script alone, never `bash -c '…' {0}`), and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row, and the workflow's line in `What this run established` names what its grant does not read instead of env values and `apiKeyHelper`. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, though never while that job keeps any unread step of that agent or a launch of it with an unread input the rule is read from, nor while any other job but the receiving one holds more of them than before, so renaming a job while quoting its launch, as another job adds that launch plainly, is a widening; one the job's unread step or unread input may already have met is named, not claimed; and an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. diff --git a/STABILITY.md b/STABILITY.md index a8ec3a9e8..1302ec28a 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -362,15 +362,15 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin ``` - **Shell is not parsed.** A value is read only as a plain list of words, which every parser involved splits at its blanks and nowhere else; every other form is named as a non-blocking limit, publishes none of its text and is never guessed at. -- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets. A `run:` is an agent launch only when it is one line of plain words (letters, digits and `_ . / : = , % + -`, separated by spaces or tabs), run by `bash`, `sh` or no declared `shell:`, whose program, after any `NAME=value` assignments (skipped, never published), has the file name `claude` and passes `-p`/`--print`, or is `codex` followed by `exec` (`codex e`); it lists its documented permission flags under their primary spelling, and a `--settings` or `--mcp-config` flag leaves it unread. Any other `run:` that mentions `claude` or `codex` as a word of its own is one `unread_agent_runs` entry per agent CLI it mentions, holding only `job`, `step` and `agent`. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). +- **What is read.** A step whose `uses:` is `anthropics/claude-code-action`, `anthropics/claude-code-base-action` (also published as `anthropics/claude-code-action/base-action`) or `openai/codex-action` (at any ref, in any letter case) is an agent launch listing the documented inputs it sets. A `run:` is an agent launch only when it is one line of plain words (letters, digits and `_ . / : = , % + -`, separated by spaces or tabs), run by `bash`, `sh` or no declared `shell:` (a template only as `bash` or `sh` running the script alone, option words it still runs the script under then `{0}` last, so never one such as `bash -c '…' {0}` that may run a command of its own, or one with `-s`, `-n` or `--rcfile`), whose program, after any `NAME=value` assignments (skipped, never published), has the file name `claude` and passes `-p`/`--print`, or is `codex` followed by `exec` (`codex e`); it lists its documented permission flags under their primary spelling, and a `--settings` or `--mcp-config` flag leaves it unread. Any other `run:` that mentions `claude` or `codex` as a word of its own is one `unread_agent_runs` entry per agent CLI it mentions, holding only `job`, `step` and `agent`. Each `actions/checkout` step lists its `with.ref`, `null` for the default. The support page has [the input and flag tables](docs/host-boundary-support.md#known-unread-surfaces). Values are text: no action is fetched, no command run, no expression evaluated. `job_secrets` names the secrets the launch's job references and the workflow's `env` passes; it is context for the row and is not compared. `job`, `step`, values and secret names are published labels (#802). - **Argument inputs.** `claude_args` and `codex-args` are read only when they are a plain list of words (the characters above and parentheses, separated by blanks or newlines, with no `--settings` or `--mcp-config` flag), and published as those words one space apart, so reformatting one is quiet. Any other value — a quote, a `${{ }}` expression, `$`, a backtick, a backslash, a `#` comment, a shell separator or redirection, a glob, JSON, a `--settings` or `--mcp-config` flag, any other character — is `unresolved_reason: unread_arguments`: its `value` is ``, a short digest, so an edit is a `changed` row showing the digest, and no documented widening rule is read from it. So a quoted `claude_args` such as the #823 reproduction's `--allowedTools "Read"` is compared only by its digest. - **What is compared.** Each job's multiset of launches (`job`, `agent`, `form`, `unresolved_reason`, `settings` by `name`, `value` and `unresolved_reason`, and `widening_rules`) and of checkout refs (`job`, `ref`, `unresolved_reason`), never the step label, so a rename or reorder is quiet. An added, removed or changed entry is one `changed` row on the workflow naming `job/step` and the value on each side. Of a CLI launch only the documented permission flags and their values are compared, so `--model` and any undocumented flag are not; a flag `claude` reads as variadic (`--allowedTools`, `--disallowedTools`, `--add-dir`) takes every following word up to the next word starting with `-`, as the CLI reads it, so a prompt written after one is compared, and published, as one of its values; any other prompt is not. `unread_agent_runs` is never compared: adding, removing or editing an unread step gives no row. - **What is withheld.** A JSON object in a `settings` or `mcp_config` input publishes, as canonical JSON, its key names, numbers, booleans and `null`, with each string replaced by ``: a short digest of what the host readers digest for that string, so editing it is still a `changed` row while none of its text is published. `env` and `headers` values, `apiKeyHelper` and every secret-named value are `` and not digested, as the host readers redact them, so rotating one is quiet. The strings a host reader publishes are kept: a `permissions.allow`, `ask` or `deny` rule, and the value of a documented Claude Code setting (`defaultMode`, the switches, `enabledMcpjsonServers` entries), as the settings reader publishes them; and an MCP server's command name and its URL's scheme and host, as the MCP reader publishes them, each followed by the digest when it drops something the digest reads (a command's arguments, a URL's query). So `{"mcpServers":{"remote":{"command":"npx","args":["mcp-remote","https://…","--header","Authorization: Bearer …"]}}}` publishes `{"mcpServers":{"remote":{"args":["","","",""],"command":"npx"}}}`, and a hook publishes its event names and no command, as `.mcp.json` and `.claude/settings.json` publish none of them. JSON passed through `claude_args`, `codex-args` or a `run:` is never read, so never published. A codex `--config` override in a plain list of words (`-c`, `--config=`, `-c`, `-c=`) publishes its key, and its value as `` under `env`, `headers` or a secret-named key, as written for `sandbox_mode`, `default_permissions`, `approval_policy` and `model`, and as a `` digest under any other key, such as an MCP server's `command` or `url`. The word after a secret-named word such as `--token` or `password` is ``, as the host readers redact it among an MCP server's arguments, and the value is then published redacted (`redacted`, below). Other argument text — a prompt word, a flag's value — is published as written through the #802 label redaction, except that a URL in it publishes its scheme, host and port, with `` for any path, as an MCP server's URL does (#723), so a change only to such a URL's path is not reported; a `${{ }}` expression is one word while this is decided, so one in a URL's userinfo is withheld with it. Text in a `settings` or `mcp_config` input that starts like JSON and does not parse is withheld whole (`unparsed_json`); one that is neither a JSON object nor a plain file path (path characters, and a `${{ }}` expression only as a plain context reference) publishes only a `` digest. -- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, the last of a repeated `--permission-mode` counting, as the action and the CLI keep it, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only becomes a named unread step has not left it, nor has one whose job still has any named unread step of that agent, wherever it stands and whether or not it was there before, since an unread step carries no text that tells which launch it is, or a launch of it holding an expression or an unread argument input the rule is read from), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. +- **Direction.** Only a documented rule a job's launches gain widens: bypassed permission checks (`--dangerously-skip-permissions` or `--permission-mode bypassPermissions` in a plain `claude_args` or `run:`, the last of a repeated `--permission-mode` counting, as the action and the CLI keep it, or Claude Code settings written as JSON in the `settings` input whose `defaultMode` is `bypassPermissions`, read as the settings reader reads it; one rule), `--dangerously-bypass-approvals-and-sandbox`, a `danger-full-access` sandbox (`sandbox: danger-full-access`, `--sandbox danger-full-access` in any spelling clap reads, `-sdanger-full-access` and `-s=danger-full-access` included, or `permission-profile: :danger-full-access`, Codex's built-in full-access profile; and in a `codex exec` step that passes no `--sandbox`, which takes precedence, a `--config` override — `-c`, `--config=`, `-c` or `-c=` — of `sandbox_mode` to `danger-full-access` or of `default_permissions` to `:danger-full-access`, the last override of a key counting and `default_permissions` outranking `sandbox_mode`, as the CLI resolves them; never such an override in `codex-args`, after which the action appends its own `--sandbox` or `default_permissions` selection), `safety-strategy: unsafe`, or a `*` entry in `allowed_bots` (any bot), `allowed_non_write_users` or `allow-users` (any user). A `settings` path names a file this audit does not read, and meets none. The rules each launch meets are decided from its declared text when the workflow is read, before anything is withheld, and published as `widening_rules` (`rule`, and the `setting` it was read from), so redaction never hides one. A rule is read only from text this audit reads exactly: an argument input holding a `${{ }}` expression is not read at all, a user gate's entries that hold none are read, and a `sandbox`, `permission-profile`, `safety-strategy` or `settings` value holding one meets none; such a setting is published with `holds_expression: true` (omitted otherwise), and a row that changes it says the text the expression reaches is not read. Three gains are named in the `why` and not claimed: where, in the job, a step that may launch that agent in a form this audit does not read (an unread `run:`, or an action whose `with:` is not a mapping) is gone and a launch this audit reads is added, as for a job whose permissions were not explicit, because the added launch may be that step rewritten; where the job's launch held, before, a `${{ }}` expression or an argument input that was not a plain list of words in an input the rule is read from, which may already have met it; and where the rule moved between jobs, because the launch that met it in another job left that job (the job no longer exists, or the same launch now runs elsewhere while the receiving job's own launches of that agent still run there or in the job it left; a job that remains may still run its launch in a form this audit does not read, so a launch that only becomes a named unread step has not left it, nor has one whose job still has any named unread step of that agent, wherever it stands and whether or not it was there before, since an unread step carries no text that tells which launch it is, or a launch of it holding an expression or an unread argument input the rule is read from; nor while any job but the receiving one has more such steps and launches of that agent than it had before, a job new at the head holding any, because the launch may be one of them and the job it left may be that job under a new name, so renaming a job while quoting its launch or running it through `npx`, as another job adds that plain launch, is claimed, and so are two jobs renamed at once while one holds an unread step, in the safe direction), as a step reference moved between jobs adds no scope. An unread step that remains takes no gain from another launch. Otherwise the grant earns `workflow_agent_widened_changed` (or `_added` for a new workflow), the row is `widened` with `expands: true`, its `why` names the rule and step, and the Stop hook announces it. Every other edit is `changed` with `expands: false`, including a tool rule such as `--allowedTools Bash` (rating its reach is #824's), `acceptEdits`, a new plugin, an unread argument input and a head-ref checkout. `access` and `risk` still describe the token and triggers alone. - **The note on a workflow row.** Whatever the row is about, when its workflow runs an agent its `why` ends with the job facts beside each agent step: an untrusted-input trigger (`issue_comment`, `issues`, `pull_request_target`, `workflow_run`), the job's write scopes, the job's secrets, and a checkout of pull request code in the job. It moves no direction and is not a verdict. An unread step is not an agent step here. A removed workflow gets none. -- **Unread and unreadable values are a named limit, not a blocking one.** An unread `run:` step, an argument input that is not a plain list of words (`unread_arguments`), an agent action whose `with:` is not a mapping (`form: unresolved`, `inputs_not_a_mapping`), a setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), a setting holding credential-shaped text (`redacted`), and a ref that is not a string each record a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`. An unread `run:` step is never compared, so it gives no row whatever is edited, and it never says the step starts or does not start an agent. For the others, adding, removing or re-forming the entry, or its gaining a rule, is still a row; only an edit inside it that gains no rule is not reported (an unread argument input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them: a change that only adds an unread step prints `No static host-grant changes detected`, and `audit --host` names the step. +- **Unread and unreadable values are a named limit, not a blocking one.** An unread `run:` step, an argument input that is not a plain list of words (`unread_arguments`), an agent action whose `with:` is not a mapping (`form: unresolved`, `inputs_not_a_mapping`), a setting that is not a string (`not_a_string`) or holds text that starts like JSON and does not parse (`unparsed_json`), a setting holding credential-shaped text (`redacted`), and a ref that is not a string each record a non-blocking `unsupported` coverage issue naming its `job/step`, as an unread secret value does (#693): GitHub coverage stays `complete`. An unread `run:` step is never compared, so it gives no row whatever is edited, and it never says the step starts or does not start an agent. For the others, adding, removing or re-forming the entry, or its gaining a rule, is still a row; only an edit inside it that gains no rule is not reported (an unread argument input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them: a change that only adds an unread step prints `No static host-grant changes detected`, and `audit --host` names the step. The workflow's coverage line then ends `so no row (text this entry does not read, such as a step's env or an unread agent step, is not compared; audit --host names each unread agent step)` instead of the redacted-values note another file's line carries. - **Credential-shaped text.** Other text the #802 label redaction rewrites — a token shape, a credential assignment, a bearer or header value, a URL's userinfo, and prose such as "never print bearer tokens" in a system prompt — is published redacted with `unresolved_reason: redacted`. In a setting it is compared as published, beside the rules read from its declared text, and named by the non-blocking limit above, so a permission change or a rule gained beside it is still a row and only an edit inside what is redacted is not reported. A checkout ref names the code a job runs, so a redacted one refuses as a redacted step reference does (#767): a changed workflow's comparison is refused, and an unchanged one is named in `unchanged_limits`. -- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through a variable or a function, and a step's `env:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them, or an unread `run:`, is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. The one exception is a rule that moved between jobs (Direction, above): when another job adds the same launch while this job keeps no named unread step of that agent and no launch of it whose input the rule is read from this audit did not read, the row says the launch moved there, as a step reference moved between jobs does, so a launch that became a script or an action outside the table in the same change reads as moved; while the job keeps such a step, it never does. +- **What is not read.** An action outside the table, a composite action (#701), a script the step runs, an agent CLI reached through a variable or a function, and a step's `env:` and `if:`. The support page lists them under Known unread surfaces. A launch this audit read that becomes one of them, or an unread `run:`, is a row saying the step no longer declares an agent launch this audit reads, never that it no longer starts an agent. The one exception is a rule that moved between jobs (Direction, above): when another job adds the same launch while this job keeps no named unread step of that agent and no launch of it whose input the rule is read from this audit did not read, and no job but the receiving one holds more of them than it did, the row says the launch moved there, as a step reference moved between jobs does, so a launch that became a script or an action outside the table in the same change reads as moved; while the job keeps such a step, or any job but the receiving one gains one, it never does. **Compatibility.** - **A committed `0.4`, `0.5` or `0.6` baseline holding a workflow grant** is loaded but incomparable: it never read agent launches or checkout refs, so its silence is not evidence that none changed. `audit --host --drift` reports `comparison_status: incomparable` with `baseline_workflow_agent_launches_unavailable` among `incomparable_reasons` (beside the #771 and #693 reasons for a `0.4`/`0.5` one), `has_drift: null` and `next_action: null`, and exits `20` under `--fail-on-drift`; `preflight` raises a `high`, `actor: human` `host_grant_drift` signal naming it. To migrate, follow [the #771 steps](#workflow-step-action-references-contract-v40-771) from a checkout of the reviewed default branch, keeping the old file as `host-grants.v0.6.json`: review `audit --host`, move the baseline aside, `audit --host --save-baseline`, and confirm drift is comparable with `has_drift: false`. diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index 8e43ed4d4..d3bd792b6 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,7 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, an unread argument input or an unresolved launch is named there and in the `why` of a row reporting its launch, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, an unread argument input or an unresolved launch is named there and in the `why` of a row reporting its launch, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL and the env and header key names its grant already publishes, redacted and bounded, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or arguments; an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note, and a workflow's names instead what its grant does not read and that `audit --host` names each unread agent step (#823); sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/docs/host-boundary-support.md b/docs/host-boundary-support.md index 4fc99d373..b7ec36daa 100644 --- a/docs/host-boundary-support.md +++ b/docs/host-boundary-support.md @@ -80,10 +80,12 @@ the changed inputs the candidate rules at the end of this section name (#821): it never says the step no longer starts an agent. The one exception is a rule that moved between jobs (below): when another job adds the same launch while this job keeps no named unread step of that agent and no launch of it - whose input the rule is read from this audit did not read, the row says the - launch moved there, as a step reference moved between jobs does, so a - launch that became a script or an action outside the table in the same - change reads as moved; while the job keeps such a step, it never does. + whose input the rule is read from this audit did not read, and no job but + the one adding it holds more of them than it did, the row says the launch + moved there, as a step reference moved between jobs does, so a launch that + became a script or an action outside the table in the same change reads as + moved; while the job keeps such a step, or any job but the receiving one + gains one, it never does. Review changes to those files and fields as you would a change to the workflow, hook or server entry that holds them. @@ -185,7 +187,14 @@ never guessed at. Four things are listed on the workflow grant, each naming its `&`, `|`, `<`, `>`, parenthesis, brace, glob, `~`, `!`, `@` or second line, and every POSIX shell runs it as exactly those words. It must run under `bash`, `sh` or no declared `shell:` (the step's, else its job's or its - workflow's `defaults.run.shell`). After any `NAME=value` assignments, which + workflow's `defaults.run.shell`); a template is read only as `bash` or `sh` + running the script alone — option words it still runs the script under + (`set` flags, `-l`, `-i`, `-r`, `-o`/`-O` with an option name other than + `noexec`, `--noprofile`, `--norc`, `--posix`, `--login`, `--restricted`, + `--noediting`, `--verbose`), then `{0}` last, as in + `bash --noprofile --norc -eo pipefail {0}` — so one such as + `bash -c '…' {0}`, which may run a command of its own, or one with `-s`, + `-n` or `--rcfile`, is not. After any `NAME=value` assignments, which are skipped and never published, the program's file name must be `claude` with `-p`/`--print` among its arguments, or `codex` followed by `exec` (`codex e`): `claude -p …`, `./node_modules/.bin/claude -p …` and @@ -225,7 +234,8 @@ never guessed at. Four things are listed on the workflow grant, each naming its a line continuation, another program such as `npx`, `timeout`, `sudo` or `echo`, a subcommand that is not a headless launch (`claude mcp add`, `codex login`), `codex` with an option before `exec`, a script named after - an agent, or a declared `shell:` other than `bash` or `sh`. It is listed in + an agent, or a declared `shell:` other than `bash` or `sh` running the + script alone. It is listed in `unread_agent_runs` once for each agent CLI it mentions, with no other field, and is a non-blocking limit (below). Nothing more: none of its text is published, it is never compared, so adding, removing or editing it gives @@ -297,6 +307,16 @@ names the rule and step. Three gains are named in the `why` and not claimed: or merging that job's `npm i -g @anthropic-ai/claude-code` step into it, while another job adds that plain step is a widening, and so is moving that step to another job while the job it left keeps a `claude mcp add` step. + Nor has it left while any job but the one gaining the rule has more such + steps and launches of that agent than it had before — a job new at the + head holding any — because the launch may be one of them, and the job it + left may be that job under a new name. So renaming the job while quoting + that launch or running it through `npx`, or removing the job while another + job gains it quoted, as a third job adds the plain step, is a widening; + renaming a job with the install step it keeps beside the launch, or beside + another job that keeps its unread step, is a move. Two jobs renamed at + once, one of them holding an unread step, cannot be told apart from those, + so they claim the gain, in the safe direction. Any other edit — `--allowedTools Read` to `--allowedTools Bash`, `acceptEdits`, a new plugin, an argument input this audit does not read, a @@ -398,7 +418,11 @@ input's edit is a `changed` row by its digest). `diff`, `verify` and `check` carry no limit for any of them, as for an unread secret value (#693): a `diff` of a change that only adds an unread agent step says `No static host-grant changes detected`, and `audit --host` is where the step -is named. +is named. The workflow's coverage line says so: `… compared; changed, but no +grant this entry compares changed, so no row (text this entry does not read, +such as a step's env or an unread agent step, is not compared; audit --host +names each unread agent step)`, where another file's line names redacted +values such as env values and `apiKeyHelper`. A workflow's labels are published redacted (#802). A job id, a step's `id` or `name`, a trigger and a permission scope name go through the same redaction as diff --git a/docs/host-grants-baseline-schema.v0.7.json b/docs/host-grants-baseline-schema.v0.7.json index 5024eee85..82115c889 100644 --- a/docs/host-grants-baseline-schema.v0.7.json +++ b/docs/host-grants-baseline-schema.v0.7.json @@ -1739,7 +1739,7 @@ }, "HostWorkflowUnreadAgentRunV7": { "additionalProperties": false, - "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", + "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` run on the script alone (so ``bash -c '\u2026' {0}``\ntoo) \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ diff --git a/docs/host-grants-inventory-schema.v0.7.json b/docs/host-grants-inventory-schema.v0.7.json index 53cb666cc..aecb069b6 100644 --- a/docs/host-grants-inventory-schema.v0.7.json +++ b/docs/host-grants-inventory-schema.v0.7.json @@ -1797,7 +1797,7 @@ }, "HostWorkflowUnreadAgentRunV7": { "additionalProperties": false, - "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", + "description": "A ``run:`` step that mentions a known agent CLI and is not read as an agent launch (#823 review cycle 4).\n\nAny ``run:`` holding ``claude`` or ``codex`` as a word of its own that is\nnot an agent launch this reader reads \u2014 more than one line or command, a\nquote, an expansion, a redirection, a comment, a continuation, a\n``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a\nsubcommand that is not a headless launch, or a declared ``shell:`` other\nthan ``bash`` or ``sh`` run on the script alone (so ``bash -c '\u2026' {0}``\ntoo) \u2014 once for each agent CLI it mentions. It is a\nnamed, non-blocking limit and nothing more: none of the step's text is\npublished, it is never compared, so adding, removing or editing it gives\nno row, and it never says that the step starts, or does not start, an\nagent. ``job`` and ``step`` are published labels (#802).", "properties": { "agent": { "enum": [ diff --git a/src/agents_shipgate/core/host_grants.py b/src/agents_shipgate/core/host_grants.py index 4725a6729..f507f380a 100644 --- a/src/agents_shipgate/core/host_grants.py +++ b/src/agents_shipgate/core/host_grants.py @@ -1773,7 +1773,11 @@ def agent_rule_text(rule: str, detail: str) -> str: _ASSIGNMENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*=") _SECRET_REFERENCE_RE = re.compile(r"\bsecrets\.([A-Za-z_][A-Za-z0-9_]*)") -_EXPRESSION_RE = re.compile(r"\$\{\{(.*?)\}\}", re.S) +#: One ``${{ … }}`` expression's body and its closing ``}}``, or an unterminated +#: ``${{`` to the end of the text, which names no secret. A lazy match that +#: must find ``}}`` scanned to the end from every ``${{``, so text holding many +#: unterminated ones took quadratic time (#823 review cycle 7). +_EXPRESSION_RE = re.compile(r"\$\{\{(.*?)(\}\}|\Z)", re.S) def pull_request_code_ref(ref: str | None) -> bool: @@ -1887,14 +1891,54 @@ def _declared_shell(step: dict[Any, Any], job: dict[Any, Any], workflow: dict[An return None +#: An option word of a declared ``bash``/``sh`` template after which the shell +#: still runs the script: a cluster of ``set`` flags and ``-l``, ``-i`` or +#: ``-r``, or one of these long options. Never ``-c``, which runs a command +#: string of its own, ``-s``, which reads commands from standard input, ``-n``, +#: which runs none, ``--rcfile`` or anything else. A cluster ending in ``o`` +#: or ``O`` takes an option name next. +_SHELL_OPTION_RE = re.compile( + r"[-+](?:[abefhiklmprtuvxBCEHPT]+[oO]?|[oO])" + r"|--(?:noprofile|norc|posix|login|restricted|noediting|verbose)" +) +_SHELL_OPTION_NAME_RE = re.compile(r"[a-z_]+") +#: The option name that runs no command. +_SHELL_NO_EXEC = "noexec" + + def _read_shell(shell: Any) -> bool: - """Whether a ``run:`` under this declared shell is read: none declared, ``bash`` or ``sh``.""" + """Whether a ``run:`` under this declared shell is read. + + None declared, ``bash`` or ``sh`` (by any path), or a template that runs + one of them on the script alone: option words it still runs the script + under, then ``{0}`` last, as in ``bash --noprofile --norc -eo pipefail + {0}``. Any other template, such as ``bash -c '…' {0}``, may run a command + of its own or none, so the step is not read (#823 review cycle 7). + """ if shell is None: return True - if not isinstance(shell, str) or not shell.split(): + if not isinstance(shell, str): + return False + words = shell.split() + if not words or words[0].rsplit("/", 1)[-1] not in _READ_SHELLS: + return False + if len(words) == 1: + return True + if words[-1] != "{0}": return False - return shell.split()[0].rsplit("/", 1)[-1] in _READ_SHELLS + index = 1 + while index < len(words) - 1: + word = words[index] + if not _SHELL_OPTION_RE.fullmatch(word): + return False + index += 1 + if not word.startswith("--") and word[-1] in "oO": + name = words[index] if index < len(words) - 1 else "" + if not _SHELL_OPTION_NAME_RE.fullmatch(name) or name == _SHELL_NO_EXEC: + return False + index += 1 + return True def _read_flags( @@ -2333,8 +2377,9 @@ def _job_secrets(job: dict[Any, Any], workflow_env: Any) -> list[str]: while pending: value = pending.pop() if isinstance(value, str): - for expression in _EXPRESSION_RE.findall(value): - names.update(_SECRET_REFERENCE_RE.findall(expression)) + for expression, closed in _EXPRESSION_RE.findall(value): + if closed: + names.update(_SECRET_REFERENCE_RE.findall(expression)) elif isinstance(value, (dict, list)) and id(value) not in seen: seen.add(id(value)) pending.extend(value.values() if isinstance(value, dict) else value) @@ -2762,9 +2807,14 @@ class AgentRuleGains: that job had gains the rule (#823 review cycle 3). The launch has not left while the job it met the rule in has any unread step of that agent, wherever it stands, or a read launch of that agent whose input - the rule is read from it did not read (#823 review cycles 5 and 6). - Each names the launch it left, the way a step reference moved between - jobs adds no scope (#771). + the rule is read from it did not read (#823 review cycles 5 and 6); + nor while any other job but this one has more such steps and launches + than it had before, as a job new at the head holding one has, because + the launch may be one of them: the job it met the rule in may be that + job under a new name (#823 review cycle 7). So two jobs renamed at + once, one of them holding an unread step, claim the gain, in the safe + direction. Each names the launch it left, the way a step reference + moved between jobs adds no scope (#771). """ claimed: list[AgentWidening] @@ -2815,18 +2865,38 @@ def compared(entries: list[dict[str, Any]]) -> set[tuple[Any, ...]]: # A launch as it is compared, less its job. return {agent_launch_key(entry)[1:] for entry in entries} - def may_still_meet(source: _RuleKey) -> bool: - # The losing job may still run the launch that met the rule, in a - # form this reader does not read: it has any unread step of that - # agent, or one of its read launches of that agent holds text in an - # input the rule is read from that this reader did not read. Such a - # launch has not left the job. An unread step carries no text to - # tell which launch it is, so any one counts, wherever it stands and - # whether or not it was there before (#823 review cycles 5 and 6). + def unread_forms(grant_unread: dict[tuple[str, str], list[dict[str, Any]]], + grant_launched: dict[tuple[str, str], list[dict[str, Any]]], + job: str, family: str, rule: str, detail: str) -> int: + # How many steps of the job may run a launch of that agent that this + # reader cannot tell meets the rule: its unread steps of that agent, + # and its read launches of it holding text in an input the rule is + # read from that this reader did not read. + return len(grant_unread.get((job, family), [])) + sum( + 1 for entry in grant_launched.get((job, family), []) + if entry.get("form") == "read" and _unread_setting(entry, rule, detail) + ) + + def may_still_meet(source: _RuleKey, target: _RuleKey) -> bool: + # The launch that met the rule may still run, in a form this reader + # does not read, somewhere other than the job gaining the rule. Such a + # launch has not left for that job. An unread step carries no text to + # tell which launch it is, so: + # - in the losing job, any one counts, wherever it stands and whether + # or not it was there before (#823 review cycles 5 and 6); + # - in any other job, one counts when that job has more of them than + # it had before, as a job new at the head holding one has — so a + # renamed losing job cannot hide the launch it kept, quoted, under + # its new name (#823 review cycle 7). job, family, rule, detail = source - if unread_after.get((job, family)): + if unread_forms(unread_after, launched_after, job, family, rule, detail): return True - return any(_unread_setting(entry, rule, detail) for entry in launched_after.get((job, family), [])) + others = {name for name, agent in (*unread_after, *launched_after) if agent == family} - {job, target[0]} + return any( + unread_forms(unread_after, launched_after, other, family, rule, detail) + > unread_forms(unread_before, launched_before, other, family, rule, detail) + for other in others + ) def same_launch(source: _RuleKey, target: _RuleKey) -> bool: # The launch that met the rule in the losing job now runs here, the @@ -2838,7 +2908,7 @@ def same_launch(source: _RuleKey, target: _RuleKey) -> bool: # review cycle 3). if not compared(old[source]) & compared(new[target]): return False - if may_still_meet(source): + if may_still_meet(source, target): return False family = target[1] remaining = compared([ @@ -2848,12 +2918,14 @@ def same_launch(source: _RuleKey, target: _RuleKey) -> bool: jobs_after = {str(context["job"]) for context in (after or {}).get("permission_contexts", [])} - def job_left(source: _RuleKey, _target: _RuleKey) -> bool: - # The losing job no longer exists, as when it is renamed. A job that - # remains may still run its launch in a form this reader does not - # read, such as `npx`, so its launch is not taken to have left - # (#823 review cycle 3). - return source[0] not in jobs_after + def job_left(source: _RuleKey, target: _RuleKey) -> bool: + # The losing job no longer exists, as when it is renamed, and no + # other job may run its launch unread. A job that remains may still + # run its launch in a form this reader does not read, such as `npx`, + # so its launch is not taken to have left (#823 review cycle 3); a + # job renamed while it runs its launch unread is a new job holding + # an unread step (#823 review cycle 7). + return source[0] not in jobs_after and not may_still_meet(source, target) # A rule moved when the launch that met it left the losing job. The same # launch arriving is matched first, so which of two gaining jobs a rule diff --git a/src/agents_shipgate/report/host_comparison.py b/src/agents_shipgate/report/host_comparison.py index 1a17e43cf..16d0a4d5a 100644 --- a/src/agents_shipgate/report/host_comparison.py +++ b/src/agents_shipgate/report/host_comparison.py @@ -281,19 +281,32 @@ def _side_text(item: HostComparisonCoverageItem) -> str: #: comparison compares them (#812). _REDACTED_VALUES = "(redacted values such as env values and apiKeyHelper are not compared)" +#: The same note for a workflow, which holds no `apiKeyHelper`: what its +#: grant does not compare, and where the unread agent steps are named (#823 +#: review cycle 7). `diff`, `verify` and `check` carry no limit for such a +#: step, so without this a change that only adds one reads as covered. +_WORKFLOW_UNREAD_TEXT = ( + "(text this entry does not read, such as a step's env or an unread agent step, " + "is not compared; audit --host names each unread agent step)" +) + def _redacted_values_note(source: str) -> str: - """The note on redacted values, only for a file that can hold them (#812 review cycle 5). + """The note on what is not compared, only for a file that can hold it (#812 review cycle 5). Not for a file the inventory reads as instructions (`AGENTS.md`, a `CLAUDE.md` link to it, a skill or a rule): it holds no `env` value or `apiKeyHelper`, and a docs-only change would print the note on every such - file it could not prove unchanged. The kind is the one the inventory gives - the file's path when it reads it; a path redacted past recognition keeps - the note. + file it could not prove unchanged. A workflow's note names what its grant + does not read instead (#823). The kind is the one the inventory gives the + file's path when it reads it; a path redacted past recognition keeps the + note. """ - return "" if _source_kind(source) == "instructions" else f" {_REDACTED_VALUES}" + kind = _source_kind(source) + if kind == "instructions": + return "" + return f" {_WORKFLOW_UNREAD_TEXT if kind == 'workflow' else _REDACTED_VALUES}" #: What each candidate rule names (#821), as the line says it. None of these diff --git a/src/agents_shipgate/schemas/host_grants.py b/src/agents_shipgate/schemas/host_grants.py index 44cad7cae..57c3ffec5 100644 --- a/src/agents_shipgate/schemas/host_grants.py +++ b/src/agents_shipgate/schemas/host_grants.py @@ -776,7 +776,8 @@ class HostWorkflowUnreadAgentRunV7(BaseModel): quote, an expansion, a redirection, a comment, a continuation, a ``${{ }}`` expression, another program such as ``npx`` or ``timeout``, a subcommand that is not a headless launch, or a declared ``shell:`` other - than ``bash`` or ``sh`` — once for each agent CLI it mentions. It is a + than ``bash`` or ``sh`` run on the script alone (so ``bash -c '…' {0}`` + too) — once for each agent CLI it mentions. It is a named, non-blocking limit and nothing more: none of the step's text is published, it is never compared, so adding, removing or editing it gives no row, and it never says that the step starts, or does not start, an diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index a8ee873b8..39e87c0de 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -236,7 +236,11 @@ def paths(self) -> list[Path]: # triggers, write scopes, secrets and checkout refs the engine already # published on the grant, so it adds no claim; # `tests/test_workflow_agent_launches.py` holds diff, verify, the PR - # comment, check and the control envelope to the same row. + # comment, check and the control envelope to the same row. A + # workflow's coverage line naming what its grant does not read, in + # place of the redacted-values note, is fixed text chosen by the + # source's kind the inventory gives its path, and decides nothing, so + # it adds no claim either. {}, ), Surface( diff --git a/tests/test_workflow_agent_launches.py b/tests/test_workflow_agent_launches.py index 1aac61c1e..661c9318e 100644 --- a/tests/test_workflow_agent_launches.py +++ b/tests/test_workflow_agent_launches.py @@ -409,6 +409,29 @@ def test_job_secrets_read_a_yaml_alias_that_holds_itself_or_fans_out_once(): assert time.monotonic() - started < 5 +def test_job_secrets_read_unterminated_expressions_in_linear_time(): + """#823 review cycle 7 (C7-F2): every unterminated `${{` scanned to the end, 92 s at 300 KB. + + An unterminated expression names no secret, as before; a closed one + after it still does. + """ + + unterminated = "${{ secrets.NEVER " * 20_000 # about 360 KB, no closing `}}` + job = {"env": {"X": unterminated, "Y": "${{ secrets.CLOSED }}"}, "steps": [{"run": "claude -p Review"}]} + started = time.monotonic() + launch, = _launches(_workflow(jobs={"review": job})) + assert launch["job_secrets"] == ["CLOSED"] + # The same text in an agent action's settings input, which took as long. + row, = _rows(_workflow(_agent()), _workflow(_agent(settings="${{" * 100_000))) + assert (row.direction, row.expands) == ("changed", False) + assert time.monotonic() - started < 5 + # The body of an expression that closes after an unterminated one is still read. + launch, = _launches(_workflow(jobs={"review": { + "env": {"X": "${{ vars.A ${{ secrets.INNER }}"}, "steps": [{"run": "claude -p Review"}], + }})) + assert launch["job_secrets"] == ["INNER"] + + # --- the only forms read: plain lists of words (#823 review cycle 4) -------------------- # # Four review cycles each found a shell form the `run:` reader mis-read, so no @@ -581,8 +604,25 @@ def test_a_run_this_reader_does_not_parse_is_a_named_limit_that_gives_no_row_and {"step": "${{ matrix.shell }}"}, {"job": "powershell"}, {"workflow": "pwsh"}, + # #823 review cycle 7: a bash or sh template that may run a command of its own + {"step": "bash -c 'claude -p --dangerously-skip-permissions Review' {0}"}, + {"step": "bash -ec true {0}"}, + {"job": "bash --rcfile .ci/rc {0}"}, + {"step": "bash --init-file .ci/rc {0}"}, + {"step": "sh -e {0} extra"}, + {"step": "bash -e"}, + {"step": "bash -o {0}"}, + {"workflow": "bash -O ${{ vars.OPTION }} {0}"}, + # one that runs no command, or reads its commands from elsewhere + {"step": "bash -n {0}"}, + {"step": "bash -o noexec {0}"}, + {"step": "sh -s {0}"}, + {"step": "bash --version {0}"}, ], - ids=["step-pwsh", "step-python", "step-cmd", "step-expression", "job-default", "workflow-default"], + ids=["step-pwsh", "step-python", "step-cmd", "step-expression", "job-default", "workflow-default", + "bash-c-template", "short-cluster-with-c", "rcfile", "init-file", "words-after-the-script", + "no-script-slot", "option-without-its-name", "option-name-from-an-expression", + "noexec-flag", "noexec-option", "stdin", "version"], ) def test_a_run_under_a_declared_shell_other_than_bash_or_sh_is_not_read(shell): step = {"run": "claude -p --dangerously-skip-permissions go"} @@ -599,7 +639,11 @@ def test_a_run_under_a_declared_shell_other_than_bash_or_sh_is_not_read(shell): assert _unread(workflow) == [{"job": "review", "step": "steps[0]", "agent": "claude"}] -@pytest.mark.parametrize("shell", ["bash", "sh", "bash -e {0}", "/bin/bash --noprofile --norc -eo pipefail {0}"]) +@pytest.mark.parametrize( + "shell", + ["bash", "sh", "bash -e {0}", "/bin/bash --noprofile --norc -eo pipefail {0}", "sh -eu {0}", + "bash -l {0}", "bash -O extglob -o pipefail {0}"], +) def test_a_run_under_bash_or_sh_is_read(shell): step = {"run": "claude -p --dangerously-skip-permissions go", "shell": shell} launch, = _launches(_workflow(step)) @@ -1123,8 +1167,17 @@ def test_renaming_a_job_that_launches_a_bypassing_agent_is_not_a_widening(): review=[{"run": "make"}]), _jobs(lint=[{"run": "make"}], review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # #823 review cycle 7: a job renamed with the install step it had + # beside the launch; the unread step is in the job gaining the rule + (_jobs(review=[{"run": "npm i -g @anthropic-ai/claude-code"}, {"name": "agent", "run": f"claude -p {BYPASS} Review"}]), + _jobs(**{"code-review": [{"run": "npm i -g @anthropic-ai/claude-code"}, + {"name": "agent", "run": f"claude -p {BYPASS} Review"}]})), + # a job renamed beside another job that keeps the unread step it had + (_jobs(review=[_named(BYPASS)], setup=[{"run": "claude mcp add x"}]), + _jobs(**{"code-review": [_named(BYPASS)]}, setup=[{"run": "claude mcp add x"}])), ], - ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved", "moved-beside-a-step-that-is-no-launch"], + ids=["step-moved", "renamed-and-edited", "swapped", "gate-moved", "moved-beside-a-step-that-is-no-launch", + "renamed-with-its-install-step", "renamed-beside-an-unread-step-another-job-keeps"], ) def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [] @@ -1194,13 +1247,42 @@ def test_a_launch_that_left_one_job_for_another_moves_its_rules(before, after): review=[{"run": "make"}]), _jobs(lint=[{"name": "agent", **_agent(settings='{"permissions":{"defaultMode":"${{ vars.MODE }}"}}')}], review=[{"name": "agent", **_agent(settings='{"permissions":{"defaultMode":"bypassPermissions"}}')}])), + # #823 review cycle 7 (R6): the job it met it in is renamed and quotes + # the prompt, so under its new name it still runs the launch, unread + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), + _jobs(**{"lint-renamed": [{"name": "agent", "run": f'claude -p {BYPASS} "Review"'}]}, + review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}])), + # (R7) renamed and run through `npx`, the other job spelling the bypass as a mode + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}]), + _jobs(**{"lint-renamed": [{"name": "agent", "run": f"npx @anthropic-ai/claude-code -p {BYPASS} Review"}]}, + review=[{"name": "agent", "run": "claude -p --permission-mode bypassPermissions Review"}])), + # (R3) the job it met it in is removed, and an existing job gains the quoted launch + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], review=[{"run": "echo hi"}], + docs=[{"run": "make"}]), + _jobs(review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], + docs=[{"run": "make"}, {"name": "agent", "run": f'claude -p {BYPASS} "Review"'}])), + # the same while the job it left remains + (_jobs(lint=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}, {"run": "make"}], + review=[{"run": "echo hi"}], docs=[{"run": "make"}]), + _jobs(lint=[{"run": "make"}], review=[{"name": "agent", "run": f"claude -p {BYPASS} Review"}], + docs=[{"run": "make"}, {"name": "agent", "run": f'claude -p {BYPASS} "Review"'}])), + # renamed into an action whose argument input this audit does not read + (_jobs(lint=[_named(BYPASS)], review=[{"run": "echo hi"}]), + _jobs(**{"lint-renamed": [_named(f'{BYPASS} --append-system-prompt "Review"')]}, + review=[_named(BYPASS)])), + # Two jobs renamed at once, one holding an unread step: which new job + # is which is not told, so the gain is claimed (the safe direction). + (_jobs(lint=[_named(BYPASS)], setup=[{"run": "npm i -g @anthropic-ai/claude-code"}]), + _jobs(review=[_named(BYPASS)], prepare=[{"run": "npm i -g @anthropic-ai/claude-code"}])), ], ids=["second-job", "narrowed-there-widened-here", "unread-in-the-job-it-met-it", "edited-into-the-same-launch", "moved-and-edited", "quoted-in-the-job-it-met-it", "npx-in-the-job-it-met-it", "unread-at-another-step-in-the-job-it-met-it", "install-merged-into-the-launch", "unread-step-removed-while-the-launch-becomes-unread", "beside-an-unread-step-that-stays", "arguments-unread-in-the-job-it-met-it", - "setting-expression-in-the-job-it-met-it"], + "setting-expression-in-the-job-it-met-it", + "renamed-and-quoted", "renamed-and-npx", "removed-while-another-job-gains-it-quoted", + "left-for-another-job-quoted", "renamed-into-unread-arguments", "two-jobs-renamed"], ) def test_a_rule_another_job_gains_while_no_launch_left_is_a_widening(before, after): assert host_grant_expansion_signals(_changes(before, after)) == [f"workflow_agent_widened_changed: {SOURCE}"] @@ -1946,6 +2028,28 @@ def test_the_run_forms_of_review_cycle_4_give_no_row_and_are_a_named_coverage_is assert text not in joined +def test_a_change_that_only_adds_an_unread_step_says_what_the_workflow_does_not_compare(tmp_path): + """#823 review cycle 7 (carried P3): the coverage line named env values and apiKeyHelper.""" + + base = _workflow({"uses": "actions/checkout@v4"}) + head = _workflow({"uses": "actions/checkout@v4"}, {"run": f"claude -p {BYPASS} 'Review'"}) + repo = _repo(tmp_path, {SOURCE: _yaml(base)}) + _git(repo, "checkout", "-qb", "change") + _write(repo, {SOURCE: _yaml(head)}) + _git(repo, "commit", "-qam", "an unread agent step") + + result = CliRunner().invoke(app, ["diff", "--workspace", str(repo), "--base", "main"]) + assert result.exit_code == 0, result.output + assert "No static host-grant changes detected." in result.output + assert ( + f"{SOURCE} (github): compared; changed, but no grant this entry compares changed, so no row " + "(text this entry does not read, such as a step's env or an unread agent step, is not " + "compared; audit --host names each unread agent step)" + ) in result.output + assert "apiKeyHelper" not in result.output + assert "dangerously" not in result.output + + @pytest.mark.parametrize( "still_there", [f'claude -p {BYPASS} "Review"', f"npx @anthropic-ai/claude-code -p {BYPASS} Review"],