From a54e3a29c417a857d88a52939707d56edd807778 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Tue, 1 Sep 2026 16:29:06 -0700 Subject: [PATCH 001/397] feat(engine): fix loops, votes, packet pins, gates, and registry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 56 commits hardening the run engine and the CLI around it. Workflow grammar — `when` predicates gain `or` and `contains_any`; steps can declare a severity floor and stall limit; cross-issue linked input form; `max_fix_loops` counting documented. Fix loops — entry scoped to per-trigger clusters; loop parks when the routing verdict repeats or the fix commit didn't move; steering notes carry into the next round; run-scoped batch gate-override covers repeated identical failures (DKT-546). Votes — approved votes route on concerned casts; vote records exposed; retry refused once a proposal has decided; silent seats named with their path; weighted score separated from ballot count; vote-step spend counted against the run budget. Claims and budget — claims can declare a variant-scaled cost; a forced reap no longer spends the attempt budget; routing metadata recorded at claim; run report pairs requested and resolved tier; aggregate builtin gets a routing floor; warning when usage is backfilled against an unclaimed step. Packet pins — pin only the closure a run's workflows reach; check the pin set is closed; validate a packet before taking a lease; repin drops gone files and adds newly required ones; registered workflow source drift detected on disk and shown in run reports. Gates — gate processes get the step's base commit; stub gate results marked with a reason and tracking issue; override-pass refused when it would skip a gate; failed gate named when a step parks; advisory pre-gate verdicts marked; gates named as make targets. Registry and scope — cross-project workflow/schema registration and a drift audit; orphaned registrations surfaced; `--deprecated` hides retired versions and `show NAME` skips them (DKT-616); a widened scope can reach an active run, with a warning when it can't; unresolvable stale targets flagged and waivable. CLI fixes — `events follow --tail` starts at the newest N; `events list` finds a run in any project; `next --run` returns the full ready set; dispatch close does backfill and verify in one step; trust changes refuse a missing actor/cwd; loop redirects no longer bind an unrelated step's output; operator pause survives a step record; in-repo docket skill copy retired. --- .github/workflows/ci.yaml | 176 +- Makefile | 79 +- README.md | 23 +- docs/design/engine-spec.md | 80 +- docs/tdd/completion-metadata.md | 121 +- docs/tdd/engine-spine.md | 66 +- docs/tdd/gates-trust.md | 57 + docs/tdd/payloads-thresholds.md | 32 +- docs/tdd/reliability-delta.md | 69 + internal/cli/dispatch.go | 313 +- internal/cli/dispatch_reconcile_test.go | 300 ++ internal/cli/dispatch_test.go | 146 + internal/cli/dispatch_waive_test.go | 102 + internal/cli/events.go | 64 +- internal/cli/events_follow.go | 48 +- internal/cli/events_follow_test.go | 139 +- internal/cli/events_test.go | 221 + internal/cli/issue_edit.go | 340 +- internal/cli/issue_edit_scope_test.go | 175 + internal/cli/issue_show.go | 12 +- internal/cli/next.go | 2 +- internal/cli/next_steps.go | 11 + internal/cli/next_steps_test.go | 114 +- internal/cli/registry.go | 233 + internal/cli/registry_audit_test.go | 252 + internal/cli/registry_fanout.go | 344 ++ internal/cli/registry_fanout_test.go | 514 ++ internal/cli/root.go | 32 +- internal/cli/run_activate.go | 29 + internal/cli/run_refresh_scope.go | 142 + internal/cli/run_refresh_scope_test.go | 195 + internal/cli/run_repin.go | 69 +- internal/cli/run_repin_drop_test.go | 142 + internal/cli/run_report.go | 292 +- internal/cli/run_report_metadata_test.go | 179 + internal/cli/run_report_pins_test.go | 155 + internal/cli/run_report_render_test.go | 178 + internal/cli/run_verify_pins.go | 38 +- internal/cli/run_verify_pins_test.go | 138 + internal/cli/schema_register.go | 101 +- internal/cli/step.go | 302 +- internal/cli/step_gates_test.go | 108 + internal/cli/step_resolve_interposed_test.go | 227 + internal/cli/step_test.go | 250 + internal/cli/trust.go | 76 +- internal/cli/trust_event_test.go | 173 + internal/cli/workflow_deprecate.go | 90 + .../workflow_deprecated_visibility_test.go | 44 +- internal/cli/workflow_lint.go | 15 + internal/cli/workflow_list.go | 121 +- internal/cli/workflow_orphans_test.go | 218 + internal/cli/workflow_register.go | 147 +- internal/cli/workflow_show.go | 30 +- internal/cli/workflow_source_test.go | 171 + internal/cli/workflow_test.go | 2 + internal/db/engineconfig.go | 19 + internal/db/gate_override_grants.go | 124 + internal/db/idempotency.go | 32 + internal/db/migrate_v10_test.go | 17 +- internal/db/migrate_v24_test.go | 111 + internal/db/migrate_v25_test.go | 112 + internal/db/migrate_v7_test.go | 2 + internal/db/proposals.go | 34 +- internal/db/rollups.go | 216 +- internal/db/runs.go | 28 + internal/db/schema.go | 141 +- internal/db/stale_target_waivers.go | 86 + internal/db/steps.go | 56 + internal/db/workflows.go | 80 +- internal/db/workflows_test.go | 63 + internal/engine/action.go | 2 +- internal/engine/action_exec.go | 23 +- internal/engine/action_test.go | 90 + internal/engine/activate.go | 236 +- internal/engine/activate_test.go | 80 +- internal/engine/aggregate.go | 120 +- internal/engine/aggregate_test.go | 182 +- internal/engine/autoregister.go | 29 +- internal/engine/autoregister_test.go | 56 +- internal/engine/backfill.go | 75 +- internal/engine/backfill_test.go | 205 + internal/engine/batch_override.go | 185 + internal/engine/batch_override_test.go | 618 +++ internal/engine/budget.go | 156 +- internal/engine/claim.go | 216 +- internal/engine/context.go | 147 +- internal/engine/dispatch.go | 131 +- internal/engine/dispatch_reconcile.go | 161 + internal/engine/dispatch_reconcile_test.go | 293 ++ internal/engine/dispatch_stale_test.go | 10 +- internal/engine/dkt470_test.go | 15 +- internal/engine/dkt545_test.go | 412 ++ internal/engine/dkt547_test.go | 417 ++ internal/engine/dkt584_test.go | 305 ++ internal/engine/dkt586_test.go | 223 + internal/engine/dkt587_test.go | 403 ++ internal/engine/dkt588_test.go | 262 + internal/engine/dkt589_test.go | 482 ++ internal/engine/dkt590_test.go | 267 + internal/engine/dkt591_test.go | 269 ++ internal/engine/dkt594_test.go | 281 ++ internal/engine/dkt609_test.go | 184 + internal/engine/dkt725_test.go | 116 + internal/engine/dkt726_test.go | 183 + internal/engine/dkt733_test.go | 165 + internal/engine/dkt741_test.go | 254 + internal/engine/dkt804_test.go | 95 + internal/engine/dkt805_test.go | 212 + internal/engine/dkt818_test.go | 110 + internal/engine/dkt820_test.go | 90 + internal/engine/dkt821_test.go | 293 ++ internal/engine/dkt861_test.go | 161 + internal/engine/dkt867_test.go | 249 + internal/engine/dkt868_test.go | 305 ++ internal/engine/dkt869_test.go | 456 ++ internal/engine/dkt870_test.go | 459 ++ internal/engine/dkt895_test.go | 150 + internal/engine/dkt982_test.go | 150 + internal/engine/dkt992_test.go | 209 + internal/engine/event.go | 103 +- internal/engine/events_project_test.go | 70 + internal/engine/events_read.go | 49 +- internal/engine/events_read_test.go | 39 +- internal/engine/events_test.go | 36 +- internal/engine/forced_reap_budget_test.go | 201 + internal/engine/gate.go | 13 + internal/engine/gate_exec.go | 1 + internal/engine/gate_preflight.go | 27 +- internal/engine/gate_preflight_test.go | 45 +- internal/engine/held.go | 58 +- internal/engine/held_test.go | 3 +- internal/engine/human.go | 198 +- internal/engine/linked.go | 308 ++ internal/engine/lookahead.go | 6 +- internal/engine/loop.go | 637 ++- internal/engine/loop_cluster_test.go | 326 ++ internal/engine/loop_reentry_test.go | 4 +- internal/engine/loop_starvation_test.go | 8 +- internal/engine/loop_test.go | 27 +- internal/engine/metadata.go | 15 +- internal/engine/metadata_test.go | 432 ++ internal/engine/orphan_registration.go | 249 + internal/engine/packet.go | 53 +- internal/engine/packet_closure.go | 192 + internal/engine/packet_closure_test.go | 301 ++ internal/engine/packet_test.go | 22 +- internal/engine/passfloor.go | 86 + internal/engine/pending_closure.go | 302 ++ internal/engine/pin_staleness.go | 408 ++ internal/engine/pregate.go | 4 +- internal/engine/pregate_test.go | 163 + internal/engine/ready.go | 7 +- internal/engine/reap_ack.go | 29 +- internal/engine/registry_audit.go | 428 ++ internal/engine/registry_audit_test.go | 319 ++ internal/engine/render.go | 37 +- internal/engine/repin.go | 399 +- internal/engine/repin_drop_test.go | 462 ++ internal/engine/repin_test.go | 15 +- internal/engine/report.go | 271 +- internal/engine/saga.go | 478 +- internal/engine/saga_resume.go | 8 +- internal/engine/scope_refresh.go | 375 ++ internal/engine/scope_snapshot.go | 214 + internal/engine/source_drift.go | 276 ++ internal/engine/stale_waiver_test.go | 344 ++ internal/engine/strand.go | 10 +- internal/engine/union_config_test.go | 42 +- internal/engine/verify_pins.go | 119 +- internal/engine/vote.go | 110 +- internal/engine/vote_record.go | 246 + internal/engine/waive.go | 109 + internal/exec/env.go | 16 + internal/exec/env_scope_test.go | 31 + internal/model/model_test.go | 37 + internal/model/relation.go | 40 + internal/model/workflow.go | 145 +- internal/output/human.go | 24 + internal/output/output.go | 23 + internal/output/output_test.go | 75 + internal/render/step_test.go | 39 + internal/trust/add.go | 22 +- internal/trust/parse.go | 8 + internal/trust/store.go | 17 + internal/trust/store_test.go | 50 +- internal/workflow/aggregate.go | 22 +- internal/workflow/expand.go | 39 +- internal/workflow/held.go | 9 + internal/workflow/lint.go | 7 + internal/workflow/match.go | 77 +- internal/workflow/match_test.go | 425 ++ internal/workflow/parse.go | 120 +- internal/workflow/schemas.go | 74 + internal/workflow/validate.go | 526 +- internal/workflow/validate_test.go | 787 +++ scripts/qa.sh | 16 +- scripts/qa/fixtures/context/step-10.golden | 1 + scripts/qa/fixtures/context/step-2.golden | 1 + scripts/qa/fixtures/context/step-3.golden | 1 + scripts/qa/fixtures/context/step-4.golden | 1 + scripts/qa/fixtures/context/step-5.golden | 1 + scripts/qa/fixtures/context/step-6.golden | 1 + scripts/qa/fixtures/context/step-7.golden | 1 + scripts/qa/fixtures/context/step-8.golden | 1 + scripts/qa/fixtures/context/step-9.golden | 1 + scripts/qa/gate-coverage-check.sh | 155 - scripts/qa/test_zg_workflow.sh | 16 +- skills/docket/SKILL.md | 4293 ----------------- 208 files changed, 29392 insertions(+), 5404 deletions(-) create mode 100644 internal/cli/dispatch_reconcile_test.go create mode 100644 internal/cli/dispatch_waive_test.go create mode 100644 internal/cli/issue_edit_scope_test.go create mode 100644 internal/cli/registry.go create mode 100644 internal/cli/registry_audit_test.go create mode 100644 internal/cli/registry_fanout.go create mode 100644 internal/cli/registry_fanout_test.go create mode 100644 internal/cli/run_refresh_scope.go create mode 100644 internal/cli/run_refresh_scope_test.go create mode 100644 internal/cli/run_repin_drop_test.go create mode 100644 internal/cli/run_report_metadata_test.go create mode 100644 internal/cli/run_report_pins_test.go create mode 100644 internal/cli/run_verify_pins_test.go create mode 100644 internal/cli/step_resolve_interposed_test.go create mode 100644 internal/cli/workflow_orphans_test.go create mode 100644 internal/cli/workflow_source_test.go create mode 100644 internal/db/gate_override_grants.go create mode 100644 internal/db/migrate_v24_test.go create mode 100644 internal/db/migrate_v25_test.go create mode 100644 internal/db/stale_target_waivers.go create mode 100644 internal/engine/batch_override.go create mode 100644 internal/engine/batch_override_test.go create mode 100644 internal/engine/dispatch_reconcile.go create mode 100644 internal/engine/dispatch_reconcile_test.go create mode 100644 internal/engine/dkt545_test.go create mode 100644 internal/engine/dkt547_test.go create mode 100644 internal/engine/dkt584_test.go create mode 100644 internal/engine/dkt586_test.go create mode 100644 internal/engine/dkt587_test.go create mode 100644 internal/engine/dkt588_test.go create mode 100644 internal/engine/dkt589_test.go create mode 100644 internal/engine/dkt590_test.go create mode 100644 internal/engine/dkt591_test.go create mode 100644 internal/engine/dkt594_test.go create mode 100644 internal/engine/dkt609_test.go create mode 100644 internal/engine/dkt725_test.go create mode 100644 internal/engine/dkt726_test.go create mode 100644 internal/engine/dkt733_test.go create mode 100644 internal/engine/dkt741_test.go create mode 100644 internal/engine/dkt804_test.go create mode 100644 internal/engine/dkt805_test.go create mode 100644 internal/engine/dkt818_test.go create mode 100644 internal/engine/dkt820_test.go create mode 100644 internal/engine/dkt821_test.go create mode 100644 internal/engine/dkt861_test.go create mode 100644 internal/engine/dkt867_test.go create mode 100644 internal/engine/dkt868_test.go create mode 100644 internal/engine/dkt869_test.go create mode 100644 internal/engine/dkt870_test.go create mode 100644 internal/engine/dkt895_test.go create mode 100644 internal/engine/dkt982_test.go create mode 100644 internal/engine/dkt992_test.go create mode 100644 internal/engine/forced_reap_budget_test.go create mode 100644 internal/engine/linked.go create mode 100644 internal/engine/loop_cluster_test.go create mode 100644 internal/engine/orphan_registration.go create mode 100644 internal/engine/packet_closure.go create mode 100644 internal/engine/packet_closure_test.go create mode 100644 internal/engine/passfloor.go create mode 100644 internal/engine/pending_closure.go create mode 100644 internal/engine/pin_staleness.go create mode 100644 internal/engine/registry_audit.go create mode 100644 internal/engine/registry_audit_test.go create mode 100644 internal/engine/repin_drop_test.go create mode 100644 internal/engine/scope_refresh.go create mode 100644 internal/engine/scope_snapshot.go create mode 100644 internal/engine/source_drift.go create mode 100644 internal/engine/stale_waiver_test.go create mode 100644 internal/engine/vote_record.go create mode 100644 internal/engine/waive.go create mode 100644 internal/render/step_test.go delete mode 100755 scripts/qa/gate-coverage-check.sh delete mode 100644 skills/docket/SKILL.md diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index 8eb8db9c..ee2fc520 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -1,106 +1,14 @@ name: ci -# GATE COVERAGE. scripts/qa/ holds seventeen scripts outside the -# test_*.sh suite; two of those (ac-commands.sh, helpers.sh) are not gates — -# see below. This block is the one place that names, for each of the -# remaining fifteen gates AS COUNTED HERE AND NOW (this figure moves as gate -# scripts are added or retired — it is not the sixteen -# docs/spec/review-strategy.md's own count names, since that count predates -# copy-verify.sh, render-verify.sh, and later additions and retirements), -# whether CI runs it and why not when it doesn't — docs/spec/review-strategy.md -# §10 (G2) is the gap this closes. -# scripts/qa/gate-coverage-check.sh checks this block against the directory -# mechanically, so this listing cannot drift silently. -# -# RUNS HERE, by name, in `repo-gates`. On `pull_request` AND on every -# `push` (git-diff scripts in base-ref mode; the base is `base.sha` for a -# PR and `github.event.before` for a push, falling back to the root commit -# when that is the all-zero SHA — see the `repo-gates` job comment below -# for the one gap that remains, which is that this job reports rather than -# blocks): -# secret-scan.sh, self-hygiene.sh -# -# RUNS ELSEWHERE IN THIS FILE, already: -# genericity.sh — invoked by `./scripts/qa.sh` (`qa` job) on -# every `pull_request` and every `push` to -# `main`/a tag (this file's own `on:` block). -# copy-verify.sh — same: invoked by `./scripts/qa.sh`. Both are -# render-verify.sh whole-repo, diff-independent checks (no -# DOCKET_* env needed — verified empty-env), -# the same shape as genericity.sh, not tied -# to a specific workflow step or run/step -# identity. -# THIS BLOCK IS AUTHORITATIVE over the -# earlier acceptance criterion that listed -# these two as excluded. That AC was written from the -# premise that both need run context; the -# premise was false — neither reads a -# DOCKET_* variable, and both exit 0 under -# `env -i` — so they are RUN, which is -# strictly more coverage than the AC asked -# for. The other five it named are still -# excluded, below. -# render-verify.sh runs here with ONE HALF -# VACUOUS: its coverage check reads -# working-tree/staged/untracked state, which -# a `pull_request` checkout does not have, -# so that half reports and cannot fail. Its -# ANSI-escape half is whole-repo and does -# bite here. See the script's own comment at -# the `changed=` assignment. That is the -# same criterion the EXCLUDED block below -# uses; render-verify still earns its place -# here because its second half is real, -# whereas those four are vacuous entire. -# gate-baseref-regression.sh — also invoked by `./scripts/qa.sh`; pins the -# gate-coverage-check.sh base-ref mode of secret-scan.sh/ -# self-hygiene.sh and this block itself. -# build.sh — its check (`go1.26.6 build ./...`) is what -# the `test` job's `go build ./...` step -# already verifies, given the same Go -# toolchain: `actions/setup-go`'s -# `go-version-file: go.mod` resolves the -# `toolchain go1.26.6` line there, which the -# `test` job's own `go version` step below -# makes self-checking rather than assumed. -# tests.sh — same equivalence, for `go test ./...`. -# -# EXCLUDED — read only working-tree/staged/untracked state, with no -# base-ref mode (the gap secret-scan.sh/self-hygiene.sh had before this fix -# round) to read a committed PR diff instead. CI has no staged or unstaged -# state, so wiring these as-is would pass every PR vacuously rather than -# fail closed. Adding base-ref mode is the same three-line pattern applied -# to secret-scan.sh/self-hygiene.sh; deferred to a follow-up rather than -# done in this round — the missing base-ref mode is the actual blocker, -# not run/step identity (each already runs with an empty DOCKET_* env): -# doc-validate.sh, citation-check.sh, tdd-preflight.sh, -# reserved-name-check.sh -# -# EXCLUDED — for other stated reasons: -# sdet-abuse.sh — needs the `go1.26.6` binary by that exact name on -# PATH (the script invokes it directly, not `go test`); -# `actions/setup-go` does not add a version-suffixed -# alias. It also cannot verify its cases match a run's -# threat model without step identity — see the script's -# own header. Either gap is enough; deferred, not new -# CI infrastructure this round. -# vuln-scan.sh — needs `govulncheck` installed; deferred as a follow-up -# so this change stays wiring, not new CI infrastructure. -# ac-commands.sh — a reporter for the `verify` step's context bundle; -# ALWAYS EXITS 0 by design (see the script's own header), -# so it has no pass/fail verdict for CI to act on. Not a -# gate. -# helpers.sh — shared functions `qa.sh`'s own test files source; not -# an executable gate. - on: + # pull_request covers every commit on a branch with an open PR. push is + # scoped to main/tags only, so that branch's commits aren't built twice — + # once for pull_request, once for a push the "**" pattern used to also + # match. pull_request: push: - # ALL branches, not just `main`. This repository's work lands as - # direct commits to feature branches, so a `main`-only push trigger meant - # the gates in `repo-gates` never ran on the path the code actually takes. branches: - - "**" + - main tags: - "*" @@ -115,25 +23,6 @@ jobs: go-version-file: go.mod cache: true - # Asserts, not just prints: the GATE COVERAGE block claims this job - # covers build.sh/tests.sh's version-pinned invocations because setup-go - # resolves go.mod's toolchain line — a bare `go version` step cannot - # fail, so it was evidence of nothing. `go build ./...` below already - # fails if go is missing entirely; this is the one property that step - # alone does not check. - # - # The expected version is READ FROM go.mod, never hard-coded here. - # go.mod carries two version lines — `go 1.26.0` and - # `toolchain go1.26.6` — and a literal here pinned only the second. A - # toolchain bump then had to be made in two files or CI failed every - # PR naming a version go.mod no longer contains. Deriving it keeps the - # property (setup-go really did resolve the toolchain line) while - # making go.mod the single place the version lives. - # - # ACCEPTING EITHER LINE is deliberate. Which one setup-go resolves is - # its behaviour, not ours; the property worth asserting is that the Go - # on PATH is one of the versions go.mod actually names, so a bump to - # either line stays green and an unrelated toolchain still fails. - run: | go_directive=$(sed -n 's/^go \([0-9.]*\)$/go\1/p' go.mod) toolchain=$(sed -n 's/^toolchain \(go[0-9.]*\)$/\1/p' go.mod) @@ -153,13 +42,6 @@ jobs: - run: go build ./... - # internal/workflow/routing_sweep_test.go sweeps the workflow corpus - # config.Config.InstanceConfigDirs resolves, which on a bare CI - # checkout is empty — there is no shared store at $HOME/.docket and - # this repo ships no `.docket/config` of its own. Pointing the sweep - # at the hermetic fixture under internal/workflow/testdata gives it a - # corpus without unioning into anyone's real shared store (see that - # directory's README.md for why it must stay override-only). - run: go test ./... env: DOCKET_SWEEP_WORKFLOWS_DIR: ${{ github.workspace }}/internal/workflow/testdata/ci-routing-corpus @@ -174,46 +56,8 @@ jobs: go-version-file: go.mod cache: true - # jq and perl are preinstalled on ubuntu-latest; qa.sh hard-fails without them. - run: ./scripts/qa.sh - # repo-gates. The scripts here read a COMMITTED diff, not - # working-tree state, so they need the PR's base commit fetched and passed - # explicitly — see the base-ref mode each script's header now documents. - # - # KNOWN LIMIT: `BASE_SHA` below is `github.event.pull_request.base.sha`, the - # base branch tip at event time — not necessarily an ancestor of the - # checked-out `pull_request` ref's own history. If the base branch advances - # after the PR's last push, the scanned range can widen to base-branch - # commits the PR did not author (INFERRED — no local seam to exercise a - # real Actions `pull_request` event to confirm this against a live run). - # - # RUNS ON PUSH AS WELL AS PULL_REQUEST. - # - # `pull_request`-only was a gate that never fired on this repository's - # actual regime. Work here lands as direct commits to a feature branch - # (`git log --merges` is empty), and `ci-nightly.yaml` tags `main`'s HEAD - # daily and publishes through the push-on-tags path. Both paths excluded - # `repo-gates`, so the credential gate ran on nothing that actually - # happens here — literally satisfied, practically absent. - # - # The base ref differs per event, which is the whole reason this was - # deferred once: - # pull_request — `base.sha`, the base branch tip at event time. - # push — `github.event.before`, the branch's previous tip. - # - # `before` IS THE ALL-ZERO SHA ON A BRANCH'S FIRST PUSH (and on a tag - # push), which is not a readable ref. That case is handled explicitly - # below rather than left to fail confusingly: it falls back to the - # repository's first commit, so the scan covers the entire branch. For a - # brand-new branch that is the correct interpretation — every commit on it - # is new — and it fails toward scanning MORE, never less. - # - # STILL UNGATED, stated plainly: nothing here can make - # this a REQUIRED check. Branch protection is not version-controlled in - # this repo, so a push whose workflow run fails still lands. This job - # reports; it does not block. Making it blocking is a repository-settings - # change, outside the tree. repo-gates: runs-on: ubuntu-latest steps: @@ -221,16 +65,6 @@ jobs: with: fetch-depth: 0 - # Resolved in its own step so the fallback is testable as shell rather - # than buried in a `${{ }}` ternary, and so both gates below read one - # value that has already been validated. - # - # Named via `env:` rather than interpolated straight into `run:`. - # GitHub substitutes `${{ }}` into the step's text before the shell - # parses it; both values are GitHub-computed rather than contributor - # text, so there is no injection here today, but the indirection is the - # shape that stays safe if a future edit swaps in something - # contributor-controlled. - id: base env: PR_BASE: ${{ github.event.pull_request.base.sha }} diff --git a/Makefile b/Makefile index 847abd06..1e49c5c3 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,22 @@ -.PHONY: build test lint vet install clean demo +# Developer targets and the gate targets `docket trust list` invokes. +# +# The gate targets exist so a trust entry can name `make ` instead of an +# absolute path into one worktree's checkout. Trust entries are bound to the +# repository, not the worktree, and the engine spawns each gate with cwd set to +# the step's worktree — an absolute path pins every worktree's gate to whichever +# checkout happened to be current when the entry was approved, and goes dead the +# day that worktree is removed. `make ` resolves against the caller's cwd, +# so it is correct from every worktree and survives their coming and going. +# +# Each gate target's NAME MATCHES ITS TRUST ENTRY NAME exactly, so a workflow's +# `gates = ["build", "tests"]` reads the same as the Makefile and the same as +# `docket trust list`. That is why the compile gate owns the name `build` and +# the binary build is `make bin`. + +.PHONY: bin test lint vet install clean demo \ + build tests self-hygiene doc-validate citation-check secret-scan \ + vuln-scan sdet-abuse tdd-preflight reserved-name-check \ + render-verify copy-verify ac-commands VERSION ?= $(shell git describe --tags --always --dirty 2>/dev/null || echo "dev") COMMIT ?= $(shell git rev-parse --short HEAD 2>/dev/null || echo "none") @@ -6,7 +24,15 @@ BUILD_DATE ?= $(shell date -u '+%Y-%m-%dT%H:%M:%SZ') LDFLAGS := -X github.com/ALT-F4-LLC/docket/internal/cli.version=$(VERSION) -X github.com/ALT-F4-LLC/docket/internal/cli.commit=$(COMMIT) -X github.com/ALT-F4-LLC/docket/internal/cli.buildDate=$(BUILD_DATE) -build: +QA := scripts/qa + +# --------------------------------------------------------------------------- +# Developer targets +# --------------------------------------------------------------------------- + +# bin — produce ./bin/docket. Formerly `build`; renamed so the compile gate can +# own that name (see the header). +bin: CGO_ENABLED=0 go build -ldflags "$(LDFLAGS)" -o ./bin/docket ./cmd/docket test: @@ -24,6 +50,53 @@ install: clean: rm -rf ./bin/ -demo: build +demo: bin @command -v vhs >/dev/null 2>&1 || { echo "vhs is required: brew install vhs"; exit 1; } vhs scripts/demo.tape + +# --------------------------------------------------------------------------- +# Gate targets — one per `docket trust list` entry +# +# Each delegates to the script that already carries the gate's logic and its +# failure explanation; this file adds a stable, worktree-independent name and +# nothing else. A gate's behavior is changed in its script, not here. +# --------------------------------------------------------------------------- + +build: + @bash $(QA)/build.sh + +tests: + @bash $(QA)/tests.sh + +self-hygiene: + @bash $(QA)/self-hygiene.sh + +doc-validate: + @bash $(QA)/doc-validate.sh + +citation-check: + @bash $(QA)/citation-check.sh + +secret-scan: + @bash $(QA)/secret-scan.sh + +vuln-scan: + @bash $(QA)/vuln-scan.sh + +sdet-abuse: + @bash $(QA)/sdet-abuse.sh + +tdd-preflight: + @bash $(QA)/tdd-preflight.sh + +reserved-name-check: + @bash $(QA)/reserved-name-check.sh + +render-verify: + @bash $(QA)/render-verify.sh + +copy-verify: + @bash $(QA)/copy-verify.sh + +ac-commands: + @bash $(QA)/ac-commands.sh diff --git a/README.md b/README.md index a37ffe40..99a297c6 100644 --- a/README.md +++ b/README.md @@ -40,26 +40,7 @@ DOCKET_INSTALL_DIR=/usr/local/bin curl -fsSL https://raw.githubusercontent.com/A ### Agent Skill Setup -If you're working with Docket through an AI coding agent, also install the companion skill so the agent has the full CLI workflow and command reference available without re-deriving usage from `--help` output. Copy this repo's [`skills/docket/SKILL.md`](skills/docket/SKILL.md) into your agent's skill directory: - -| Harness | Project-scoped | User-scoped | -|---------|-----------------|-------------| -| Claude Code | `.claude/skills/docket/SKILL.md` | `~/.claude/skills/docket/SKILL.md` | -| Codex | `.agents/skills/docket/SKILL.md` | `~/.agents/skills/docket/SKILL.md` | -| Opencode | `.opencode/skills/docket/SKILL.md` | `~/.config/opencode/skills/docket/SKILL.md` | - -```bash -# Claude Code (project-scoped) -mkdir -p .claude/skills/docket && cp skills/docket/SKILL.md .claude/skills/docket/SKILL.md - -# Codex (project-scoped) -mkdir -p .agents/skills/docket && cp skills/docket/SKILL.md .agents/skills/docket/SKILL.md - -# Opencode (project-scoped) -mkdir -p .opencode/skills/docket && cp skills/docket/SKILL.md .opencode/skills/docket/SKILL.md -``` - -See [Drop-in Skill](#drop-in-skill) for what the skill teaches. +If you're working with Docket through an AI coding agent, also install the companion skill so the agent has the full CLI workflow and command reference available without re-deriving usage from `--help` output. The skill is maintained in [ALT-F4-LLC/dotfiles.vorpal](https://github.com/ALT-F4-LLC/dotfiles.vorpal) at `src/user/claude_code/skills/docket/`; copy that directory into your agent's skill directory (e.g. `.claude/skills/docket/` project-scoped, or `~/.claude/skills/docket/` user-scoped for Claude Code). ### From Source @@ -142,7 +123,7 @@ Any agent that can run shell commands works with Docket. Point it at `docket nex ### Drop-in Skill -[`skills/docket/SKILL.md`](skills/docket/SKILL.md) is a thorough reference teaching the full Docket CLI workflow and command/flag reference in one file, so your agent doesn't need to re-derive usage from `--help` output. See [Agent Skill Setup](#agent-skill-setup) in the Installation section above for where to drop it in for Claude Code, Codex, and Opencode. +A companion skill teaching the full Docket CLI workflow and command/flag reference is maintained in [ALT-F4-LLC/dotfiles.vorpal](https://github.com/ALT-F4-LLC/dotfiles.vorpal) at `src/user/claude_code/skills/docket/`, so your agent doesn't need to re-derive usage from `--help` output. See [Agent Skill Setup](#agent-skill-setup) in the Installation section above.
Verbose JSON examples diff --git a/docs/design/engine-spec.md b/docs/design/engine-spec.md index 6651f30f..d059426b 100644 --- a/docs/design/engine-spec.md +++ b/docs/design/engine-spec.md @@ -192,9 +192,9 @@ engine verb, though not a guard predicate.) (`threshold = { "fix-loop" = "any(severity >= high)" }` works because *the user's schema* declared the order — core never knows what a severity is). Aggregations beyond comparison are **action steps**. One is builtin and generic: `action = "aggregate"` -with `params = { field, method = median|max|min, hold_spread, output }` computes over -any ordered-enum payload field — median, spread-hold, and a recorded demotion trail -work for severities, priorities, or tiers alike. Cluster membership arrives in the +with `params = { field, method = median|max|min, hold_spread, output, route_at }` +computes over any ordered-enum payload field — median, spread-hold, and a recorded +demotion trail work for severities, priorities, or tiers alike. Cluster membership arrives in the payload itself: each element is one cluster, whose `field` value is either a scalar (a one-member cluster — the identity case) or an array of the cluster's member values; the builtin's input payload is the concatenated payloads of the step's @@ -205,7 +205,15 @@ that cluster's payload index (`-held@k#i`), together gating the routing st *(amended 2026-08-07, DKT-15: one step for the whole hold made approve/reject binary over the set, so an operator who wanted two clusters escalated and two accepted could not say so)*. The routing step waits for every one of them and routes once: per -`on_fail` if any was rejected, otherwise through the threshold. The +`on_fail` if any was rejected, otherwise through the threshold. An optional +`route_at = ""` names a routing floor in the field's declared order: only +clusters whose reduced value's position is at or above it are emitted to the output +payload — what the threshold evaluates and downstream `inputs` read — while the rest +are recorded, fully reduced, on the aggregate's own `action_results` row and never +enter the loop; a held cluster is never routed below the floor (the operator's +decision, not the untrusted computed value, decides it), an unknown `route_at` value +is a register-time refusal naming it, and with `route_at` absent the output is +byte-for-byte what it always was *(amended 2026-08-23, DKT-593)*. The aggregate's output payload — per-cluster value, members, held flag, `demoted_from`, `operator_resolved` — validates against the shipped `aggregate@1` schema, and `operator_resolved` is set per cluster, on the approved ones only. The @@ -411,6 +419,24 @@ registered. `--restore` reverses a retirement. There is deliberately **no delete verb**: old versions stay registered, which is what keeps lineage readable. +A registry is **per project** (`UNIQUE(project_id, name, version)`), and +`workflow register`, `workflow deprecate`, and `schema register` therefore +accept `--project ` and `--all-projects` *(amended 2026-08-24 — DKT-615)*. +Without either flag each verb writes to the project the working directory +resolves to, unchanged, and emits the row it always emitted. With either flag it +emits a **per-project report** instead — one outcome per target +(`registered` / `unchanged` / `deprecated` / `restored` / `already-binding` / +`already-deprecated` / `conflict` / `not-registered` / `invalid`) — and each +project's own idempotency and conflict rules are decided *there*: a CONFLICT in +one project neither cancels nor hides another's registration, and the sweep +runs to the end. The process exits with the failures' shared code when they +agree and GENERAL_ERROR when they do not, having already written the report. +`workflow register`'s **environment validation runs per target**, because +`vote_rule` and `payload` references resolve against the registry of the project +being written to — the same bytes can be valid in one project and name a schema +that does not exist in the next, and storing them there anyway would defer a +guaranteed activation failure. Register schemas store-wide first. + `[limits]` — optional map of executor *class* → `{ max = N, lease_ttl = "45m" }` (bare int = shorthand for `max`). When a run pins multiple workflows, the most restrictive limit per class wins; unset values fall back to `docket config` defaults. Classes also @@ -433,18 +459,21 @@ is exactly `"write" = { max = 1, lease_ttl = "45m", max_step_duration = "2h" }`. | `payload` | `schema@ver`, optional | payload validated at `complete`; threshold fields check against it at register time; required on `action = "aggregate"` steps *(amended 2026-08-03, DKT-25)* | | `voters`, `vote_rule` | [executor hints], proposal-config name | required on `type="vote"` steps — who casts, which existing Docket threshold config tallies | | `after` | [step names], **required** except the first step and `loop = true` steps (whose ordering comes from loop entry, §11.3) | intra-workflow predecessors; `[]` = root (implicit topology was a footgun) | -| `inputs` | [`"."` \| `".*"` \| `"issue.body"` \| `"issue.diff"`] | artifacts inlined into the context bundle, in order. `issue.diff` = the engine-computed VCS diff for the issue's scope, snapshotted and fingerprinted when its producing step completed (git in v1 — the one declared VCS coupling, §7) | +| `inputs` | [`"."` \| `".*"` \| `".vote-record"` \| `"issue.body"` \| `"issue.diff"` \| `"issue.linked.."`] | artifacts inlined into the context bundle, in order. `issue.diff` = the engine-computed VCS diff for the issue's scope, snapshotted and fingerprinted when its producing step completed (git in v1 — the one declared VCS coupling, §7). `.vote-record` = the named `type="vote"` step's recorded proposal — tally outcome, weighted score, and every cast with its rationale — engine-served from the existing vote machinery; the named step must be a vote step, and the `vote-record` kind is reserved from `emits` *(amended 2026-08-22, DKT-545)*. `issue.linked..` = a CROSS-ISSUE input: the latest recorded artifact of `` held by each issue this issue is linked to by `` (a relation type or its inverse form — `depends_on`, `dependency_of`, `blocks`, `blocked_by`, `relates_to`, `duplicates`, `duplicate_of`), resolved and pinned by artifact id at activation inside the fat transaction; activation fails loudly when the relation is missing or no linked issue holds the kind, so the binding is enforced rather than an issue-body citation. V11's produced-kind table deliberately does not apply — the producer is another issue's run — and the `issue.linked` name is reserved from step names as `issue.latest` is *(amended 2026-08-22, DKT-547)* | | `gates` | [trusted gate names \| `{name, source="fence:", pre=bool}`] | `pre = true` gates run at claim with results included in the context bundle (measure-then-judge steps); the rest run in order inside `complete` (§2, §4) | | `params` | opaque KV table | arguments to `action` steps (e.g. the builtin `aggregate`) | | `min_siblings` | int, default = all | fanout join quorum (§2 Fanout joins); the default is the plain join — quorum semantics (the `on_fail` routing at join) apply only when declared below the sibling count *(clarified 2026-08-03)* | -| `threshold` | table: routing → predicate (11.2) | routing computed over the step's recorded payloads | +| `threshold` | table: routing → predicate (11.2) | routing computed over the step's recorded payloads; on `type="vote"` steps, over the tally's cast set after an APPROVED tally (11.2) *(amended 2026-08-22, DKT-545)* | +| `pass_floor` | `{ field, at }`, optional; requires `payload`, and `at` must be a value of `field`'s declared order (V37/V37a) | exit bar on a `pass` routing: when the routing resolves to `pass` but the step's recorded payload holds an element whose `field` value sits at or above `at`'s position — and the element is neither `held` nor `operator_resolved` — the step parks `waiting-human` instead of exiting, naming `--as override-pass` and `--as fix-round` as the ways out. Both values are opaque tokens compared only by position, `route_at`'s discipline; declared nowhere, nothing changes *(amended 2026-08-26, DKT-870: RUN-58's reconcile routed `pass` with all 16 clusters open, six at the order's high position — "converged" in the ledger meaning "dispositioned")* | | `on_fail` | `"fix-loop"` \| `"waiting-human"` \| `"skip"` \| `"abandon-issue"`; default `"waiting-human"` | routing for gate failure / attempts exhausted; `type="human"` steps must declare it explicitly and `"waiting-human"` is invalid there — reject routes per `on_fail` (§2's reject-routing rule; amended 2026-08-03) | | `loop` | bool, default false | marks loop-body steps (11.3) | | `after_loop` | step name | re-entry target after a loop body completes | +| `serves` | [step names], only on `loop = true` steps | scopes the body to the named steps' `fix-loop` routings — its loop CLUSTER (11.3); omitted = serves every trigger. Entries must name steps that can route `fix-loop`, and every step that can must be served by at least one body *(amended 2026-08-22, DKT-544)* | | `max_attempts` | int, default engine config | per-instance retry budget | -| `max_fix_loops` | int, default engine config | loop-entry budget per issue | +| `max_fix_loops` | int, default engine config | loop-entry budget per issue — ONE counter over EVERY `fix-loop` routing source (threshold, `on_fail`, rejected vote/human gate, quorum miss), read off whichever non-cluster step declares it. Each admitted entry post-increments the counter to its own 1-indexed ordinal; an entry whose new count exceeds the bound is refused with the counter restored, so `= N` admits exactly N entries and parks the N+1th `waiting-human`. Only a `fix-round` grant (one per resolution, effective bound = declared + grants) admits more *(amended 2026-08-23, DKT-587)*. On a `serves`-scoped loop body it is instead that CLUSTER's round budget, checked independently under the issue-level ceiling — it never raises or lowers it *(amended 2026-08-22, DKT-544)* | +| `max_stalled_rounds` | int ≥ 0, default 0 (never fires); only on a step that can route `fix-loop` and records an artifact (V38) | non-convergence tolerance over THIS step's routed volume: a `fix-loop` entry after that many consecutive measured rounds in which the element count of the step's recorded payload never fell below the smallest count any earlier round recorded is refused in the non-convergence park's exact shape — counter restored, nothing instantiated, `waiting-human` naming `--as fix-round` as the way out, an authorized entry waived. "No improvement" means no new strict minimum, so volumes oscillating around a floor still park while a genuinely shrinking set never does *(amended 2026-08-26, DKT-870: RUN-51 held 8-12 clusters flat across TEN rounds and RUN-50 7-10 across six, both ended only by operator action — the plateau was the corpus's own non-convergence signal and nothing in the engine read it)* | | `expected_cost` | number ≥ 0, default 0 | budget-floor contribution per claim (§2) | -| `when` | predicate over issue `kind`/`labels` | step is `skipped` when false | +| `when` | predicate over issue `kind`/`labels` — clauses ` <==\|!=\|contains> ` or `labels contains-any (a, b, c)` / `labels contains_any [a, b, c]`, joined by `and` throughout or by `or` throughout | step is `skipped` when false. `or` holds when at least one clause does; a predicate MIXING `and` and `or` is a VALIDATION_ERROR (V22), because the grammar has no parentheses and therefore no reading of `a and b or c` to prefer — the mixed case is expressed as two steps, which is what the disjunction removed the need for in the common case *(amended 2026-08-22, DKT-548)*. `labels contains-any (…)` is the step-level spelling of the `labels_any` [match] clause and holds when the list intersects the issue's labels — a CLAUSE, not a connective, so "kind X and any of these labels" is one homogeneous-`and` predicate rather than a mix V22 would refuse. The list needs at least one element and its values carry no whitespace *(amended 2026-08-22, DKT-550)*. The operator is spelled `contains-any` or `contains_any` and its list is delimited by `(…)` or `[…]`; all four combinations are the same clause, and the delimiters must pair — `[a, b)` is a VALIDATION_ERROR. Both spellings were admitted rather than one because `contains-any (…)` is what registered definitions carry and `contains_any [a, b]` is how a list is written everywhere else in a workflow TOML, so refusing either would make an author's first correct guess an error *(amended 2026-09-01, DKT-1000)* | | `metadata` | opaque KV table | recorded on the step; delivered in the context bundle | ### 11.2 Threshold predicates @@ -469,6 +498,23 @@ whose registered schema declares `ordered_enum` (§2). Fields and literals are validated against the registered schema at `workflow register` time. Example (standard-change): `threshold = { "fix-loop" = "any(severity >= high)" }`. +**Vote-step thresholds** *(amended 2026-08-22, DKT-545)*: on a `type="vote"` step, +`threshold` is evaluated over the proposal's recorded **casts** — one element per +cast, addressable fields `vote` / `verdict` (aliases for the cast's verdict) and +`voter` — and only after an **APPROVED** tally. A rejected tally routes per +`on_fail`, exactly as before; a manually committed proposal (an operator setting +the final outcome by hand) skips the threshold. The routing vocabulary is +restricted to `"fix-loop"` / `"waiting-human"` / `"pass"` — step-name +interposition is not available on vote steps — and operators to equality, because +casts have no registered schema and ordered comparisons are defined only over +`ordered_enum` fields (all register-time rules: V36). First match routes, no +match ⇒ `"pass"`, and a step declaring no threshold behaves exactly as it always +did. Example (an investigation read-gate): +`threshold = { "fix-loop" = "count>=2(vote == approve-with-concerns)" }` sends an +approved-but-concerned tally into the same revise loop a rejection enters, +instead of the concerns evaporating; the loop body reads what the panel said +through `inputs = [".vote-record"]` (§11.1). + ### 11.3 Loop semantics (normative) Step instances are identified `name@k#i` — `k` = loop ordinal (0 at initial @@ -498,6 +544,24 @@ the highest existing ordinal ≤ the consumer's (mirroring input binding); re-instantiation never spans steps outside the `after_loop` chain. *(Clarified 2026-08-03, S3 stage review.)* +**Cluster scoping** *(amended 2026-08-22, DKT-544)*: a `loop = true` body may declare +`serves = [step names]`, scoping it to the named steps' `fix-loop` routings. The +TRIGGERING step — the one whose routing resolved to `fix-loop` — selects its +cluster: clauses (2)–(4) then apply to the serving bodies and to the downstream +chains of THOSE bodies' `after_loop` roots only (a body or `after_loop` declarer +without `serves` serves every trigger, so a workflow declaring no `serves` anywhere +has exactly one cluster and the original behavior, unchanged). The loop counter, +its ordinal sequence, the `max_fix_loops` ceiling read off non-cluster steps, the +non-convergence refusal, and `fix-round` grants all stay issue-level across every +cluster. A `max_fix_loops` declared on a `serves`-scoped body additionally bounds +that cluster's own rounds — counted as the distinct ordinals holding its scoped +bodies' instances (bodies serving several triggers are counted wherever they ran) — +and its refusal takes the ceiling's exact shape, waived once by the same +`fix-round` resolution. The `loop-entered` event data names the trigger alongside +the ordinal. Register-time rules: `serves` is valid only on `loop = true` steps, +every entry must name a step of the workflow that can route `fix-loop` (V35), and +every step that can route `fix-loop` must be served by at least one body (V17c). + Engine-enforced numbers live core-side, never in opaque pins: per-class lease TTLs and concurrency (`[limits]` / `docket config`), attempt caps (step fields / config defaults), the per-run budget cap (`docket run start --budget N`, config default), and diff --git a/docs/tdd/completion-metadata.md b/docs/tdd/completion-metadata.md index babd66ea..c41fec76 100644 --- a/docs/tdd/completion-metadata.md +++ b/docs/tdd/completion-metadata.md @@ -11,6 +11,12 @@ metadata"). Tracker unit: **DKT-68**. Precedent: the M2a bind-to-highest patch Spec of record is engine-spec.md; deviations become DKT amendment issues per docs/design/amendments.md, never silent changes. +**Amended 2026-08-23 (DKT-592): §1.7 adds the claim-side write.** A bag that +lands only at completion is a bag that failed and crashed steps never record, +so the dispatcher's half of a requested/resolved pair now lands AT CLAIM and +only the worker-reported half stays at completion. §1.1–§1.5 are unchanged for +the completion path; §2.D's "both are worker-reported" is corrected there. + **The defect is not a missing feature; it is a missing assignment.** Every other part of the path already exists and is correct: the flag parses (`internal/cli/step.go:310`), the option field is declared and documented as @@ -197,6 +203,78 @@ would lose exactly the diagnostic an operator wants. But that is DKT-69's call to make with its own acceptance criteria, and nothing here forecloses it: the merge function is indifferent, and the decision is one call site. +### 1.7 The claim-side write (DKT-592) — amendment, 2026-08-23 + +**The completion-time write is correct and incomplete.** A bag that lands only +at `complete` is a bag that a step which never completes never records — and a +step that failed or crashed is exactly the step whose dispatch facts an +operator most wants. Two runs measured the hole: 76 of 119 steps carried the +routing keys, then 55 of 83. The rollup this note exists to feed went blind at +precisely the rows that motivate reading it. + +The rule, and it is a **split by who knows the fact, not by which verb is +convenient**: + +- A fact the **dispatcher already knows when it hands the step out** is + recorded **at claim**. §2.D calls the motivating routing keys "both + *worker*-reported"; that was wrong about the requested half. What was asked + for is settled before the work starts — it is the dispatcher's own decision — + and holding it until the work comes back makes recording it conditional on + the work succeeding. +- A fact **only the returning worker knows** stays at **completion**. The + resolved half is genuinely worker-reported: nothing before the work runs can + say what actually served it. It does **not** move to claim, and §1.1–§1.5 are + unchanged for it. + +`ClaimOptions.Metadata` (`--metadata` on `step claim`) is the channel. +Mechanically it is the pieces this note already built, with no new ones: + +- **The same merge.** `mergeMetadata` over the step's stored bag, + last-write-wins per top-level key, shallow. A definition-side bag survives; a + re-claim overlays the dead attempt's; **the completion bag later overlays the + claim's**, which is what makes a normally-completing step carry both pairs. + The pure function §1.6 promised now has three callers and still knows nothing + about which verb called it. +- **The same writer**, `db.SetStepMetadataTx`, in the claim's **transaction A** + — with the CAS that awarded the claim, after the status write and before the + `step-claimed` event, mirroring §1.3's ordering rationale: the row reaches + its final shape for the transition before the transition is recorded. A crash + between them rolls back both. +- **The same cap and the same shape check**, `MetadataMaxBytes`, measured on + the raw input, **before `conn.Begin()`**. That placement carries more weight + here than on `complete` or `fail`: transaction A performs the lazy reap and + the CAS, so validation drifting inside it would put a malformed bag one + change away from consuming an attempt. Only the remedy text differs — a claim + has no artifact or payload channel of its own, so the message names its + completion's. + +**Survival is a property of what the other writers do not touch**, and it is +worth stating as such rather than assuming it: `ReapStepTx`, +`MarkStepAttemptFailedTx`, `MarkStepClaimReapedTx` and the status writers all +leave `metadata` alone. So the claim's bag survives a failure, a reap, an +abandonment, and the return to `pending` — with no cleanup path to audit. + +The claim's **own context bundle** reports the merged bag: the snapshot is +updated in place before `AssembleContext` runs, so a worker reading +`context.metadata` sees what it was dispatched with rather than the row's state +before its own claim. + +**No schema change**, again (§1.4). `currentSchemaVersion` is untouched. + +**One spec deviation, unamended and named here rather than silently applied.** +engine-spec.md §11.4 lists a `complete args` line and no `claim args` line, so +`step claim --metadata` is surface the spec does not yet describe. Per +docs/design/amendments.md the spec is not edited from here: this owes a DKT +amendment issue proposing §11.4 gain + +``` +claim args --owner NAME [--ttl D] [--metadata '{…}'] +``` + +alongside the existing `complete args`. The engine behavior is additive and +every existing claimant is byte-identical without the flag, which is why the +implementation does not wait on the amendment — but the amendment is owed. + ## 2. Alternatives considered **A. Separate `completion_metadata` column.** Rejected. It requires a v11 @@ -228,6 +306,13 @@ appears, `--metadata` bags can carry their own provenance keys opaquely, which is precisely what the KV bag is for. Filing it as a deferred question rather than building it is the §2 discipline. +**Corrected by §1.7 (DKT-592):** "both are *worker*-reported" was wrong about +the requested half — what was asked for is the DISPATCHER's own decision, +settled before the work starts. The rejection of a provenance column stands +(the bag still carries its own provenance keys opaquely, and no schema change +was needed), but the sentence that justified it no longer describes the write +path: the requested half now lands at claim, the resolved half at completion. + **E. Refuse when the worker overwrites a definition key.** Rejected. It sounds protective and is actually core having an opinion about which keys matter. The reference instance's `model_resolved` overwriting a declared `model_requested` @@ -340,7 +425,41 @@ A new `scripts/qa/` section following the existing helper conventions 5. Genericity: `grep` the section's own fixtures for the banned words → zero hits, since the gate scans tests too. -### 4.5 Gates that must stay green +### 4.5 Go — the claim-side write (§1.7) + +In `internal/engine/metadata_test.go`, beside the completion and fail suites, +with the same neutral-key discipline: `tier_requested` / `desk_resolved` mirror +the SHAPE of a routing pair — one fact known at dispatch, one known only on +return — without naming any instance's vocabulary. + +- **The regression test:** claim with `--metadata`, assert `steps.metadata` + carries it with **no completion anywhere in the test**. +- **The failure case, which is the whole point:** claim with a bag, `fail` with + no bag of its own → the claim's keys are still on the row. +- **The crash case, which is not the failure case:** nobody calls `fail` at + all. The lease lapses, the next claim reaps it, and the bag survives the reap. +- **Both pairs on a normal completion:** claim's two keys plus completion's two + keys, all four present — the completion merge must not clobber the claim's. +- Definition-side bag present → merged, not clobbered; dispatcher's value wins + on a shared key. +- No `--metadata` → definition bag byte-identical, no row_version bump. +- The claim's own context bundle reports the merged bag. +- A claimed-then-FAILED step's keys appear in the R7 rollup — the read surface + the defect was observed in, with no read-side change. +- Refusals (invalid JSON, array, scalar, number, null) → `VALIDATION_ERROR`, + `row_version` unmoved, **status still `pending` and attempt still 0**, and + the step still claimable. The attempt assertion is the claim-specific half: + this transaction reaps and CASes, so "the refusal wrote nothing" has to mean + "it cost no attempt" too. +- The cap's own test: inclusive boundary in both directions, both numbers in + the message plus this verb's remedy, measured on raw input. +- **Source-position check**, `TestClaimValidatesMetadataBeforeTheTransactionOpens`: + the size and shape calls appear before `conn.Begin()` in + `claimStepWithGates`'s AST — the same argument + `TestFailValidatesMetadataBeforeTheTransactionOpens` makes, because a + rollback hides the difference from any runtime assertion. + +### 4.6 Gates that must stay green Full `scripts/qa.sh` plus `scripts/qa/genericity.sh`. `go build ./...` and `go test ./...`. No migration test changes, because there is no migration — diff --git a/docs/tdd/engine-spine.md b/docs/tdd/engine-spine.md index 11582c68..1d919703 100644 --- a/docs/tdd/engine-spine.md +++ b/docs/tdd/engine-spine.md @@ -287,7 +287,7 @@ offending field. Every row is a test case (§4.6). The table is the phase's cont | V19 | `max_attempts` ≥ 1; `max_fix_loops` ≥ 0; `expected_cost` ≥ 0 | §11.1 | | V20 | `threshold` keys ∈ {`fix-loop`, `waiting-human`, `pass`} ∪ step names in this workflow | §11.2 | | V21 | `threshold` predicate parses as `agg(field op literal)`, `agg ∈ {any, all, count>=n}`, `op ∈ {==, !=, >=, >, <=, <}` | §11.2 | -| V22 | `when` parses as a predicate over `kind`/`labels` only | §11.1 `when`; engine-core §4 "conditions (predicates over issue kind/labels only)" | +| V22 | `when` parses as a predicate over `kind`/`labels` only, its clauses joined by `and` throughout or `or` throughout — a mix of the two is refused (DKT-548). A clause is ` <==\|!=\|contains> ` or the set form `labels contains-any (a, b, c)`, whose list must be non-empty, comma-separated, and free of whitespace inside its values; `contains-any` is `labels`-only (DKT-550). The set operator is equivalently spelled `contains_any` and its list equivalently delimited `[a, b, c]`, with the delimiters required to pair (DKT-1000) | §11.1 `when`; engine-core §4 "conditions (predicates over issue kind/labels only)" | | V23 | `class` defaults to the `executor` value when unset | §11.1 `class`: "default = executor value" | | V24 | `[limits]` values: `max` ≥ 1, `lease_ttl`/`max_step_duration` parse as durations | §11.1 `[limits]` | | V25 | `payload` matches `name@version` shape (**shape only** at S3 — §6.14) | §11.1 `payload` | @@ -407,7 +407,7 @@ one classification feeds the lint, expansion, and the engine's readiness latch | Verb | Flags | Effect | |---|---|---| | `docket workflow register ` | `--json[=v2]` | parse + validate + lint; insert `name@version`; idempotent on identical bytes; `CONFLICT` on differing bytes at an existing `name@version` | -| `docket workflow list` | `--name`, `--limit`, `--json[=v2]` | registered workflows; a `Collection` (reliability-delta §4.1) so v2 renders `{items,total,truncated}` | +| `docket workflow list` | `--name`, `--limit`, `--orphans`, `--json[=v2]` | registered workflows; a `Collection` (reliability-delta §4.1) so v2 renders `{items,total,truncated}`; `--orphans` narrows to registrations whose NAME no file in any instance-config root declares any more (DKT-609), stamping each row with an `origin` verdict and refusing outright when there is no root to scan | | `docket workflow show [@]` | `--source`, `--json[=v2]` | the parsed definition; `@version` omitted ⇒ highest registered; `--source` emits the stored TOML verbatim | | `docket workflow init` | `--template NAME`, `--dir PATH`, `--force` | writes template files into `.docket/config/` (default), refusing to overwrite without `--force` | @@ -612,6 +612,60 @@ mid-run edit immunity; freezing the scheduler would ignore a correction that exi precisely to prevent a collision. Both are stated so neither is "fixed" into the other later. +**No AUTOMATIC path refreshes the snapshot** (DKT-741). It is written once, at +activation stage 4, and nothing rewrites it for the life of the run — not a claim, not +a fix-round, not a re-instantiation, and not an `issue edit`. So the consumers that +read it — the packet's `context.issue.scope` (§6.6) and the recorded `issue.diff` +scope (§6.7.1 D1) — cannot drift apart from each other or from what the run was +activated on. Re-snapshotting at claim time instead would break exactly that: two +steps of one run would render two different declared scopes and record their diffs +over two different path sets, and a packet would stop being reproducible from the +ledger. `docket issue edit --scope` therefore reaches the live column and nothing +else, which is correct and is also a trap: + +| what the operator wants | what `issue edit --scope` does | +|---|---| +| stop a collision the scheduler is about to allow | works, immediately — R4 reads live | +| widen an authorized scope so a live step's packet says so | **does nothing**; the packet renders the frozen snapshot | + +The second row has **two** dispositions, and which one is right depends on whether the +run's premise changed or one declaration was corrected. + +**Where the premise changed**, take the issue out of the run and re-plan it — +`docket run abandon RUN-N --issue DKT-M --reason "scope widened"`, then plan it into a +new run, whose activation snapshots the widened scope afresh. It is expensive on +purpose: a mid-run scope widen can invalidate the premise every step of that issue +already executed under, and re-planning is what re-establishes it. + +**Where the premise is intact**, `docket run refresh-scope RUN-N --issue DKT-M +--reason R` copies `issues.scope_globs` into that one run-issue's snapshot and +rewrites nothing else in it (DKT-869, `RefreshIssueScopeInRun`). DKT-741 had ruled out +any refresh verb; RUN-52 (VPL-434) then charged twice for that ruling on an intact +premise — the panel rejected work as out of scope, the operator agreed and widened it, +the already-minted `fix@2` step still rendered the old scope, and the issue was +abandoned mid-loop. The freeze keeps its default and gains an explicit exception whose +four properties are what keep it from being a hole in §9 item 5: + +1. **It carries no scope of its own.** There is no `--scope` on it; `issue create|edit + --scope` stays the sole writer of the column it copies, so the refresh cannot make + real a scope that was not declared through the one gate widening has always had. A + refresh with no widen behind it is **refused** (CONFLICT), not silently no-op'd. +2. **No step straddles it.** It refuses while any of the issue's steps is `claimed`, + `running`, or `gated`, and while a dispatch is open — the repin quiescence rule + (DKT-408) applied to the other frozen premise. `pending` and `waiting-human` are the + refreshable states. +3. **It rewrites no history.** Terminal steps keep their artifacts and the scope their + diffs were computed over; only the remaining steps' renders move. +4. **The discontinuity is in the ledger.** One `issue-scope-refreshed` event (actor + `human`) carries the old scope, the new scope, the instances reached, and the + operator's reason — so two steps of one run declaring two different scopes is a + dated, attributable fact rather than drift a reader must infer. + +`issue edit --scope` **warns**, naming the run, the frozen scope, the count of live +steps, and **both** verbs, whenever the edit changes the scope of an issue that still +has non-terminal steps in a non-terminal run (`ScopeEditFrozenForActiveRuns`). It +reports rather than refuses, because the write is real for the scheduling half. + ## 5.2 Run status and the minimal subset engine-core §1.1: `planning → active ⇄ waiting-human → done | abandoned`. All five @@ -792,6 +846,14 @@ including `unless_labels` beating `labels_any`; absent clauses matching anything **Go unit tests** (`internal/engine/activate_test.go`): - exactly-one-match: zero matches and two matches each `VALIDATION_ERROR`, each naming the issue **and** every candidate workflow (asserted by substring). +- **orphan annotation** (DKT-609, `internal/engine/dkt609_test.go`): each named + candidate whose NAME no file in any instance-config root declares any more is + marked `(no source on disk — orphaned registration, deprecation candidate)`, + and the refusal carries the remedy. It DECORATES the candidate set and never + changes it — an orphaned registration still binds, because a registration is + a row and not a file. With no root to scan the verdict is `unchecked` and the + message is byte-identical to the pre-DKT-609 one: "nothing was checked" must + never render as "nothing is orphaned". - **bind-to-highest** (§11.1 as amended 2026-08-05, DKT-40): the candidate set is the **highest registered version of each name**, so exactly-one-match applies across NAMES. `TestBindingUsesHighestVersionOfEachName` is DKT-8's M2a wedge as diff --git a/docs/tdd/gates-trust.md b/docs/tdd/gates-trust.md index fea02d4b..378bb0a4 100644 --- a/docs/tdd/gates-trust.md +++ b/docs/tdd/gates-trust.md @@ -561,6 +561,33 @@ It is deliberately NOT the pre-existing `gate_results.stub`, which marks a row migrated from an S3 `gate_trail` — that is a fact about which era produced the row, and one column carrying both would answer neither question. +### AMENDMENT (DKT-607) — a stub records its reason + +A stub entry may carry `stub_reason`, set by `docket trust add --stub +--stub-reason ""`. It records the DECISION behind the +placeholder: why no real check exists yet and which issue tracks replacing it +(e.g. `"no scanner selected yet; removal tracked by DKT-607"`). + +**The problem it solves.** DKT-265 made hollow green visible; it did not make it +EXPLAINED. Two tribunal seats on DKT-V196 independently rediscovered the same +corpus stubs (`secret-scan`, `sdet-abuse`) because the decision that they remain +stubs lived only in tribunal transcripts. The project's stub-gate policy +requires every stub to have a removal-tracking issue; this field is where that +reference becomes discoverable from the surfaces an operator actually reads. + +**Where it surfaces.** The activation gate preflight prints it under the stub's +own line (and carries it as `stub_reason` in the JSON row); a stub with NO +recorded reason gets a remedy line naming `--stub-reason`. `trust list` renders +it inside the `stub(no-real-check: …)` marker, and it rides §3.6's event beside +`stub`. + +**Constraints.** It only makes sense alongside `stub = true`: a reason on a +non-stub entry is refused at parse and at add, the closed direction. It is +OPTIONAL on a stub — every pre-DKT-607 stub entry has none and keeps loading +with an empty reason. Changing or erasing it on a re-add is a `CONFLICT`, since +the reason is the documented decision and a silent rewrite would swap one +decision for another under a re-approval. + ## 3.6 Trust changes are event-logged (T9) `trust add` and `trust rm` write a **`trust-added` / `trust-removed` event** into @@ -897,6 +924,36 @@ the next environment variable anyone invents. | `CI` | `1` | the near-universal convention for "non-interactive"; it makes tools skip prompts and progress spinners without docket having to know each tool | | `DOCKET_GATE` | the gate name | so a check can behave differently under docket if its author wants; opaque to core | | `DOCKET_REPO` | the repo root | the same value as `Dir`, for tools that need it in an env | +| `DOCKET_GATE_BASE` | the step's base commit sha, **worktree-recorded completion gates only** | so a range-shaped check can scan exactly the step's committed change — `DOCKET_GATE_BASE..HEAD` of the tree it runs in — see below *(added 2026-09-01, DKT-992)* | + +**`DOCKET_GATE_BASE` — the step's committed range** *(DKT-992)*. Executors +commit **before** `step record`, so at gate time a worktree-recorded step's +tree is clean: a working-tree-only scan measures zero lines however large the +change (RUN-66's secret-scan passed 8/8 write steps that way), and a gate +guessing `git diff HEAD~1` is wrong for every multi-commit step. The engine +already knows the step's base — the worktree's **fork point**, the same +resolution the diff stage's `runDiffBase` applies — so completion gates of a +`--worktree`-recorded step export it: + +- **Worktree-recorded step**: `DOCKET_GATE_BASE` names the commit the worktree + was created from. `git diff $DOCKET_GATE_BASE..HEAD` in the gate's own cwd + (the worktree, per DKT-9) is exactly the step's committed change — the same + range the recorded `issue.diff` describes. +- **Non-worktree step**: the variable is **unset** — that is the documented + pick between the two admissible encodings (unset, or equal to `HEAD`). The + shared checkout has no fork point, the run's pinned commit is not this + step's base (sibling work lands between them, DKT-42's over-attribution), + and a live `HEAD` read is a value docket cannot vouch for as a range + endpoint. Absence — never an invented sha — is the encoding, the same + convention as `DOCKET_SCOPE`. +- The variable is also unset when the fork point cannot be resolved, and on + the pre-claim path (a pre-gate measures the tree under review, not a + recorded completion; after integration sweeps a worktree, no honest base + survives to export). +- **Fail closed on absence**: a range-shaped gate that finds the variable + absent while the tree is clean has nothing it can honestly scan, and should + fail rather than pass having measured nothing — "we couldn't check, so + carry on" is what makes a control decorative (N3). **Excluded, by name, in addition to being absent from the allowlist:** diff --git a/docs/tdd/payloads-thresholds.md b/docs/tdd/payloads-thresholds.md index 7ba6d3cb..f9363fe9 100644 --- a/docs/tdd/payloads-thresholds.md +++ b/docs/tdd/payloads-thresholds.md @@ -863,6 +863,7 @@ unchanged. | `method` | `median` \| `max` \| `min` | yes | the reduction | | `hold_spread` | integer ≥ 0, default 0 | no | hold when spread **≥** this; `0` never holds | | `output` | string | yes (already V11's, §4.3.1) | the produced artifact kind | +| `route_at` | string, a value of `field`'s declared order | no | routing floor *(amended 2026-08-23, DKT-593 — see the amendment below)*: a cluster whose reduced value's position is **≥** its position is emitted to the output payload; the rest go to the record. Absent ⇒ every cluster emits, byte-for-byte the pre-`route_at` output | New register-time rules, each a `VALIDATION_ERROR` naming workflow, step, and param, each a test case: @@ -870,7 +871,8 @@ param, each a test case: | # | Rule | Argument | |---|---|---| | V27 | a step's `name` may not end in **`-held`**, and `action` may not name a builtin other than `aggregate` | the first reserves the materialized identity (§7.7) so a definition cannot collide with one; the second turns "my trusted `aggregate` command never runs" into a register-time sentence | -| V28 | `action = "aggregate"` requires `field`, `method ∈ {median,max,min}`, `output`; `hold_spread` an integer ≥ 0 if present; **no other keys** | the discipline every V-rule follows. A typo'd `method = "medain"` is otherwise discovered hours into a run, on a step whose inputs are already spent | +| V28 | `action = "aggregate"` requires `field`, `method ∈ {median,max,min}`, `output`; `hold_spread` an integer ≥ 0 if present; `route_at` a non-empty string if present *(DKT-593)*; **no other keys** | the discipline every V-rule follows. A typo'd `method = "medain"` is otherwise discovered hours into a run, on a step whose inputs are already spent | +| V28a | `route_at`, when declared, must name a value of `params.field`'s declared order *(DKT-593)* | the floor is a **position** in that order, and a value with no position has no floor to name — G4's discipline asked at register time. Schema-aware, so it lives in `ValidateSchemas` beside V29 rather than in the pure-bytes V28 | | V29 | an `aggregate` step must declare `payload = name@version`, and that schema must declare `params.field` as **`ordered_enum`** | median, max, and min are all defined **only** over an order. An aggregate without a declared order is a step that can never compute, and §11.2's own restriction is the same restriction | | V30 | the declared schema must **accept an aggregate-shaped document**: a synthetic probe built from the schema's own declared enum values (§7.6) is validated against it at register time | the output must satisfy *both* the instance schema and `aggregate@1` (§7.6). An instance schema with `"additionalProperties": false` makes that conjunction unsatisfiable — and the failure would otherwise land at the end of a review fan-out, hours in. The probe is deterministic and invents nothing: every value in it comes from the schema being checked, and it carries the aggregate output keys (`members`, `held`) plus one carried-through extra key so the conjunction it tests is the real one (review F3) | @@ -881,6 +883,34 @@ magic (nothing in the grammar says an action reads its predecessor's payload), it is unstable under `inputs` edits, and it hides the one declaration an author most needs to see. +### AMENDMENT (DKT-593) — `route_at`: a routing floor + +RUN-43 spent 71.6% of its output tokens on fix rounds that closed roughly as +many clusters as they opened, because every reconciled cluster — whatever its +reduced value — entered the loop, and `contracts/fix.md` rightly forbids the +FIXER from filtering ("the reconciled set is the work"). The missing piece was +never the fixer's to add: which clusters are worth a round is a ROUTING +question, and routing is core's. + +`route_at = ""` names a floor in `field`'s declared order. Like every +value the builtin touches it is an **opaque token compared only by position** +(§4.3 I3) — a severity floor for a severity order, a ripeness floor for a +ripeness order, and core cannot tell the difference. + +| # | Clause | +|---|---| +| R1 | A cluster whose **reduced value's** position is **≥** the floor's position is emitted to the output payload — the wire the threshold evaluates and every downstream `inputs` reader consumes. The comparison is over the reduced value, exactly as the threshold's would be: a cluster whose members reach `high` but whose median is `low` is a `low` cluster to both | +| R2 | A cluster below the floor goes **to the record**: fully reduced — value, members, demotion trail — into the aggregate's own `action_results` row (`output` column, the audit channel every attempt already writes), and counted in the artifact body. It never enters the artifact payload, so the threshold and the loop never see it. Routed, not erased: the input artifacts remain immutable and addressable, and the reduction's trail is attributed in the row | +| R3 | A **held cluster is never routed below the floor.** Its computed value is exactly the value the hold refuses to trust — the members disagree, and the operator resolving it may accept a different value (`--value`) — so routing it out by that value would spend the decision the hold exists to ask. It is emitted, it gates the step (§7.7), and the floor's opinion waits for the operator's | +| R4 | `Held` indices address the **emitted** payload — the payload the artifact records and `-held@k#i` resolves against (H2a) — which coincides with input positions whenever nothing was routed below | +| R5 | **Absent ⇒ byte-for-byte today's output.** No key, no floor, nothing recorded, G2's identity property untouched — the same no-cliff discipline `hold_spread = 0` follows | +| R6 | Every cluster below the floor is legal: the emitted payload is then the **empty array**, a valid `aggregate@1` document over which any threshold predicate finds nothing — the round is simply not entered | +| R7 | Validation is split as V28/V28a: presence and type are pure bytes (V28); membership in the declared order needs the schema and lives in `ValidateSchemas` (V28a). At run time — a definition can arrive through a restored database — the same two refusals fire in `ParseAggregateParams` and `Aggregate` respectively, the second naming the value per G4 | + +This is **core routing, not fixer filtering**: the set the fixer receives has +already been routed, so `contracts/fix.md`'s prohibition stands unchanged — +there is nothing left in the reconciled set to filter. + ## 7.2 The input shape: what a "cluster" is §2 promises the output is "per-cluster value, members, held flag", and diff --git a/docs/tdd/reliability-delta.md b/docs/tdd/reliability-delta.md index 38369a95..50601585 100644 --- a/docs/tdd/reliability-delta.md +++ b/docs/tdd/reliability-delta.md @@ -592,6 +592,75 @@ the counters are authoritative only for claims that ended after v23. The wire fields are `omitempty`, so every row with no counted outcome serializes byte-identically to v22's rendering. +### AMENDMENT — the span extends to v24 (DKT-546, 2026-08-22) + +**What changed.** v24 adds ONE table, `gate_override_grants`: one operator +ruling that a gate's failure signature — gate name + exit code + reason +classification — is environmental for the remainder of ONE run. A grant is +minted by `step resolve --as override-pass --batch` (one row per failed +completion gate of the parked step, in the resolution's own transaction), and +spent by the routing stage: a later step of the same run whose EVERY failing +gate matches a grant routes the same generic `pass` the operator's own +override-pass records, instead of parking. `covered_steps` counts the spends, +bumped in the routing transaction; both edges are event-logged +(`gate-override-granted` / `step-batch-overridden`, attributed human / +threshold), so the feed walks from every auto-pass back to the person. The +grant dies with its run — the `run_id` FK is the whole scope rule, and a new +run re-asks. A cover blocked by an interposed threshold target (DKT-470's +shape) parks as before, with the block named. + +**What it fixes.** Refit mining of 25 runs / 34 issues found the dominant +operator toil is environmental gate parks — build failed 30/46 recorded +verdicts, tests 26/45, self-hygiene 18/34 — virtually every one +operator-overridden as a sandbox artifact, not a code defect, and each park +resolved individually: in RUN-42 the operator's own resolution was +"override-pass each as it parks", the same ruling re-made per step. No +mechanism let one ruling cover subsequent identical failures in the same run. + +**Why the ratified arithmetic is untouched.** Like v11–v23, v24 is an +amendment, not a stage: one additive table, `CREATE TABLE IF NOT EXISTS` +throughout so the migration is idempotent and re-runnable, and a rewind guard +that probes the TABLE (the v7/v8 form, since v24 adds no column). It +BACK-FILLS NOTHING — no operator granted a batch override before the verb for +granting one existed — and it is dormant: a run that never records a grant +reads byte-identically to v23 on every verb. + +### AMENDMENT — the span extends to v25 (DKT-742, 2026-08-25) + +**What changed.** v25 adds ONE table, `stale_target_waivers`: one operator +ruling that a specific stale-target warning — one (step instance, target sha) +pair — has been adjudicated for the remainder of ONE run. A waiver is minted +by `dispatch waive-target` (one row per named step instance, all for one +target sha, in one transaction, each row event-logged as +`stale-target-waived`, attributed human), and consulted READ-ONLY by the +DKT-193/424/451 advisory judge at `dispatch open`/`verify` and at held +resolutions: a would-be warning whose (instance, target) pair matches a +waiver — the sha compared as a case-insensitive prefix of at least 7 hex +characters, because the advisory renders it at 12 — is dropped from the +answer. There is deliberately no "spent" counter and no per-application event: +the advisory is recomputed by `dispatch verify`, which writes nothing by +contract, and a suppressed warning changes no step's state. The waiver dies +with its run — the `run_id` FK is the whole scope rule, and a new run +re-warns. + +**What it fixes.** The stale-target advisory had no memory: RUN-52 fired the +IDENTICAL adjudicated warning four times across DISPATCH-295/297/301 (the +shared HEAD moves at every integration, so the pair recurs under a different +rendered reason each time), each firing costing an investigation and the +first an operator gate, until the operator issued a standing waiver that +lived only in session memory — where the engine could not see it. A different +target sha on the same row, or the same sha on an unnamed row, is a different +question and still warns, which is what keeps a new divergence from riding an +old ruling. + +**Why the ratified arithmetic is untouched.** Like v11–v24, v25 is an +amendment, not a stage: one additive table, `CREATE TABLE IF NOT EXISTS` +throughout so the migration is idempotent and re-runnable, and a rewind guard +that probes the TABLE (the v24 form, since v25 adds no column). It BACK-FILLS +NOTHING — no operator waived a stale-target warning before the verb for +waiving one existed — and it is dormant: a run that never records a waiver +reads byte-identically to v24 on every verb. + ### 2.1 The never-mutate rule engine-spec.md §3 requires v4 DBs open unchanged and existing verbs stay diff --git a/internal/cli/dispatch.go b/internal/cli/dispatch.go index 62b210ac..8a4a8786 100644 --- a/internal/cli/dispatch.go +++ b/internal/cli/dispatch.go @@ -2,6 +2,7 @@ package cli import ( "encoding/json" + "errors" "fmt" "io" "os" @@ -133,6 +134,23 @@ func warnStaleTargets(w *output.Writer, stale []engine.StaleTarget) { } } +// warnUnclaimedTargets renders DKT-993's advisory, one line per back-filled +// step no worker ever claimed. +// +// The rows ARE recorded — the engine's contract is warn-and-record, not refuse +// — so the line says so explicitly. A conductor that reads "recorded" and meant +// it can move on; one that did not now has the step id its journal mis-joined +// onto, which is the whole gap RUN-66 left: the bad rows landed in silence and +// surfaced only as unexplained spend in `run report`. +func warnUnclaimedTargets(w *output.Writer, unclaimed []engine.UnclaimedTarget) { + for _, u := range unclaimed { + w.Warn("%s (%s) is %s and was never claimed — no worker ever held it, "+ + "so it reported nothing to reconstruct. The row is recorded; check "+ + "the journal that produced it before trusting the spend", + u.Step, exec.Render(u.Instance), u.Status) + } +} + // warnPinDrift renders DKT-408's advisory, one line per unsound pin, plus the // recovery verb — steps reading these refs will refuse at claim/render, and // the conductor should learn that before spending a wave, not from it. @@ -207,24 +225,7 @@ func runDispatchVerify(cmd *cobra.Command, w *output.Writer) error { return runErr(err) } if mismatch != nil { - // The differing BYTES, both sides, escaped on their way to a terminal: - // a manifest row carries an executor hint and a metadata bag, which are - // workflow-author strings and therefore untrusted text (gates-trust - // §5.7 R11). - // - // The per-row summary rides ABOVE the refusal (DKT-243). The refusal - // itself still names the first offending row and shows its bytes — - // unchanged — but a dispatch where several steps moved mid-flight used - // to report one of them and hide the rest, costing a manual per-step - // confirm round before a `close` that reconciles the same state - // without complaint. - return cmdErr(fmt.Errorf( - "%s does not match its current rendering (manifest row %d)\n%s"+ - " stored: %s\n"+ - " recomputed: %s", - result.Dispatch, mismatch.Position, renderRowVerdicts(result.Rows), - renderRowOrAbsent(mismatch.Stored), renderRowOrAbsent(mismatch.Computed)), - output.ErrConflict) + return mismatchConflict("", result, mismatch) } warnStaleTargets(w, result.StaleTargets) @@ -238,6 +239,35 @@ func runDispatchVerify(cmd *cobra.Command, w *output.Writer) error { return nil } +// mismatchConflict is P9's refusal, rendered once for both the verbs that can +// raise it (`dispatch verify` and `dispatch close --backfill-from`). +// +// The differing BYTES, both sides, escaped on their way to a terminal: a +// manifest row carries an executor hint and a metadata bag, which are +// workflow-author strings and therefore untrusted text (gates-trust §5.7 R11). +// +// The per-row summary rides ABOVE the refusal (DKT-243). The refusal itself +// still names the first offending row and shows its bytes — unchanged — but a +// dispatch where several steps moved mid-flight used to report one of them and +// hide the rest, costing a manual per-step confirm round before a `close` that +// reconciles the same state without complaint. +// +// `prefix` is empty for `dispatch verify`, so that verb's refusal is +// byte-identical to what it has always printed; the reconcile path passes its +// stage name (DKT-580) so an operator learns WHICH of three stages refused +// without the diagnostic itself being reworded. +func mismatchConflict( + prefix string, result *engine.VerifyResult, mismatch *engine.RowMismatch, +) error { + return cmdErr(fmt.Errorf( + "%s%s does not match its current rendering (manifest row %d)\n%s"+ + " stored: %s\n"+ + " recomputed: %s", + prefix, result.Dispatch, mismatch.Position, renderRowVerdicts(result.Rows), + renderRowOrAbsent(mismatch.Stored), renderRowOrAbsent(mismatch.Computed)), + output.ErrConflict) +} + // renderRowVerdicts is DKT-243's per-row block: every stored row that is not // plainly matched, named with what happened to it. // @@ -335,7 +365,27 @@ needed a dispatch that was not open. Without the flag, closing a manifest that is not open is still a conflict. The acceptance clears the discrepancy for next and dispatch open, not only for -the dispatch row — those two verbs refuse on exactly the same conditions.`, +the dispatch row — those two verbs refuse on exactly the same conditions. + +--backfill-from PATH closes a wave in ONE invocation: it back-fills the usage in +PATH, verifies the manifest, and closes it, refusing at whichever stage fails +with that stage named. PATH is the same JSON array ` + "`backfill-usage --from-json`" + ` +reads — [{"step","unit","quantity"}, ...] — and "-" reads stdin. + + docket dispatch close --run RUN-44 --backfill-from usage.json + +It is exactly the three verbs, in the one order they can run in, calling the +same engine functions they call. Each stage is all-or-nothing on its own: a +failed back-fill runs no verify and no close, a failed verify runs no close, and +nothing is ever half-closed. --source and --on-duplicate mean what they mean on +` + "`backfill-usage`" + ` and apply to the back-fill stage, as does its +never-claimed-target warning: a conductor that traded three invocations for one +loses no advisory. + +An open dispatch is REQUIRED with --backfill-from, because the verify stage +needs a manifest to compare. --accept-missing-usage still applies, to the close +stage; the standalone verbs are unchanged and remain the way to run the stages +one at a time.`, RunE: func(cmd *cobra.Command, args []string) error { return runDispatchClose(cmd, getWriter(cmd)) }, @@ -350,6 +400,14 @@ func runDispatchClose(cmd *cobra.Command, w *output.Writer) error { } accept, _ := cmd.Flags().GetBool("accept-missing-usage") + // DKT-580: with `--backfill-from` this verb is the whole wave-close + // pipeline. Without it, every line below is what it has always been — + // criterion 2 is that the plain close did not change, so the reconcile is a + // branch taken on the flag rather than a rewrite of the default path. + if from, _ := cmd.Flags().GetString("backfill-from"); from != "" { + return runDispatchReconcile(cmd, w, runID, from, accept) + } + outcome, err := engine.NewEngine().CloseDispatch(conn, runID, accept, model.NowMS()) if err != nil { return runErr(err) @@ -363,6 +421,104 @@ func runDispatchClose(cmd *cobra.Command, w *output.Writer) error { return nil } +// runDispatchReconcile is `dispatch close --backfill-from` (DKT-580): one +// invocation for back-fill, verify, and close. +// +// The conductor's wave close was four typed commands with a temp file between +// two of them, repeated once per wave. Nothing in it is a decision — the order +// is forced and the arguments are the same run — so every retype was a chance +// to point at the wrong journal or skip the verify, and a skipped verify is how +// a manifest closes over a ready set that moved underneath it. +func runDispatchReconcile( + cmd *cobra.Command, w *output.Writer, runID int, from string, accept bool, +) error { + rows, err := reconcileRows(cmd, from) + if err != nil { + return err + } + source, _ := cmd.Flags().GetString("source") + onDuplicate, _ := cmd.Flags().GetString("on-duplicate") + + outcome, err := engine.NewEngine().ReconcileDispatch( + getDB(cmd), runID, rows, source, onDuplicate, accept, model.NowMS()) + if err != nil { + return reconcileErr(err) + } + + // The same advisories the standalone verbs emit, from the same stages: a + // conductor that switched to one invocation must not lose a warning it + // would have seen across three. + for _, sk := range outcome.Backfill.Skipped { + w.Warn("skipped %s (%s) attempt %d: %q usage already recorded", + sk.Step, exec.Render(sk.Instance), sk.Attempt, sk.Unit) + } + warnUnclaimedTargets(w, outcome.Backfill.Unclaimed) + warnStaleTargets(w, outcome.Verify.StaleTargets) + + var message string + if !w.JSONMode { + message = renderReconcileOutcome(outcome) + } + w.Success(outcome, message) + return nil +} + +// reconcileErr maps a staged failure onto the taxonomy. +// +// A verify-stage MISMATCH is the one refusal whose diagnostic cannot be +// rebuilt from the error text — P9's evidence is the differing bytes and +// DKT-243's verdict block, both carried on the result — so it is rendered +// through the same function `dispatch verify` renders it with, with the stage +// named in front. Every other stage failure keeps its own message and its own +// exit code: StageError unwraps to the stage's error, so runErr's lookup finds +// the code the standalone verb would have exited with. +func reconcileErr(err error) error { + var stage *engine.StageError + if errors.As(err, &stage) && stage.Mismatch != nil { + return mismatchConflict( + stage.Stage+" stage: ", stage.Verify, stage.Mismatch) + } + return runErr(err) +} + +// reconcileRows reads `--backfill-from`. +// +// It is `backfill-usage --from-json`'s reader, deliberately: the file a +// conductor pipes into one verb is the file it pipes into the other, and a +// second parser would be a second set of rules for the same bytes. +// +// A DIRECTORY IS REFUSED BY NAME. DKT-580 describes the flag as taking "a wave +// transcript dir or usage json", and core cannot read the former: a transcript +// tree is the harness's own format, and core inventing a reader for it would be +// core holding an opinion about how a relay records its spend — the same +// opinion `--source` and `unit` exist to avoid holding. The refusal names the +// tool that turns one into the other rather than failing as an unreadable file. +func reconcileRows(cmd *cobra.Command, path string) ([]engine.BackfillRow, error) { + if path != "-" { + if info, err := os.Stat(path); err == nil && info.IsDir() { + return nil, cmdErr(fmt.Errorf( + "--backfill-from %s is a directory; core reads usage rows, not "+ + "transcript trees. Pass the usage JSON a journal tool "+ + `(wave-usage) produces — a JSON array of `+ + `{"step","unit","quantity"} — or "-" to pipe it on stdin`, + path), output.ErrValidation) + } + } + return backfillRowsFromJSON(cmd, path) +} + +// renderReconcileOutcome is the human view: one line per stage, in the order +// they ran, each reusing the wording its standalone verb prints. +func renderReconcileOutcome(o *engine.ReconcileOutcome) string { + var b strings.Builder + fmt.Fprintf(&b, "Back-filled %d usage row(s) across %d step(s) as %q\n", + o.Backfill.Written, o.Backfill.Steps, o.Backfill.Source) + fmt.Fprintf(&b, "%s verified: the manifest matches the current ready set (%s)\n", + o.Verify.Dispatch, verdictTally(o.Verify.Rows)) + b.WriteString(renderCloseOutcome(o.Close)) + return b.String() +} + // ---- backfill-usage -------------------------------------------------------- var dispatchBackfillUsageCmd = &cobra.Command{ @@ -399,6 +555,16 @@ skipped. Cross-wave duplicates are structural — a gate probed in wave N and seated in wave N+1 emits usage in both journals — and refusing the batch for them meant hand-filtering rows before every re-run (DKT-241). +A step NO WORKER EVER CLAIMED is back-filled anyway, and WARNED ABOUT by name +and status on stderr — plus an ` + "`unclaimed`" + ` array in ` + "`--json`" + `. Trusting the +relay stays the default: core cannot know that a step which never claimed cost +nothing, so it records the row. What it will not do is stay silent. A wave +journal whose join heuristic mis-attributes a claimant's usage lands it on a +` + "`pending`" + ` or swept-` + "`superseded`" + ` step, and ` + "`run report`" + ` then shows spend +on work that never ran with nothing to say where it came from. The warning names +the step so the journal that produced the row can be checked; a step that ran +and simply could not report — the case this verb exists for — never triggers it. + ` + "`docket run report`" + ` lists what is already recorded, per step, under ` + "`step_usage`" + `: that is the verb to check before re-running. @@ -420,6 +586,12 @@ type backfillOutcome struct { // Skipped names the rows --on-duplicate=skip passed over. `omitempty`, so // the default refusing mode's payload is unchanged. Skipped []engine.SkippedRow `json:"skipped,omitempty"` + // Unclaimed names the target steps no worker ever claimed (DKT-993). It + // rides the payload because `Warn` is silent in JSON mode and a conductor + // runs with `--json`: an advisory only a human could see is the same + // silence this field exists to end. `omitempty`, so an ordinary back-fill's + // payload is byte-identical to what it was. + Unclaimed []engine.UnclaimedTarget `json:"unclaimed,omitempty"` } func runDispatchBackfillUsage(cmd *cobra.Command, w *output.Writer) error { @@ -449,10 +621,12 @@ func runDispatchBackfillUsage(cmd *cobra.Command, w *output.Writer) error { w.Warn("skipped %s (%s) attempt %d: %q usage already recorded", sk.Step, exec.Render(sk.Instance), sk.Attempt, sk.Unit) } + warnUnclaimedTargets(w, result.Unclaimed) outcome := &backfillOutcome{ Run: model.FormatRunID(runID), Rows: result.Written, Steps: result.Steps, Source: result.Source, Skipped: result.Skipped, + Unclaimed: result.Unclaimed, } var message string @@ -616,6 +790,77 @@ func renderCloseOutcome(o *engine.CloseOutcome) string { return b.String() } +// ---- waive-target ---------------------------------------------------------- + +var dispatchWaiveTargetCmd = &cobra.Command{ + Use: "waive-target", + Short: "Stop an adjudicated stale-target warning from re-firing", + Long: `Record that a stale-target warning was investigated and ruled acceptable. + +The stale-target advisory (` + "`dispatch open`/`verify`" + `) re-computes on every +invocation and, without this verb, has no memory: a warning an operator already +investigated re-fires unchanged at every subsequent open and verify of the same +(step, target) pair, costing an investigation each time. + +A waiver names one target sha and one or more step instances — exactly the +strings the warning itself printed: + + docket dispatch waive-target --run RUN-52 --target 12f5006a \ + --step "review@1#0" --step "review@1#1" \ + --note "fix integrated as b68bf23; divergence is the later format pass" + +The sha may be the 12-character prefix the warning renders (7 hex characters +minimum); matching is case-insensitive prefix against the recorded target. + +THE WARNING MACHINERY STILL RUNS. A waiver suppresses only the exact pair it +names: the same step warning about a DIFFERENT sha, or the same sha on a step +no waiver names, warns exactly as before — a new divergence never rides an old +ruling. Waivers are RUN-SCOPED and die with their run; each one is recorded as +a ` + "`stale-target-waived`" + ` event, so the feed shows what standing precedent was +minted and why.`, + RunE: func(cmd *cobra.Command, args []string) error { + return runDispatchWaiveTarget(cmd, getWriter(cmd)) + }, +} + +// waiveTargetOutcome is what the verb reports: the run and the waivers minted. +type waiveTargetOutcome struct { + Run string `json:"run"` + Waivers []engine.WaivedTarget `json:"waivers"` +} + +func runDispatchWaiveTarget(cmd *cobra.Command, w *output.Writer) error { + conn := getDB(cmd) + + runID, err := dispatchRunID(cmd) + if err != nil { + return err + } + steps, _ := cmd.Flags().GetStringSlice("step") + target, _ := cmd.Flags().GetString("target") + note, _ := cmd.Flags().GetString("note") + + waivers, err := engine.NewEngine().WaiveStaleTargets( + conn, runID, steps, target, note, model.NowMS()) + if err != nil { + return runErr(err) + } + + outcome := &waiveTargetOutcome{Run: model.FormatRunID(runID), Waivers: waivers} + var message string + if !w.JSONMode { + var b strings.Builder + fmt.Fprintf(&b, "Waived the stale-target warning for %.12s on %d step(s):", + target, len(waivers)) + for _, wv := range waivers { + fmt.Fprintf(&b, "\n %s (waiver %d)", wv.Instance, wv.ID) + } + message = b.String() + } + w.Success(outcome, message) + return nil +} + // ---- shared flags ---------------------------------------------------------- // dispatchRunID resolves the `--run` every dispatch verb requires. @@ -649,7 +894,7 @@ func ackSeqs(cmd *cobra.Command) ([]int64, error) { func init() { for _, c := range []*cobra.Command{ dispatchOpenCmd, dispatchVerifyCmd, dispatchCloseCmd, dispatchAbandonCmd, - dispatchBackfillUsageCmd, + dispatchBackfillUsageCmd, dispatchWaiveTargetCmd, } { c.Flags().String("run", "", "The run whose dispatch this is (required)") _ = c.MarkFlagRequired("run") @@ -665,6 +910,20 @@ func init() { dispatchCloseCmd.Flags().Bool("accept-missing-usage", false, "Close despite missing-usage discrepancies, recording the acceptance") + // DKT-580's one-verb wave close. `--source` and `--on-duplicate` are the + // SAME flags `backfill-usage` declares, with the same defaults, because + // they configure the same stage; they are inert without `--backfill-from`. + dispatchCloseCmd.Flags().String("backfill-from", "", + `Back-fill this JSON array of {"step","unit","quantity"}, verify, `+ + `then close, in one invocation; "-" reads stdin`) + dispatchCloseCmd.Flags().String("on-duplicate", "refuse", + "With --backfill-from: what to do with a row whose (step, attempt, "+ + "unit) is already recorded: refuse the batch, or skip that row "+ + "and report it") + dispatchCloseCmd.Flags().String("source", "", + "With --backfill-from: who measured it; recorded on every row "+ + "(default \"backfilled\")") + dispatchAbandonCmd.Flags().String("reason", "", "Why the manifest is being given up on; recorded in the event") @@ -684,7 +943,19 @@ func init() { dispatchBackfillUsageCmd.Flags().String("source", "", "Who measured it; recorded on every row (default \"backfilled\")") + // DKT-742's waiver: the step instances and target sha exactly as the + // stale-target warning printed them. + dispatchWaiveTargetCmd.Flags().StringSlice("step", nil, + "Step instance the warning named, e.g. \"review@1#0\" (repeatable)") + dispatchWaiveTargetCmd.Flags().String("target", "", + "The warned target sha — full, or the >=7-char prefix the warning renders") + dispatchWaiveTargetCmd.Flags().String("note", "", + "Why the warning is adjudicated acceptable; recorded on every waiver") + _ = dispatchWaiveTargetCmd.MarkFlagRequired("step") + _ = dispatchWaiveTargetCmd.MarkFlagRequired("target") + dispatchCmd.AddCommand(dispatchOpenCmd, dispatchVerifyCmd, - dispatchCloseCmd, dispatchAbandonCmd, dispatchBackfillUsageCmd) + dispatchCloseCmd, dispatchAbandonCmd, dispatchBackfillUsageCmd, + dispatchWaiveTargetCmd) rootCmd.AddCommand(dispatchCmd) } diff --git a/internal/cli/dispatch_reconcile_test.go b/internal/cli/dispatch_reconcile_test.go new file mode 100644 index 00000000..66de4be9 --- /dev/null +++ b/internal/cli/dispatch_reconcile_test.go @@ -0,0 +1,300 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// `dispatch close --backfill-from` — DKT-580's one-verb wave reconcile at the +// CLI edge: the flag's parsing, its per-stage refusals, and the promise that +// the flagless close is untouched. + +// dispatchCloseCmdWithDB builds a `dispatch close` command carrying every flag +// the real one declares, so a test exercises the same GetString/GetBool +// lookups runDispatchClose makes. +func dispatchCloseCmdWithDB(conn *sql.DB, runRef string) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("run", runRef, "") + cmd.Flags().Bool("accept-missing-usage", false, "") + cmd.Flags().String("backfill-from", "", "") + cmd.Flags().String("on-duplicate", "refuse", "") + cmd.Flags().String("source", "", "") + return cmd +} + +// waveWithUnreportedUsage is the state every wave close starts from: a +// dispatch is open and its one claimable step finished WITHOUT reporting +// usage, which is what an executor that cannot observe its own spend does. +// It returns the manifest's step id. +func waveWithUnreportedUsage(t *testing.T, conn *sql.DB, runID int) int { + t.Helper() + + manifest, err := engine.NewEngine().OpenDispatch(conn, runID, 0, nil, model.NowMS()) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(manifest.Rows) == 0 { + t.Fatal("premise: the manifest must have a row to complete") + } + stepID, err := model.ParseStepID(manifest.Rows[0].Step) + testsupport.Must(t, err, "parsing %q: %v", manifest.Rows[0].Step, err) + + claim, err := engine.ClaimStep(conn, stepID, engine.ClaimOptions{ + Owner: "worker", NowMS: model.NowMS(), + }) + testsupport.Must(t, err, "claim: %v", err) + err = engine.NewEngine().CompleteStep(conn, stepID, engine.CompleteOptions{ + Token: claim.Token, Artifact: []byte("done"), NowMS: model.NowMS(), + }) + testsupport.Must(t, err, "complete: %v", err) + return stepID +} + +// usageJSON writes the batch a wave journal tool emits. +func usageJSON(t *testing.T, rows string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "usage.json") + testsupport.Must(t, os.WriteFile(path, []byte(rows), 0o600), + "writing the usage batch: %v", nil) + return path +} + +// TestDispatchCloseBackfillFromRunsAllThreeStages is criterion 1 at the CLI: +// one invocation, three stages, and a payload that reports each of them. +func TestDispatchCloseBackfillFromRunsAllThreeStages(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + stepID := waveWithUnreportedUsage(t, conn, runID) + + // The premise: the flagless close REFUSES here. Without that, the + // back-fill stage running would be unobservable. + bare := dispatchCloseCmdWithDB(conn, runRef) + wBare, _ := bufWriter(true) + if err := runDispatchClose(bare, wBare); err == nil { + t.Fatal("premise: a plain close must refuse over the missing usage") + } + + path := usageJSON(t, fmt.Sprintf( + `[{"step":%q,"unit":"tokens","quantity":48211}]`, + model.FormatStepID(stepID))) + + cmd := dispatchCloseCmdWithDB(conn, runRef) + testsupport.Must(t, cmd.Flags().Set("backfill-from", path), + "setting --backfill-from: %v", nil) + testsupport.Must(t, cmd.Flags().Set("source", "wave-journal:wf-7"), + "setting --source: %v", nil) + + w, buf := bufWriter(true) + err := runDispatchClose(cmd, w) + testsupport.Must(t, err, "dispatch close --backfill-from: %v", err) + + var env struct { + Data engine.ReconcileOutcome `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Backfill == nil || env.Data.Backfill.Written != 1 { + t.Fatalf("the payload's backfill stage is %+v, want 1 row written:\n%s", + env.Data.Backfill, buf.String()) + } + if env.Data.Backfill.Source != "wave-journal:wf-7" { + t.Errorf("source = %q, want the one passed", env.Data.Backfill.Source) + } + if env.Data.Verify == nil || env.Data.Verify.Dispatch == "" { + t.Errorf("the payload's verify stage names no dispatch:\n%s", buf.String()) + } + if env.Data.Close == nil || env.Data.Close.Status == "" { + t.Fatalf("the payload's close stage is %+v:\n%s", env.Data.Close, buf.String()) + } +} + +// TestDispatchCloseUnchangedWithoutTheFlag is criterion 2: the flagless close +// still emits the plain CloseOutcome, not the reconcile's three-stage shape. +// A conductor parsing `.data.dispatch` must not have to learn a new payload +// because a flag it does not pass exists. +func TestDispatchCloseUnchangedWithoutTheFlag(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + stepID := waveWithUnreportedUsage(t, conn, runID) + + // Back-fill through the STANDALONE verb, exactly as before this change. + _, err := engine.NewEngine().BackfillUsage(conn, runID, []engine.BackfillRow{ + {Step: stepID, Unit: "tokens", Quantity: 7}, + }, "", "", model.NowMS()) + testsupport.Must(t, err, "backfill-usage: %v", err) + + cmd := dispatchCloseCmdWithDB(conn, runRef) + w, buf := bufWriter(true) + testsupport.Must(t, runDispatchClose(cmd, w), "dispatch close: %v", nil) + + var env struct { + Data map[string]any `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if _, ok := env.Data["dispatch"]; !ok { + t.Errorf("the plain close payload lost its `dispatch` key:\n%s", buf.String()) + } + for _, key := range []string{"backfill", "verify", "close"} { + if _, ok := env.Data[key]; ok { + t.Errorf("the plain close payload grew a %q key; the reconcile "+ + "shape must ride on the flag only:\n%s", key, buf.String()) + } + } +} + +// TestDispatchCloseBackfillFromNamesTheFailedStage is the per-stage half of +// criterion 1: a back-fill that refuses names ITS OWN stage, exits on the +// stage's own code, and leaves the dispatch open — no partial close. +func TestDispatchCloseBackfillFromNamesTheFailedStage(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + waveWithUnreportedUsage(t, conn, runID) + + // A step that does not exist: the back-fill resolves every step before it + // writes anything, so this refuses NOT_FOUND with nothing written. + path := usageJSON(t, `[{"step":"STEP-999999","unit":"tokens","quantity":1}]`) + + cmd := dispatchCloseCmdWithDB(conn, runRef) + testsupport.Must(t, cmd.Flags().Set("backfill-from", path), + "setting --backfill-from: %v", nil) + + w, _ := bufWriter(true) + err := runDispatchClose(cmd, w) + if err == nil { + t.Fatal("the close succeeded over a back-fill naming a missing step") + } + if !strings.Contains(err.Error(), engine.StageBackfill+" stage") { + t.Errorf("the refusal %q does not name the stage that failed", err) + } + var cerr *CmdError + if !errors.As(err, &cerr) { + t.Fatalf("error %v carries no exit code", err) + } + if cerr.Code != output.ErrNotFound { + t.Errorf("code = %q, want %q — a staged failure keeps the stage's own "+ + "taxonomy code", cerr.Code, output.ErrNotFound) + } + + // The dispatch is untouched: a later `dispatch close` still has something + // to close, which it would not if the failure had half-closed the run. + var status string + testsupport.Must(t, conn.QueryRow( + `SELECT status FROM dispatches WHERE run_id = ? ORDER BY id DESC LIMIT 1`, + runID).Scan(&status), "reading the dispatch row: %v", nil) + if status != "open" { + t.Errorf("the dispatch is %q after a failed back-fill stage, want open", status) + } +} + +// TestDispatchCloseBackfillFromRefusesADirectory pins the diagnostic for the +// other half of DKT-580's flag description. Core reads usage rows, not +// transcript trees, and the refusal must say so and name the way across rather +// than failing as an unreadable file. +func TestDispatchCloseBackfillFromRefusesADirectory(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + waveWithUnreportedUsage(t, conn, runID) + + cmd := dispatchCloseCmdWithDB(conn, runRef) + testsupport.Must(t, cmd.Flags().Set("backfill-from", t.TempDir()), + "setting --backfill-from: %v", nil) + + w, _ := bufWriter(true) + err := runDispatchClose(cmd, w) + if err == nil { + t.Fatal("a directory was accepted as a usage batch") + } + if !strings.Contains(err.Error(), "is a directory") || + !strings.Contains(err.Error(), "wave-usage") { + t.Errorf("the refusal %q neither names the mistake nor the way out", err) + } + var cerr *CmdError + if !errors.As(err, &cerr) || cerr.Code != output.ErrValidation { + t.Errorf("code = %v, want %q", cerr, output.ErrValidation) + } +} + +// TestDispatchCloseDeclaresTheReconcileFlags walks the REAL cobra command, not +// a test-built stand-in: every other test here constructs its own flag set, so +// without this one a flag could be missing from `init()` and the suite would +// still be green while the shipped binary had no `--backfill-from`. +// +// It also pins criterion 2 at the surface: `backfill-usage` keeps every flag it +// had, and `close` gained exactly three. +func TestDispatchCloseDeclaresTheReconcileFlags(t *testing.T) { + for _, name := range []string{"backfill-from", "on-duplicate", "source", + "accept-missing-usage", "run"} { + if dispatchCloseCmd.Flags().Lookup(name) == nil { + t.Errorf("`dispatch close` declares no --%s", name) + } + } + // The standalone verb is untouched. + for _, name := range []string{"step", "unit", "quantity", "from-json", + "on-duplicate", "source", "run"} { + if dispatchBackfillUsageCmd.Flags().Lookup(name) == nil { + t.Errorf("`dispatch backfill-usage` lost --%s", name) + } + } + // And `verify` did not quietly acquire the reconcile's flags: the stages + // stay separately invocable, they do not merge. + if dispatchVerifyCmd.Flags().Lookup("backfill-from") != nil { + t.Error("`dispatch verify` declares --backfill-from; the flag belongs " + + "to close alone") + } + if got := dispatchCloseCmd.Flags().Lookup("on-duplicate").DefValue; got != "refuse" { + t.Errorf("close's --on-duplicate defaults to %q, want %q — it must "+ + "mean what it means on backfill-usage", got, "refuse") + } +} + +// TestDispatchCloseBackfillFromReadsStdin pins the "-" form, which is what a +// conductor piping a journal tool's output uses — the temp file between two +// commands is one of the drift surfaces DKT-580 is about. +func TestDispatchCloseBackfillFromReadsStdin(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + stepID := waveWithUnreportedUsage(t, conn, runID) + + cmd := dispatchCloseCmdWithDB(conn, runRef) + testsupport.Must(t, cmd.Flags().Set("backfill-from", "-"), + "setting --backfill-from: %v", nil) + cmd.SetIn(strings.NewReader(fmt.Sprintf( + `[{"step":%q,"unit":"tokens","quantity":900}]`, + model.FormatStepID(stepID)))) + + w, buf := bufWriter(true) + testsupport.Must(t, runDispatchClose(cmd, w), + "dispatch close --backfill-from -: %v", nil) + + var env struct { + Data engine.ReconcileOutcome `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Backfill == nil || env.Data.Backfill.Written != 1 { + t.Fatalf("stdin batch wrote %+v, want 1 row:\n%s", + env.Data.Backfill, buf.String()) + } + if env.Data.Close == nil { + t.Fatalf("the close stage did not run:\n%s", buf.String()) + } +} diff --git a/internal/cli/dispatch_test.go b/internal/cli/dispatch_test.go index 3cc0c3a8..03175370 100644 --- a/internal/cli/dispatch_test.go +++ b/internal/cli/dispatch_test.go @@ -1,6 +1,7 @@ package cli import ( + "bytes" "database/sql" "encoding/json" "strings" @@ -111,3 +112,148 @@ func TestDispatchVerifyCLIConflictRendersRowsAndExitCode(t *testing.T) { } } } + +// ---- DKT-993: the never-claimed back-fill target --------------------------- + +// backfillCmdWithDB builds a `dispatch backfill-usage` command carrying every +// flag the real one declares, so the test exercises the same flag lookups +// runDispatchBackfillUsage makes. +func backfillCmdWithDB(conn *sql.DB, runRef string) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("run", runRef, "") + cmd.Flags().StringSlice("step", nil, "") + cmd.Flags().StringSlice("unit", nil, "") + cmd.Flags().Float64Slice("quantity", nil, "") + cmd.Flags().String("from-json", "", "") + cmd.Flags().String("on-duplicate", "refuse", "") + cmd.Flags().String("source", "", "") + return cmd +} + +// stderrWriter is bufWriter with the warning channel visible: DKT-993's whole +// subject is what reaches stderr, and bufWriter discards it. +func stderrWriter(jsonMode bool) (*output.Writer, *bytes.Buffer, *bytes.Buffer) { + stdout, stderr := &bytes.Buffer{}, &bytes.Buffer{} + return &output.Writer{JSONMode: jsonMode, Stdout: stdout, Stderr: stderr}, + stdout, stderr +} + +// pendingStepID returns a step of the run that nothing has ever claimed — +// `second@0` in the CLI fixture, the shape RUN-66's STEP-3146 had. +func pendingStepID(t *testing.T, conn *sql.DB, runID int) int { + t.Helper() + var id int + err := conn.QueryRow( + `SELECT id FROM steps WHERE run_id = ? AND status = 'pending' AND attempt = 0 + ORDER BY id LIMIT 1`, runID).Scan(&id) + testsupport.Must(t, err, "finding a never-claimed step: %v", err) + return id +} + +// TestBackfillUsageCLIWarnsOnANeverClaimedStep is DKT-993's acceptance at the +// edge the operator actually reads: the warning is PRINTED, it names the step +// and its status, and the row is still recorded. +func TestBackfillUsageCLIWarnsOnANeverClaimedStep(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + stepID := pendingStepID(t, conn, runID) + + cmd := backfillCmdWithDB(conn, model.FormatRunID(runID)) + testsupport.Must(t, cmd.Flags().Set("step", model.FormatStepID(stepID)), + "setting --step: %v", nil) + testsupport.Must(t, cmd.Flags().Set("unit", "tokens"), "setting --unit: %v", nil) + testsupport.Must(t, cmd.Flags().Set("quantity", "8724"), + "setting --quantity: %v", nil) + + w, stdout, stderr := stderrWriter(false) + testsupport.Must(t, runDispatchBackfillUsage(cmd, w), + "backfill-usage: %v", nil) + + warning := stderr.String() + for _, want := range []string{ + "Warning", model.FormatStepID(stepID), "pending", "never claimed", + } { + if !strings.Contains(warning, want) { + t.Errorf("stderr %q does not mention %q", warning, want) + } + } + + // Warn-and-record: the verb still reports the row it wrote. + if !strings.Contains(stdout.String(), "1 usage row") { + t.Errorf("stdout %q does not report the recorded row; the contract is "+ + "warn-and-record, not reject", stdout.String()) + } +} + +// TestBackfillUsageCLIJSONCarriesUnclaimed is the other half of the channel +// rule: `Warn` is suppressed in JSON mode, and a conductor runs with `--json` +// — an advisory only a human could see is the same silence DKT-993 ends. +func TestBackfillUsageCLIJSONCarriesUnclaimed(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + stepID := pendingStepID(t, conn, runID) + + cmd := backfillCmdWithDB(conn, model.FormatRunID(runID)) + testsupport.Must(t, cmd.Flags().Set("step", model.FormatStepID(stepID)), + "setting --step: %v", nil) + testsupport.Must(t, cmd.Flags().Set("unit", "tokens"), "setting --unit: %v", nil) + testsupport.Must(t, cmd.Flags().Set("quantity", "18"), + "setting --quantity: %v", nil) + + w, stdout, _ := stderrWriter(true) + testsupport.Must(t, runDispatchBackfillUsage(cmd, w), + "backfill-usage: %v", nil) + + var env struct { + Data struct { + Rows int `json:"rows"` + Unclaimed []engine.UnclaimedTarget `json:"unclaimed"` + } `json:"data"` + } + if err := json.Unmarshal(stdout.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, stdout.String()) + } + if env.Data.Rows != 1 { + t.Errorf("rows = %d, want 1 — the row is recorded", env.Data.Rows) + } + if len(env.Data.Unclaimed) != 1 || + env.Data.Unclaimed[0].Step != model.FormatStepID(stepID) || + env.Data.Unclaimed[0].Status != "pending" { + t.Fatalf("`unclaimed` = %+v, want one entry naming %s as pending:\n%s", + env.Data.Unclaimed, model.FormatStepID(stepID), stdout.String()) + } +} + +// TestBackfillUsageCLIQuietForAClaimedStep is the "existing valid back-fills +// unchanged" half at the CLI: no warning, and a payload with no `unclaimed` +// key at all — a conductor parsing this shape must not have to learn a new one. +func TestBackfillUsageCLIQuietForAClaimedStep(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + stepID := waveWithUnreportedUsage(t, conn, runID) + + cmd := backfillCmdWithDB(conn, model.FormatRunID(runID)) + testsupport.Must(t, cmd.Flags().Set("step", model.FormatStepID(stepID)), + "setting --step: %v", nil) + testsupport.Must(t, cmd.Flags().Set("unit", "tokens"), "setting --unit: %v", nil) + testsupport.Must(t, cmd.Flags().Set("quantity", "48211"), + "setting --quantity: %v", nil) + + w, stdout, stderr := stderrWriter(true) + testsupport.Must(t, runDispatchBackfillUsage(cmd, w), + "backfill-usage: %v", nil) + + if stderr.Len() != 0 { + t.Errorf("a claimed step's back-fill warned: %q", stderr.String()) + } + var env struct { + Data map[string]any `json:"data"` + } + if err := json.Unmarshal(stdout.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, stdout.String()) + } + if _, ok := env.Data["unclaimed"]; ok { + t.Errorf("an ordinary back-fill's payload grew an `unclaimed` key:\n%s", + stdout.String()) + } +} diff --git a/internal/cli/dispatch_waive_test.go b/internal/cli/dispatch_waive_test.go new file mode 100644 index 00000000..cd9ce668 --- /dev/null +++ b/internal/cli/dispatch_waive_test.go @@ -0,0 +1,102 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// DKT-742 at the CLI boundary: `dispatch waive-target` records the waivers, +// reports them on both channels, and refuses garbage with the flag named. +// WHICH warnings a waiver then suppresses is the engine's to prove +// (internal/engine/stale_waiver_test.go); what this file asserts is the verb. + +func waiveTargetCmdWithDB(conn *sql.DB, runRef, target, note string, steps []string) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("run", runRef, "") + cmd.Flags().StringSlice("step", steps, "") + cmd.Flags().String("target", target, "") + cmd.Flags().String("note", note, "") + return cmd +} + +func TestDispatchWaiveTargetCLIRecordsAndReports(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + runRef := model.FormatRunID(runID) + + w, buf := bufWriter(true) + cmd := waiveTargetCmdWithDB(conn, runRef, "cafe1234cafe", + "adjudicated: the divergence is the later format pass", + []string{"review@0#0", "review@0#1"}) + err := runDispatchWaiveTarget(cmd, w) + testsupport.Must(t, err, "dispatch waive-target: %v", err) + + var env struct { + Data struct { + Run string `json:"run"` + Waivers []engine.WaivedTarget `json:"waivers"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Run != runRef || len(env.Data.Waivers) != 2 { + t.Fatalf("payload = %+v, want two waivers on %s", env.Data, runRef) + } + for i, instance := range []string{"review@0#0", "review@0#1"} { + wv := env.Data.Waivers[i] + if wv.Instance != instance || wv.Target != "cafe1234cafe" || wv.ID == 0 { + t.Errorf("waiver %d = %+v, want %s on cafe1234cafe with a real id", + i, wv, instance) + } + } + + // The rows really landed, run-scoped. + var count int + err = conn.QueryRow( + `SELECT COUNT(*) FROM stale_target_waivers WHERE run_id = ?`, runID, + ).Scan(&count) + testsupport.Must(t, err, "counting waivers: %v", err) + if count != 2 { + t.Errorf("stale_target_waivers rows = %d, want 2", count) + } +} + +func TestDispatchWaiveTargetCLIHumanMessageNamesTheSteps(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + + w, buf := bufWriter(false) + cmd := waiveTargetCmdWithDB(conn, model.FormatRunID(runID), + "cafe1234cafe", "", []string{"review@0#0"}) + err := runDispatchWaiveTarget(cmd, w) + testsupport.Must(t, err, "dispatch waive-target: %v", err) + + out := buf.String() + if !strings.Contains(out, "cafe1234cafe") || !strings.Contains(out, "review@0#0") { + t.Errorf("the human message does not name the target and step:\n%s", out) + } +} + +func TestDispatchWaiveTargetCLIRefusesANonHexTarget(t *testing.T) { + conn := newTestDB(t) + runID := activatedDispatchRunForCLI(t, conn) + + w, _ := bufWriter(true) + cmd := waiveTargetCmdWithDB(conn, model.FormatRunID(runID), + "not-a-sha", "", []string{"review@0#0"}) + err := runDispatchWaiveTarget(cmd, w) + if err == nil { + t.Fatal("a non-hex target was accepted") + } + if !strings.Contains(err.Error(), "hex") { + t.Errorf("the refusal %q does not say what a target must look like", err) + } +} diff --git a/internal/cli/events.go b/internal/cli/events.go index e9ef70f6..c430dc2d 100644 --- a/internal/cli/events.go +++ b/internal/cli/events.go @@ -62,11 +62,18 @@ transition's own opaque payload, carried verbatim. Without --run the feed is repo-wide, which is the only place events that belong to no run — trust grants — are visible. +--run names ONE run and is answered wherever that run lives: a run in another +project returns that run's events rather than an empty page, matching +` + "`run report`" + ` and ` + "`step show`" + `. A run that does not exist is refused. + --tail N returns the NEWEST N events instead of the oldest N, for the mid-incident question "what just happened". They still arrive oldest-first, so the last seq is still the cursor to store. --tail and --since are mutually exclusive: one jumps to the end of the feed, the other walks it forward. +With --follow, --tail N is the STARTING CURSOR: the stream opens with the newest +N and then follows live, instead of replaying the whole retained history first. + Human mode escapes stored strings on their way to the terminal; --json carries the raw bytes, because the consumer there is a program.`, Args: cobra.NoArgs, @@ -100,6 +107,13 @@ var _ output.Collection = eventsListResult{} // eventsProjectScope resolves the feed's project scope (v12): the invoking // project by default, the whole store under --all-projects. Store-level // events — trust changes — appear either way; see EventQuery.ProjectID. +// +// IT IS NOT THE WHOLE STORY UNDER `--run` (DKT-583). ListEvents IGNORES this +// scope when the query names a run, because a run belongs to one project and +// the run filter is therefore already the narrower scope — scoping again could +// only answer a neighbour's run with a successful EMPTY feed. The decision +// lives in the engine so `--follow`, which shares this helper, cannot drift +// away from `list`. func eventsProjectScope(cmd *cobra.Command) int { if all, _ := cmd.Flags().GetBool("all-projects"); all { return 0 @@ -116,6 +130,32 @@ func (p eventsListPayload) MarshalJSON() ([]byte, error) { return json.Marshal(p.events) } +// eventsTailFlag reads and validates `--tail` for BOTH entry points. +// +// It is shared rather than duplicated because `--follow` starts its stream at +// the tail (DKT-752): if the two paths validated the flag separately, one of +// them would eventually accept a combination the other refuses, and the pair +// that matters most — `--tail` with `--since` — is precisely the one whose +// meaning is undefined. +func eventsTailFlag(cmd *cobra.Command) (int, error) { + tail, _ := cmd.Flags().GetInt("tail") + if tail < 0 { + return 0, cmdErr( + fmt.Errorf("--tail must not be negative: %d", tail), output.ErrValidation) + } + // `--tail` answers "what just happened" and `--since` answers "what have I + // not seen". Combining them reads as a cursor but silently skips whatever + // falls between the cursor and the tail window, so it is refused rather + // than served with a plausible-looking short page. + if tail > 0 && cmd.Flags().Changed("since") { + return 0, cmdErr( + fmt.Errorf("--tail and --since are mutually exclusive: "+ + "--since walks the feed forward, --tail jumps to its end"), + output.ErrValidation) + } + return tail, nil +} + func runEventsList(cmd *cobra.Command, w *output.Writer) error { conn := getDB(cmd) @@ -132,20 +172,9 @@ func runEventsList(cmd *cobra.Command, w *output.Writer) error { limit = eventsDefaultLimit } - tail, _ := cmd.Flags().GetInt("tail") - if tail < 0 { - return cmdErr( - fmt.Errorf("--tail must not be negative: %d", tail), output.ErrValidation) - } - // `--tail` answers "what just happened" and `--since` answers "what have I - // not seen". Combining them reads as a cursor but silently skips whatever - // falls between the cursor and the tail window, so it is refused rather - // than served with a plausible-looking short page. - if tail > 0 && cmd.Flags().Changed("since") { - return cmdErr( - fmt.Errorf("--tail and --since are mutually exclusive: "+ - "--since walks the feed forward, --tail jumps to its end"), - output.ErrValidation) + tail, err := eventsTailFlag(cmd) + if err != nil { + return err } runRef, _ := cmd.Flags().GetString("run") @@ -288,7 +317,8 @@ func eventDetail(data json.RawMessage) string { func init() { eventsListCmd.Flags().Int64( "since", 0, "Return events with seq strictly greater than this cursor") - eventsListCmd.Flags().String("run", "", "Filter to one run (RUN-N)") + eventsListCmd.Flags().String( + "run", "", "Filter to one run (RUN-N), in whichever project owns it") eventsListCmd.Flags().Bool( "all-projects", false, "Show every project's events instead of this project's (store-level events show either way)") @@ -296,7 +326,9 @@ func init() { // have I not seen", which to reach the END of a long feed means paging // through all of it first. The rows still arrive oldest-first. eventsListCmd.Flags().Int( - "tail", 0, "Return the newest N events (still oldest-first); excludes --since") + "tail", 0, + "Return the newest N events (still oldest-first); with --follow, start the stream "+ + "there; excludes --since") eventsListCmd.Flags().Int( "limit", 0, fmt.Sprintf("Maximum events to return (default %d)", eventsDefaultLimit)) diff --git a/internal/cli/events_follow.go b/internal/cli/events_follow.go index a4cb6f9c..06721791 100644 --- a/internal/cli/events_follow.go +++ b/internal/cli/events_follow.go @@ -64,6 +64,11 @@ func runEventsFollow(cmd *cobra.Command) error { limit = eventsDefaultLimit } + tail, err := eventsTailFlag(cmd) + if err != nil { + return err + } + interval, _ := cmd.Flags().GetDuration("interval") if interval <= 0 { return cmdErr( @@ -89,7 +94,7 @@ func runEventsFollow(cmd *cobra.Command) error { if err := followEvents(ctx, followOptions{ conn: conn, query: engine.EventQuery{ - Since: since, RunID: runID, Limit: limit, + Since: since, RunID: runID, Limit: limit, Tail: tail, ProjectID: eventsProjectScope(cmd), }, interval: interval, @@ -127,6 +132,25 @@ func followEvents(ctx context.Context, opts followOptions) error { // per-cycle closure because it is the ONE thing that must survive a cycle. cursor := opts.query.Since + // tail is the STARTING CURSOR, and it applies to the FIRST CYCLE ONLY + // (DKT-752). + // + // `--tail N` means "start me at the newest N". Before this, the follow + // dropped the flag entirely and opened at seq 0, so a watcher arming + // `--follow --tail 20` received the whole retained history — thousands of + // rows on a busy store — before reaching anything live. + // + // It cannot survive past the first cycle: `Tail` is a SELECTION over the + // whole feed, not a window above a cursor, so a second tailed read would + // re-print the same newest N every interval and never advance. So the first + // cycle selects the newest N, the cursor advances over what it returned by + // the ordinary W3 rule, and every cycle after it is a plain `--since` read. + // + // A first cycle that returns NOTHING leaves the cursor at Since (0 — `--tail` + // and `--since` are mutually exclusive), which is correct: an empty feed has + // nothing to skip past, and whatever is written next lands above 0. + tail := opts.query.Tail + // fatal carries a TERMINAL condition out of the loop (W8). // // It exists because RunWatch tolerates two errors before giving up — the @@ -161,6 +185,13 @@ func followEvents(ctx context.Context, opts followOptions) error { }, func(ctx context.Context, w *output.Writer) error { query := opts.query query.Since = cursor + query.Tail = tail + // The first tailed read is a selection over the whole feed, so it starts + // from no cursor; ListEvents ignores Limit in that mode and the window is + // N itself. + if tail > 0 { + query.Since = 0 + } page, err := engine.ListEvents(opts.conn, query) if err != nil { // GONE is terminal; anything else is transient and gets RunWatch's @@ -179,6 +210,19 @@ func followEvents(ctx context.Context, opts followOptions) error { cursor = page.Events[n-1].Seq } + // The window that actually bounded this page: N under the opening tailed + // read, `--limit` for every cycle after it. It is captured BEFORE `tail` + // is spent so `truncated` describes the bound that applied. + // + // `tail` is cleared only after a SUCCESSFUL read: a cycle that failed + // transiently returned above without reaching here, so the retry is still + // the opening read and still starts at the newest N. + effectiveLimit := opts.query.Limit + if tail > 0 { + effectiveLimit = tail + tail = 0 + } + // A quiet cycle prints NOTHING — not an empty array, not "No events." // A follower's output must be the events and only the events, or a // consumer piping it would receive a page of nothing every interval @@ -190,7 +234,7 @@ func followEvents(ctx context.Context, opts followOptions) error { } result := eventsListResult{ - Events: page.Events, Total: page.Total, limit: opts.query.Limit} + Events: page.Events, Total: page.Total, limit: effectiveLimit} var message string if !w.JSONMode { // A follow with no project scope is the cross-project feed, so it diff --git a/internal/cli/events_follow_test.go b/internal/cli/events_follow_test.go index ff702cb9..4b9711a3 100644 --- a/internal/cli/events_follow_test.go +++ b/internal/cli/events_follow_test.go @@ -38,7 +38,12 @@ func followTrustEvent(t *testing.T, conn *sql.DB, name string) { sum := sha256.Sum256([]byte(name)) err := engine.RecordTrustEvent( conn, engine.EventTrustAdded, - engine.TrustGrant{Name: name, ArgvSHA256: fmt.Sprintf("%x", sum), Repo: "repo"}, + // Actor and Cwd are REQUIRED (DKT-595) — the writer refuses an + // unattributed grant — so even this plumbing fixture supplies them. + engine.TrustGrant{ + Name: name, ArgvSHA256: fmt.Sprintf("%x", sum), Repo: "repo", + Actor: "tester", Cwd: "/repo", + }, time.Now().UnixMilli()) if err != nil { t.Fatalf("RecordTrustEvent(%q): %v", name, err) @@ -234,6 +239,138 @@ func TestFollowCursorNeverRepeatsOrSkips(t *testing.T) { } } +// TestFollowTailStartsAtTheNewestN is DKT-752: `--follow --tail N` opens at the +// newest N and then follows, instead of replaying the whole retained history. +// +// THE ASSERTION IS ON THE HISTORICAL HALF, not on the total. A follow that +// printed the newest 3 and then the live rows is right; a follow that printed +// all 30 first and then the live rows is the bug, and both end with the same +// live events on stdout. So the test pins WHICH events preceded the live ones: +// at most N of them, and specifically the last N seqs in the store when the +// follow opened. +func TestFollowTailStartsAtTheNewestN(t *testing.T) { + conn := newTestDB(t) + + // A history comfortably larger than the tail, so "the newest N" and + // "everything" are different answers. + const history = 30 + for i := 0; i < history; i++ { + followTrustEvent(t, conn, fmt.Sprintf("old-%d", i)) + } + before, err := engine.ListEvents(conn, engine.EventQuery{}) + testsupport.Must(t, err, "ListEvents: %v", err) + if len(before.Events) != history { + t.Fatalf("fixture wrote %d events, want %d", len(before.Events), history) + } + const tail = 3 + newest := make([]int64, 0, tail) + for _, e := range before.Events[history-tail:] { + newest = append(newest, e.Seq) + } + oldest := before.Events[0].Seq + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + var stdout, stderr syncBuffer + opts := followOpts(conn, &stdout, &stderr) + opts.query.Tail = tail + + done := make(chan error, 1) + go func() { done <- followEvents(ctx, opts) }() + + // The opening page must land before the live rows are written, or a slow + // first cycle could tail a feed that already contains them and the + // historical half would be unidentifiable. + waitFor(t, func() bool { return len(followSeqs(t, stdout.String())) >= tail }) + opened := followSeqs(t, stdout.String()) + + // Then it FOLLOWS: rows written after the tail still arrive. + const live = 2 + for i := 0; i < live; i++ { + followTrustEvent(t, conn, fmt.Sprintf("live-%d", i)) + } + waitFor(t, func() bool { return len(followSeqs(t, stdout.String())) >= tail+live }) + cancel() + testsupport.Must(t, <-done, "a cancelled follow returned an error") + + seqs := followSeqs(t, stdout.String()) + if len(seqs) < tail+live { + t.Fatalf("the follow printed %v, want the newest %d then %d live events", + seqs, tail, live) + } + + // The acceptance criterion: AT MOST N historical events, and they are the + // newest N. + historical := seqs[:len(seqs)-live] + if len(historical) != tail { + t.Errorf("the follow printed %d historical events before the live ones, want at most %d"+ + " (--tail was ignored and the full history replayed): %v", + len(historical), tail, historical) + } + for i := range newest { + if i >= len(historical) || historical[i] != newest[i] { + t.Fatalf("the follow opened on %v, want the newest %d seqs %v", + historical, tail, newest) + } + } + for _, seq := range seqs { + if seq == oldest { + t.Errorf("the follow printed seq %d, the OLDEST retained event: "+ + "--tail did not move the starting cursor", oldest) + } + } + + // The opening page itself was the tail, not a first slice of the history. + if len(opened) > tail { + t.Errorf("the opening cycle printed %d events, want at most --tail=%d: %v", + len(opened), tail, opened) + } + + // Nothing is printed twice: the tail is a STARTING CURSOR, so the cycle + // after it reads with --since rather than re-selecting the same newest N. + seen := make(map[int64]bool, len(seqs)) + for _, seq := range seqs { + if seen[seq] { + t.Errorf("seq %d was printed twice; --tail was applied to more than the first cycle", + seq) + } + seen[seq] = true + } +} + +// TestFollowTailOnAnEmptyFeedStillFollows is the boundary the tail's +// "first cycle only" rule turns on: an opening tailed read over an EMPTY store +// returns nothing, so the cursor never advances — and the follow must still +// print what is written next rather than sitting on a spent tail forever. +func TestFollowTailOnAnEmptyFeedStillFollows(t *testing.T) { + conn := newTestDB(t) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + var stdout, stderr syncBuffer + opts := followOpts(conn, &stdout, &stderr) + opts.query.Tail = 5 + + done := make(chan error, 1) + go func() { done <- followEvents(ctx, opts) }() + + const live = 3 + for i := 0; i < live; i++ { + followTrustEvent(t, conn, fmt.Sprintf("after-%d", i)) + } + + waitFor(t, func() bool { return len(followSeqs(t, stdout.String())) >= live }) + cancel() + testsupport.Must(t, <-done, "a cancelled follow returned an error") + + if seqs := followSeqs(t, stdout.String()); len(seqs) != live { + t.Fatalf("a tailed follow over an initially empty feed printed %v, want %d events", + seqs, live) + } +} + // TestFollowStopsOnGone is W8, the clause that keeps `--follow` honest about a // prune: a cursor that falls below the retained minimum TERMINATES the follow // with GONE rather than resuming at the new minimum. diff --git a/internal/cli/events_test.go b/internal/cli/events_test.go index 3817b1ba..6f083ced 100644 --- a/internal/cli/events_test.go +++ b/internal/cli/events_test.go @@ -1,10 +1,17 @@ package cli import ( + "context" + "database/sql" + "encoding/json" "strings" "testing" + "github.com/ALT-F4-LLC/docket/internal/db" "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" ) // renderEventList's issue column (DKT-74): engine.Event.Issue was already on @@ -38,3 +45,217 @@ func TestRenderEventListHoldsIssueColumnOpenWhenAbsent(t *testing.T) { t.Fatalf("expected issue column held open with %q, got %q in %q", "-", fields[4], out) } } + +// DKT-862 — a gate-recorded line says whether the verdict routed anything. +// +// `gate-recorded ... detail=ac-commands exit=2 verdict=fail` was byte-identical +// whether the gate BLOCKED the step or was a `pre = true` advisory input to it. +// On RUN-61 three pre-gate failures sat in this feed beside the `step-routed` +// that contradicted them, with nothing on the line to reconcile the two. +// +// The marker rides as its own key in `data`, because eventDetail renders the +// payload as sorted `key=value` pairs and INTERPRETS NOTHING — the genericity +// line this feed holds with the report's metadata rollup. +func TestGateRecordedLineMarksTheAdvisoryVerdict(t *testing.T) { + events := []engine.Event{{ + Seq: 11394, AtMS: 1000, Kind: engine.EventGateRecorded, + Run: "RUN-61", Issue: "DKT-862", Step: "verify@1", + Data: json.RawMessage( + `{"detail":"ac-commands","exit":2,"verdict":"fail","pre":true}`), + }, { + Seq: 11400, AtMS: 1001, Kind: engine.EventGateRecorded, + Run: "RUN-61", Issue: "DKT-862", Step: "implement@1", + Data: json.RawMessage(`{"detail":"build","exit":2,"verdict":"fail"}`), + }} + + lines := strings.Split(renderEventList(events, false), "\n") + if len(lines) != 2 { + t.Fatalf("expected one line per event, got %d: %q", len(lines), lines) + } + if !strings.Contains(lines[0], "pre=true") { + t.Errorf("the pre-gate's gate-recorded line does not mark a verdict "+ + "that routed nothing:\n %s", lines[0]) + } + // The verdict and exit stay exactly where they were (DKT-63): the marker + // qualifies the line, it does not rewrite the closed vocabulary a program + // reads out of `--json`. + for _, needle := range []string{"detail=ac-commands", "exit=2", "verdict=fail"} { + if !strings.Contains(lines[0], needle) { + t.Errorf("the marker cost the line %q:\n %s", needle, lines[0]) + } + } + // And the blocking failure beside it stays unmarked, which is what makes + // the marker readable as a difference rather than as decoration. + if strings.Contains(lines[1], "pre=") { + t.Errorf("a blocking gate's line carries the advisory marker:\n %s", + lines[1]) + } +} + +// eventsListCmdWithDB is `events list`'s flag set on a pristine command, so each +// case parses its own flags rather than inheriting the previous case's values. +func eventsListCmdWithDB(conn *sql.DB) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().Int64("since", 0, "") + cmd.Flags().String("run", "", "") + cmd.Flags().Bool("all-projects", false, "") + cmd.Flags().Int("tail", 0, "") + cmd.Flags().Int("limit", 0, "") + return cmd +} + +// eventsListJSON is the verb's envelope, read the way the analysts in DKT-583 +// read it: `ok`, and the page under `data`. +type eventsListJSON struct { + OK bool `json:"ok"` + Data struct { + Events []struct { + Seq int64 `json:"seq"` + Run string `json:"run"` + } `json:"events"` + Total int `json:"total"` + } `json:"data"` +} + +// eventsFor drives `events list --json` from one project's cwd. +func eventsFor( + t *testing.T, conn *sql.DB, invoking int, runRef string, allProjects bool, +) eventsListJSON { + t.Helper() + cmd := eventsListCmdWithDB(conn) + if err := cmd.Flags().Set("run", runRef); err != nil { + t.Fatalf("setting --run=%q: %v", runRef, err) + } + if allProjects { + if err := cmd.Flags().Set("all-projects", "true"); err != nil { + t.Fatalf("setting --all-projects: %v", err) + } + } + cmd.SetContext(context.WithValue(cmd.Context(), projectKey, invoking)) + + w, buf := bufWriter(true) + if err := runEventsList(cmd, w); err != nil { + t.Fatalf("runEventsList(--run %q): %v", runRef, err) + } + + var out eventsListJSON + if err := json.Unmarshal(buf.Bytes(), &out); err != nil { + t.Fatalf("unmarshalling the envelope: %v\n%s", err, buf.String()) + } + return out +} + +// twoProjectRuns is DKT-583's shape: a machine-global store with two projects, +// one run each, and events on both. +func twoProjectRuns(t *testing.T, conn *sql.DB) (here, there int, hereRun, thereRun string) { + t.Helper() + here, err := db.EnsureProject(conn, "/src/here.git", "here.git", 1) + testsupport.Must(t, err, "registering the local project: %v", err) + there, err = db.EnsureProject(conn, "/src/there.git", "there.git", 2) + testsupport.Must(t, err, "registering the other project: %v", err) + if here == there { + t.Fatalf("the fixture needs two distinct projects, got %d twice", here) + } + + mine, err := db.InsertRun(conn, here, "here", 0, 1) + testsupport.Must(t, err, "starting the local run: %v", err) + theirs, err := db.InsertRun(conn, there, "there", 0, 2) + testsupport.Must(t, err, "starting the other run: %v", err) + + event := func(runID int, kind string) { + t.Helper() + _, err := conn.Exec( + `INSERT INTO events (at_ms, kind, run_id, data) VALUES (?, ?, ?, '{}')`, + 1, kind, runID) + testsupport.Must(t, err, "writing an event: %v", err) + } + event(mine.ID, "run-activated") + event(theirs.ID, "run-activated") + event(theirs.ID, "step-claimed") + + return here, there, model.FormatRunID(mine.ID), model.FormatRunID(theirs.ID) +} + +// TestEventsListRunInAnotherProjectIsNotAnEmptySuccess is DKT-583, verbatim. +// +// From one project's cwd, `events list --run RUN-N --json` for a run owned by +// ANOTHER project answered `{"ok":true,"data":{"events":null,"total":0}}` while +// `--all-projects` returned hundreds of rows for the same run. `run report`, +// `step show`, and `step artifact` all answer cross-project from that same cwd, +// so the empty page read as fact rather than as a scoping accident — four +// analysts recorded "this run has no events". +// +// THE INVARIANT IS THE ABSENCE OF THE FALSE POSITIVE: a real run must never be +// answered with a successful empty feed. +func TestEventsListRunInAnotherProjectIsNotAnEmptySuccess(t *testing.T) { + conn := newTestDB(t) + here, _, _, thereRun := twoProjectRuns(t, conn) + + out := eventsFor(t, conn, here, thereRun, false) + + if !out.OK { + t.Fatalf("`events list --run %s` failed outright: %+v", thereRun, out) + } + if out.Data.Total == 0 || len(out.Data.Events) == 0 { + t.Fatalf("`events list --run %s` from another project answered ok with an "+ + "empty feed — DKT-583's exact defect: %+v", thereRun, out.Data) + } + if len(out.Data.Events) != 2 || out.Data.Total != 2 { + t.Fatalf("answered %d events (total %d), want the run's 2", + len(out.Data.Events), out.Data.Total) + } + for _, e := range out.Data.Events { + if e.Run != thereRun { + t.Errorf("the feed carries an event from another run: %+v", e) + } + } +} + +// A run inside the invoking project is unchanged: the scope it never needed is +// gone, and the answer is the same one it always gave. +func TestEventsListRunInTheInvokingProjectIsUnchanged(t *testing.T) { + conn := newTestDB(t) + here, _, hereRun, _ := twoProjectRuns(t, conn) + + out := eventsFor(t, conn, here, hereRun, false) + + if len(out.Data.Events) != 1 || out.Data.Total != 1 { + t.Fatalf("the invoking project's own run answered %d events (total %d), want 1", + len(out.Data.Events), out.Data.Total) + } + if out.Data.Events[0].Run != hereRun { + t.Errorf("answered %q, want %q", out.Data.Events[0].Run, hereRun) + } +} + +// --all-projects with --run still answers the same run: the workaround the +// analysts found keeps working. +func TestEventsListAllProjectsWithRunStillAnswers(t *testing.T) { + conn := newTestDB(t) + here, _, _, thereRun := twoProjectRuns(t, conn) + + out := eventsFor(t, conn, here, thereRun, true) + + if len(out.Data.Events) != 2 || out.Data.Total != 2 { + t.Fatalf("--all-projects --run %s answered %d events (total %d), want 2", + thereRun, len(out.Data.Events), out.Data.Total) + } +} + +// The run-less feed is STILL scoped to the invoking project: dropping the scope +// predicate under --run must not drop it everywhere. +func TestEventsListWithoutRunStaysScopedToTheInvokingProject(t *testing.T) { + conn := newTestDB(t) + here, _, hereRun, _ := twoProjectRuns(t, conn) + + out := eventsFor(t, conn, here, "", false) + + if len(out.Data.Events) != 1 { + t.Fatalf("the unfiltered feed answered %d events, want only this project's 1: %+v", + len(out.Data.Events), out.Data.Events) + } + if out.Data.Events[0].Run != hereRun { + t.Errorf("the scoped feed carries %q, want this project's %q", + out.Data.Events[0].Run, hereRun) + } +} diff --git a/internal/cli/issue_edit.go b/internal/cli/issue_edit.go index b78050bf..763b6f59 100644 --- a/internal/cli/issue_edit.go +++ b/internal/cli/issue_edit.go @@ -1,6 +1,7 @@ package cli import ( + "encoding/json" "fmt" "io" "os" @@ -8,6 +9,7 @@ import ( "github.com/ALT-F4-LLC/docket/internal/config" "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/output" "github.com/spf13/cobra" @@ -16,157 +18,259 @@ import ( var editCmd = &cobra.Command{ Use: "edit [id]", Short: "Edit an existing issue", - Args: cobra.ExactArgs(1), + Long: `Edit an existing issue. + +A MID-RUN EDIT DOES NOT REACH AN ALREADY-ACTIVATED RUN'S PACKETS. Activation +freezes the issue's description, title, kind, labels and scope into the run, +and every packet renders from that snapshot — so editing --description or +--scope here changes what the NEXT activation will freeze, never what a step +already in flight will read. + +--scope is the one whose live column still does something: the scheduler reads +it live for its mutual-exclusion check, so a correction takes effect against +collisions immediately. It does not change any live step's rendered scope or +the paths that step's diff is recorded over. To make a widened scope real for +work already in a run, refresh that run's snapshot as a second, explicit act — +` + "`docket run refresh-scope RUN-N --issue DKT-M --reason R`" + ` — or, where the +run's premise changed rather than one declaration, take the issue out of the +run and re-plan it. This command warns, naming the run and both verbs, when a +--scope edit lands on an issue with live steps in a live run.`, + Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - w := getWriter(cmd) - conn := getDB(cmd) + return runIssueEdit(cmd, args, getWriter(cmd)) + }, +} - id, err := issueArg(args[0]) - if err != nil { - return err - } +// runIssueEdit is the verb's body, taking its Writer explicitly the way every +// `watchable` verb's does. The advisory this verb now raises (DKT-741) travels +// on stderr, and a test that could not substitute that stream would have to +// assert the warning somewhere other than where an operator reads it. +func runIssueEdit(cmd *cobra.Command, args []string, w *output.Writer) error { + conn := getDB(cmd) - // Verify issue exists. The row is kept: reparenting compares projects - // issue-to-issue, so the edited issue's own home is needed below. - issue, err := getIssueOrErr(conn, id, fmt.Sprintf("issue %s", args[0])) - if err != nil { - return err - } + id, err := issueArg(args[0]) + if err != nil { + return err + } - updates := make(map[string]interface{}) - filesChanged := false + // Verify issue exists. The row is kept: reparenting compares projects + // issue-to-issue, so the edited issue's own home is needed below. + issue, err := getIssueOrErr(conn, id, fmt.Sprintf("issue %s", args[0])) + if err != nil { + return err + } - if cmd.Flags().Changed("title") { - title, _ := cmd.Flags().GetString("title") - updates["title"] = title - } + updates := make(map[string]interface{}) + filesChanged := false - if cmd.Flags().Changed("description") { - description, _ := cmd.Flags().GetString("description") - if description == "-" { - const maxStdinSize = 1 << 20 // 1 MiB - data, err := io.ReadAll(io.LimitReader(os.Stdin, maxStdinSize)) - if err != nil { - return cmdErr(fmt.Errorf("reading description from stdin: %w", err), output.ErrGeneral) - } - description = strings.TrimRight(string(data), "\n") - } - updates["description"] = description - } + if cmd.Flags().Changed("title") { + title, _ := cmd.Flags().GetString("title") + updates["title"] = title + } - if cmd.Flags().Changed("status") { - status, _ := cmd.Flags().GetString("status") - if err := model.ValidateStatus(model.Status(status)); err != nil { - return cmdErr(err, output.ErrValidation) + if cmd.Flags().Changed("description") { + description, _ := cmd.Flags().GetString("description") + if description == "-" { + const maxStdinSize = 1 << 20 // 1 MiB + data, err := io.ReadAll(io.LimitReader(os.Stdin, maxStdinSize)) + if err != nil { + return cmdErr(fmt.Errorf("reading description from stdin: %w", err), output.ErrGeneral) } - updates["status"] = status + description = strings.TrimRight(string(data), "\n") } + updates["description"] = description + } - if cmd.Flags().Changed("priority") { - priority, _ := cmd.Flags().GetString("priority") - if err := model.ValidatePriority(model.Priority(priority)); err != nil { - return cmdErr(err, output.ErrValidation) - } - updates["priority"] = priority + if cmd.Flags().Changed("status") { + status, _ := cmd.Flags().GetString("status") + if err := model.ValidateStatus(model.Status(status)); err != nil { + return cmdErr(err, output.ErrValidation) } + updates["status"] = status + } - if cmd.Flags().Changed("type") { - kind, _ := cmd.Flags().GetString("type") - if err := model.ValidateIssueKind(model.IssueKind(kind)); err != nil { - return cmdErr(err, output.ErrValidation) - } - updates["kind"] = kind + if cmd.Flags().Changed("priority") { + priority, _ := cmd.Flags().GetString("priority") + if err := model.ValidatePriority(model.Priority(priority)); err != nil { + return cmdErr(err, output.ErrValidation) } + updates["priority"] = priority + } - if cmd.Flags().Changed("assignee") { - assignee, _ := cmd.Flags().GetString("assignee") - updates["assignee"] = assignee + if cmd.Flags().Changed("type") { + kind, _ := cmd.Flags().GetString("type") + if err := model.ValidateIssueKind(model.IssueKind(kind)); err != nil { + return cmdErr(err, output.ErrValidation) } + updates["kind"] = kind + } - if cmd.Flags().Changed("file") { - fileFlag, _ := cmd.Flags().GetStringSlice("file") - if err := db.SetIssueFiles(conn, id, fileFlag, config.DefaultAuthor()); err != nil { - return cmdErr(fmt.Errorf("setting files: %w", err), output.ErrGeneral) - } - filesChanged = true + if cmd.Flags().Changed("assignee") { + assignee, _ := cmd.Flags().GetString("assignee") + updates["assignee"] = assignee + } + + if cmd.Flags().Changed("file") { + fileFlag, _ := cmd.Flags().GetStringSlice("file") + if err := db.SetIssueFiles(conn, id, fileFlag, config.DefaultAuthor()); err != nil { + return cmdErr(fmt.Errorf("setting files: %w", err), output.ErrGeneral) } + filesChanged = true + } - if cmd.Flags().Changed("parent") { - parent, _ := cmd.Flags().GetString("parent") - if strings.EqualFold(parent, "0") || strings.EqualFold(parent, "none") { - updates["parent_id"] = nil - } else { - // The parent must exist AND share the edited issue's project - // (DKT-22) — issue-to-issue, never against the invoking cwd. - parentIssue, err := resolveParentIssue(conn, parent, issue.ProjectID) - if err != nil { - return err - } - newParentID := parentIssue.ID - if newParentID == id { - return cmdErr(fmt.Errorf("cannot set parent to self"), output.ErrValidation) - } - isCycle, err := db.IsDescendant(conn, id, newParentID) - if err != nil { - return cmdErr(fmt.Errorf("checking for cycles: %w", err), output.ErrGeneral) - } - if isCycle { - return cmdErr(fmt.Errorf("cannot reparent: would create a cycle"), output.ErrConflict) - } - updates["parent_id"] = newParentID + if cmd.Flags().Changed("parent") { + parent, _ := cmd.Flags().GetString("parent") + if strings.EqualFold(parent, "0") || strings.EqualFold(parent, "none") { + updates["parent_id"] = nil + } else { + // The parent must exist AND share the edited issue's project + // (DKT-22) — issue-to-issue, never against the invoking cwd. + parentIssue, err := resolveParentIssue(conn, parent, issue.ProjectID) + if err != nil { + return err + } + newParentID := parentIssue.ID + if newParentID == id { + return cmdErr(fmt.Errorf("cannot set parent to self"), output.ErrValidation) } + isCycle, err := db.IsDescendant(conn, id, newParentID) + if err != nil { + return cmdErr(fmt.Errorf("checking for cycles: %w", err), output.ErrGeneral) + } + if isCycle { + return cmdErr(fmt.Errorf("cannot reparent: would create a cycle"), output.ErrConflict) + } + updates["parent_id"] = newParentID } + } + + // `--scope` counts as a change. Without this the early return below + // would report "No changes specified" for a scope-only edit and, in + // JSON mode, emit the issue unchanged — after having written the + // scope. Scope lives on its own column and never enters `updates`, + // so it has to be counted here explicitly. + scopeChanged := cmd.Flags().Changed(scopeFlag) + if err := applyScope(cmd, conn, id); err != nil { + return err + } - // `--scope` counts as a change. Without this the early return below - // would report "No changes specified" for a scope-only edit and, in - // JSON mode, emit the issue unchanged — after having written the - // scope. Scope lives on its own column and never enters `updates`, - // so it has to be counted here explicitly. - scopeChanged := cmd.Flags().Changed(scopeFlag) - if err := applyScope(cmd, conn, id); err != nil { - return err + // DKT-741: the write above landed on `issues.scope_globs` and reached + // no already-activated run — packets and recorded diffs read the + // activation snapshot (§5.1.1), and no EDIT refreshes it (DKT-869's + // `run refresh-scope` is a separate, explicitly-named second act, which + // is what the advisory now points at). Named AFTER the write because + // the answer is a property of the run's frozen snapshot, which the + // write did not touch; an operator who meant to widen scope to unblock + // live work needs to learn here that this edit alone did not do it. + // + // BOTH CHANNELS, the way every advisory in this CLI travels: the + // stderr line for a human (suppressed in JSON mode by design), and + // `warnings` in the envelope for the relay that ran the edit — which + // is who ran it when this was found. + var scopeWarnings []string + if scopeChanged { + scopeWarnings = engine.ScopeEditFrozenForActiveRuns(conn, id) + for _, warning := range scopeWarnings { + w.Warn("%s", warning) } + } - if len(updates) == 0 && !filesChanged && !scopeChanged { - if w.JSONMode { - issue, err := db.GetIssue(conn, id) - if err != nil { - return cmdErr(fmt.Errorf("fetching issue: %w", err), output.ErrGeneral) - } - w.Success(withIssueVersion(issue), "") - } else { - w.Info("No changes specified") + if len(updates) == 0 && !filesChanged && !scopeChanged { + if w.JSONMode { + issue, err := db.GetIssue(conn, id) + if err != nil { + return cmdErr(fmt.Errorf("fetching issue: %w", err), output.ErrGeneral) } - return nil + w.Success(withIssueVersion(issue), "") + } else { + w.Info("No changes specified") } + return nil + } - ifVersion, err := ifVersionOf(cmd) - if err != nil { - return err - } + ifVersion, err := ifVersionOf(cmd) + if err != nil { + return err + } - if len(updates) > 0 || ifVersion != nil { - if err := db.UpdateIssueCAS(conn, id, updates, config.DefaultAuthor(), ifVersion); err != nil { - if e := casError(err, fmt.Sprintf("issue %s", args[0])); e != nil { - return e - } - return cmdErr(fmt.Errorf("updating issue: %w", err), output.ErrGeneral) + if len(updates) > 0 || ifVersion != nil { + if err := db.UpdateIssueCAS(conn, id, updates, config.DefaultAuthor(), ifVersion); err != nil { + if e := casError(err, fmt.Sprintf("issue %s", args[0])); e != nil { + return e } + return cmdErr(fmt.Errorf("updating issue: %w", err), output.ErrGeneral) } + } - updated, err := db.GetIssue(conn, id) - if err != nil { - return cmdErr(fmt.Errorf("fetching updated issue: %w", err), output.ErrGeneral) - } + updated, err := db.GetIssue(conn, id) + if err != nil { + return cmdErr(fmt.Errorf("fetching updated issue: %w", err), output.ErrGeneral) + } - if err := hydrateIssueAssociations(conn, updated); err != nil { - return err - } + if err := hydrateIssueAssociations(conn, updated); err != nil { + return err + } - w.Success(withIssueVersion(updated), fmt.Sprintf("Updated %s: %s", model.FormatID(id), updated.Title)) + w.Success( + withEditWarnings(withIssueVersion(updated), scopeWarnings), + fmt.Sprintf("Updated %s: %s", model.FormatID(id), updated.Title)) - return nil - }, + return nil +} + +// editWarningsPayload is the edited issue plus the advisories the edit raised +// (DKT-741), for the JSON reader that `w.Warn` deliberately does not reach. +// +// It SPLICES a key into whatever shape the wrapped payload already marshals +// to, rather than being a new shape of its own, so v1 and v2 each keep their +// own issue rendering and gain the same one additive key. With no warnings it +// is not constructed at all (withEditWarnings returns the payload unchanged), +// so every edit that raises none emits byte-identical JSON to what it always +// did — including the field ORDER a re-marshal here would otherwise sort. +type editWarningsPayload struct { + inner any + warnings []string +} + +func (p editWarningsPayload) MarshalJSON() ([]byte, error) { + return spliceWarnings(json.Marshal(p.inner))(p.warnings) +} + +// VersionedPayload implements output.Versioned: under v2 the inner payload's +// own v2 rendering carries the warnings, so neither envelope loses them. +func (p editWarningsPayload) VersionedPayload() any { + if v, ok := p.inner.(output.Versioned); ok { + return editWarningsPayload{inner: v.VersionedPayload(), warnings: p.warnings} + } + return p +} + +// withEditWarnings wraps payload only when there is something to say. +func withEditWarnings(payload any, warnings []string) any { + if len(warnings) == 0 { + return payload + } + return editWarningsPayload{inner: payload, warnings: warnings} +} + +// spliceWarnings adds a `warnings` array to an already-marshaled JSON object. +// Curried so the (bytes, error) pair of a Marshal call feeds it directly. +func spliceWarnings(raw []byte, err error) func([]string) ([]byte, error) { + return func(warnings []string) ([]byte, error) { + if err != nil { + return nil, err + } + var fields map[string]json.RawMessage + if err := json.Unmarshal(raw, &fields); err != nil { + return nil, fmt.Errorf("adding warnings to the issue payload: %w", err) + } + encoded, err := json.Marshal(warnings) + if err != nil { + return nil, fmt.Errorf("encoding warnings: %w", err) + } + fields["warnings"] = encoded + return json.Marshal(fields) + } } func init() { diff --git a/internal/cli/issue_edit_scope_test.go b/internal/cli/issue_edit_scope_test.go new file mode 100644 index 00000000..93d9da71 --- /dev/null +++ b/internal/cli/issue_edit_scope_test.go @@ -0,0 +1,175 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-741 at the verb level. The engine half — that the freeze is real and +// that ScopeEditFrozenForActiveRuns names the right runs — is pinned in +// internal/engine/dkt741_test.go. What is pinned HERE is the wiring an +// operator and a conductor actually meet: `issue edit --scope` on an issue +// with live steps in a live run says so, on both channels, and names the verb +// that would work. + +// scopedIssueInLiveRun seeds the shape the warning is about: an issue with a +// declared scope, bound into an ACTIVATED run (so a snapshot exists) that +// still holds one pending step. +// +// The rows go in directly rather than through `run activate`, because the verb +// under test reads them and does not care how they were produced — and the +// engine package already drives the real activation for the same fixture. +func scopedIssueInLiveRun(t *testing.T, conn *sql.DB, scope, snapshotScope string) (int, int) { + t.Helper() + + issueID, err := db.CreateIssue(conn, &model.Issue{ + Title: "widen me", Status: model.StatusBacklog, + Priority: model.PriorityNone, Kind: model.IssueKindTask, + }, nil, nil) + testsupport.Must(t, err, "creating the issue: %v", err) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issueID, scope), "declaring scope") + + wf, _, err := db.InsertWorkflow(conn, &model.Workflow{ + Name: "parks", Version: 1, SourcePath: "parks.toml", + SourceSHA256: "0", Body: "", Parsed: "{}", + }, 1) + testsupport.Must(t, err, "registering a workflow: %v", err) + + run, err := db.InsertRun(conn, 1, "test run", 0, 1) + testsupport.Must(t, err, "starting the run: %v", err) + testsupport.Must(t, db.AddRunIssue(conn, run.ID, issueID), "binding the issue") + + _, err = conn.Exec( + `UPDATE runs SET status = ? WHERE id = ?`, string(model.RunActive), run.ID) + testsupport.Must(t, err, "activating the run: %v", err) + _, err = conn.Exec( + `UPDATE run_issues SET workflow_id = ?, issue_snapshot = ? + WHERE run_id = ? AND issue_id = ?`, + wf.ID, `{"title":"widen me","kind":"task","labels":[],"scope":`+snapshotScope+`}`, + run.ID, issueID) + testsupport.Must(t, err, "writing the snapshot: %v", err) + _, err = conn.Exec( + `INSERT INTO steps + (run_id, issue_id, workflow_id, step_name, instance, kind, status, + created_at_ms, updated_at_ms) + VALUES (?, ?, ?, 'fix', 'fix@2', 'executor', ?, 1, 1)`, + run.ID, issueID, wf.ID, db.StepPending) + testsupport.Must(t, err, "inserting the step: %v", err) + + return run.ID, issueID +} + +// editWithScope drives the real verb body with a substituted Writer. +func editWithScope(t *testing.T, conn *sql.DB, issueID int, jsonMode bool, globs ...string) (string, string) { + t.Helper() + cmd := cmdWithDB(conn) + addScopeFlag(cmd) + addIfVersionFlag(cmd) + for _, g := range globs { + testsupport.Must(t, cmd.Flags().Set(scopeFlag, g), "setting --scope") + } + + w, stdout := bufWriter(jsonMode) + stderr := &strings.Builder{} + w.Stderr = stderr + testsupport.Must(t, + runIssueEdit(cmd, []string{model.FormatID(issueID)}, w), "issue edit") + return stdout.String(), stderr.String() +} + +// TestIssueEditScopeWarnsOnALiveRun is the acceptance criterion: the widen is +// accepted, and the operator is told in the same breath that it did not reach +// the run and what would. +func TestIssueEditScopeWarnsOnALiveRun(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, + `["cli/src/command/start.rs"]`, `["cli/src/command/start.rs"]`) + + _, stderr := editWithScope(t, conn, issueID, false, + "cli/src/command/start.rs", "script/install.sh", "makefile") + + for _, want := range []string{ + "Warning:", + model.FormatRunID(runID), + "run abandon " + model.FormatRunID(runID) + " --issue " + model.FormatID(issueID), + "re-plan", + } { + if !strings.Contains(stderr, want) { + t.Errorf("stderr does not name %q:\n%s", want, stderr) + } + } + + // The edit is REPORTED, not refused — the live column is what the + // scheduler's mutual-exclusion check reads, and that half does take. + if got := scopeColumn(t, conn, issueID); !got.Valid || + got.String != `["cli/src/command/start.rs","script/install.sh","makefile"]` { + t.Errorf("scope_globs = %v, want the widened declaration written", got) + } +} + +// TestIssueEditScopeWarningRidesTheJSONEnvelope is the other channel, and the +// one that matters most here: RUN-52's widen was typed by a CONDUCTOR, and +// w.Warn is suppressed in JSON mode by design. +func TestIssueEditScopeWarningRidesTheJSONEnvelope(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, + `["internal/a/**"]`, `["internal/a/**"]`) + + stdout, stderr := editWithScope(t, conn, issueID, true, + "internal/a/**", "internal/b/**") + + if stderr != "" { + t.Errorf("stderr = %q in JSON mode, want the envelope to be the only "+ + "channel", stderr) + } + + var envelope struct { + OK bool `json:"ok"` + Data struct { + ID string `json:"id"` + Warnings []string `json:"warnings"` + } `json:"data"` + } + testsupport.Must(t, json.Unmarshal([]byte(stdout), &envelope), + "decoding the envelope") + if !envelope.OK { + t.Fatalf("envelope is not ok: %s", stdout) + } + if envelope.Data.ID != model.FormatID(issueID) { + t.Errorf("data.id = %q, want %q — the issue shape must survive the "+ + "added key", envelope.Data.ID, model.FormatID(issueID)) + } + if len(envelope.Data.Warnings) != 1 { + t.Fatalf("data.warnings = %v, want one warning", envelope.Data.Warnings) + } + if !strings.Contains(envelope.Data.Warnings[0], model.FormatRunID(runID)) { + t.Errorf("the enveloped warning does not name the run:\n%s", + envelope.Data.Warnings[0]) + } +} + +// TestIssueEditWithoutAWarningIsUnchanged pins the no-op case byte-for-byte: +// an edit that raises nothing must emit exactly the JSON it always did, with +// no `warnings` key and no re-ordered fields from a splice that ran anyway. +func TestIssueEditWithoutAWarningIsUnchanged(t *testing.T) { + conn := newTestDB(t) + _, issueID := scopedIssueInLiveRun(t, conn, + `["internal/a/**"]`, `["internal/a/**"]`) + + // Same globs as the snapshot froze: nothing was hidden, so nothing is said. + stdout, stderr := editWithScope(t, conn, issueID, true, "internal/a/**") + + if stderr != "" { + t.Errorf("stderr = %q, want silence", stderr) + } + if strings.Contains(stdout, "warnings") { + t.Errorf("the envelope grew a warnings key on an edit that hid "+ + "nothing:\n%s", stdout) + } +} diff --git a/internal/cli/issue_show.go b/internal/cli/issue_show.go index 6895b375..4e001026 100644 --- a/internal/cli/issue_show.go +++ b/internal/cli/issue_show.go @@ -283,7 +283,17 @@ Under --json the issue's id is served under BOTH keys: .data.id, the original spelling, and .data.issue, the noun every other verb keys its primary entity by (run status -> run, step show -> step, dispatch open -> dispatch). They always hold the same value, on --json and --json=v2 alike, -and issue list rows carry the same pair.`, +and issue list rows carry the same pair. + +data IS FLAT. .data.issue is a STRING, the issue id (e.g. "AGT-311") — it +is NOT a nested issue object, and there is no .data.issue.title or +.data.issue.labels to reach into. The issue's own fields sit top-level in +the SAME object beside id and issue: title, description, status, priority, +kind, assignee, labels, files, docs, created_at, updated_at, sub_issues, +relations, linked_proposals, comments, activity, and — only when the issue +has one — scope, resolution, run_disposition. A parse that expects a +nested object under .data.issue and calls a field getter on what it finds +is calling that getter on the id string.`, Args: cobra.MinimumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return watchable(cmd, args, runIssueShow) diff --git a/internal/cli/next.go b/internal/cli/next.go index 2f01214c..a0387ef8 100644 --- a/internal/cli/next.go +++ b/internal/cli/next.go @@ -195,7 +195,7 @@ func init() { nextCmd.Flags().StringSliceP("priority", "p", nil, "Filter by priority (repeatable)") nextCmd.Flags().StringSliceP("label", "l", nil, "Filter by label (repeatable)") nextCmd.Flags().StringSliceP("type", "T", nil, "Filter by type (repeatable)") - nextCmd.Flags().Int("limit", 10, "Maximum number of results") + nextCmd.Flags().Int("limit", 10, "Maximum number of results (issue mode; with --run the full ready set is returned unless --limit is passed)") // --run switches `next` into STEP mode (TDD §6.3.1). Without it the verb // is exactly what it was: a workflow-free repo never passes this flag and // never leaves the issue-mode path. diff --git a/internal/cli/next_steps.go b/internal/cli/next_steps.go index ff514575..0314aa33 100644 --- a/internal/cli/next_steps.go +++ b/internal/cli/next_steps.go @@ -75,6 +75,17 @@ func runNextSteps(cmd *cobra.Command, runRef string, w *output.Writer) error { if err := validateLimit(cmd, limit); err != nil { return err } + // DKT-564: the issue-mode DEFAULT limit does not apply in step mode. Here + // the answer IS a dispatch manifest — the conduct pipeline dispatches it + // verbatim — so a cut the caller never asked for strands the remaining + // steps un-dispatched, and a caller reading v1 JSON cannot even tell it + // happened. A run's ready set is bounded by the run, so returning all of + // it is the safe contract. An EXPLICIT --limit is still honoured: a caller + // who typed a cut asked for one, and the v2 envelope reports it as + // truncated. + if !cmd.Flags().Changed("limit") { + limit = 0 // 0 means no limit (readyRows) + } ready, err := engine.NewEngine().NextSteps(conn, runID, limit, model.NowMS()) if err != nil { diff --git a/internal/cli/next_steps_test.go b/internal/cli/next_steps_test.go index 295f65b5..e70ef494 100644 --- a/internal/cli/next_steps_test.go +++ b/internal/cli/next_steps_test.go @@ -4,9 +4,12 @@ import ( "bytes" "database/sql" "encoding/json" + "fmt" "strings" "testing" + "github.com/spf13/cobra" + "github.com/ALT-F4-LLC/docket/internal/db" "github.com/ALT-F4-LLC/docket/internal/engine" "github.com/ALT-F4-LLC/docket/internal/model" @@ -265,35 +268,45 @@ func TestNextStepModeHumanStepCarriesNoExecutor(t *testing.T) { } } -// TestNextStepModeTruncationContract pins the v2 truncation contract: the -// pre-limit total rides in the envelope so a limited call cannot silently drop -// work (reliability-delta §4.2). -func TestNextStepModeTruncationContract(t *testing.T) { - conn := newTestDB(t) +// activatedRunOverIssues seeds `count` independent task issues into one +// activated run, so `next --run` sees `count` independent root steps ready at +// once. +func activatedRunOverIssues(t *testing.T, conn *sql.DB, count int) int { + t.Helper() - // Two issues, so two independent root steps are ready at once. registerForRun(t, conn, runWorkflow) - var issueIDs []int - for _, title := range []string{"one", "two"} { + run, err := db.InsertRun(conn, 1, "", 0, model.NowMS()) + testsupport.Must(t, err, "starting run: %v", err) + for i := range count { id, err := db.CreateIssue(conn, &model.Issue{ - Title: title, Description: "body", Status: model.StatusBacklog, - Priority: model.PriorityNone, Kind: model.IssueKindTask, + Title: fmt.Sprintf("issue %d", i), Description: "body", + Status: model.StatusBacklog, Priority: model.PriorityNone, + Kind: model.IssueKindTask, }, nil, nil) testsupport.Must(t, err, "creating issue: %v", err) - issueIDs = append(issueIDs, id) - } - run, err := db.InsertRun(conn, 1, "", 0, model.NowMS()) - testsupport.Must(t, err, "starting run: %v", err) - for _, id := range issueIDs { - err := db.AddRunIssue(conn, run.ID, id) + err = db.AddRunIssue(conn, run.ID, id) testsupport.Must(t, err, "adding issue: %v", err) } if _, err := engine.Activate(conn, run.ID, engine.ActivateOptions{NowMS: model.NowMS()}); err != nil { t.Fatalf("activate: %v", err) } + return run.ID +} + +// stepCollection is the v2 collection body `next --run` emits: a Collection +// reshapes to data.{items,total,truncated}, the uniform envelope every list +// verb carries. +type stepCollection struct { + Items []map[string]any `json:"items"` + Total int `json:"total"` + Truncated bool `json:"truncated"` +} - cmd := nextCmdWithDB(conn, 1) - err = cmd.Flags().Set("run", model.FormatRunID(run.ID)) +// stepModeEnvelope drives `next --run` under v2 and decodes that body. +func stepModeEnvelope(t *testing.T, cmd *cobra.Command, runID int) stepCollection { + t.Helper() + + err := cmd.Flags().Set("run", model.FormatRunID(runID)) testsupport.Must(t, err, "setting --run: %v", err) w := &output.Writer{ JSONMode: true, JSONVersion: output.JSONV2, @@ -303,32 +316,73 @@ func TestNextStepModeTruncationContract(t *testing.T) { err = runNext(cmd, nil, w) testsupport.Must(t, err, "runNext: %v", err) - // Under v2 a Collection reshapes to data.{items,total,truncated} — the - // uniform envelope every list verb emits. var envelope struct { - Data struct { - Items []map[string]any `json:"items"` - Total int `json:"total"` - Truncated bool `json:"truncated"` - } `json:"data"` + Data stepCollection `json:"data"` } if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { t.Fatalf("unmarshal: %v\n%s", err, buf.String()) } + return envelope.Data +} + +// TestNextStepModeTruncationContract pins the v2 truncation contract: the +// pre-limit total rides in the envelope so a limited call cannot silently drop +// work (reliability-delta §4.2). The limit is set EXPLICITLY — since DKT-564 +// step mode ignores the issue-mode default, so a cut only happens when the +// caller asked for one. +func TestNextStepModeTruncationContract(t *testing.T) { + conn := newTestDB(t) + + // Two issues, so two independent root steps are ready at once. + runID := activatedRunOverIssues(t, conn, 2) + + cmd := nextCmdWithDB(conn, 10) + err := cmd.Flags().Set("limit", "1") + testsupport.Must(t, err, "setting --limit: %v", err) + data := stepModeEnvelope(t, cmd, runID) - if len(envelope.Data.Items) != 1 { - t.Errorf("returned %d steps under --limit 1", len(envelope.Data.Items)) + if len(data.Items) != 1 { + t.Errorf("returned %d steps under --limit 1", len(data.Items)) } - if envelope.Data.Total != 2 { + if data.Total != 2 { t.Errorf("total = %d, want the PRE-LIMIT 2 — a post-limit count cannot "+ "distinguish 'exactly one ready' from 'one returned, more dropped'", - envelope.Data.Total) + data.Total) } - if !envelope.Data.Truncated { + if !data.Truncated { t.Error("truncated = false with 2 ready and --limit 1") } } +// TestNextStepModeIgnoresDefaultLimit is DKT-564: step mode's answer IS a +// dispatch manifest, dispatched verbatim, so the issue-mode DEFAULT of 10 must +// not cut it — a conductor who never typed --limit would strand every step +// past the tenth un-dispatched, with nothing in v1 JSON to reveal the drop. +func TestNextStepModeIgnoresDefaultLimit(t *testing.T) { + conn := newTestDB(t) + + // More ready steps than the default limit of 10. + const issues = 14 + runID := activatedRunOverIssues(t, conn, issues) + + // The default limit is in force on the flag and untouched by the caller, + // exactly as `docket next --run RUN-N` leaves it. + cmd := nextCmdWithDB(conn, 10) + data := stepModeEnvelope(t, cmd, runID) + + if data.Total != issues { + t.Fatalf("total = %d, want %d ready root steps", data.Total, issues) + } + if len(data.Items) != issues { + t.Errorf("returned %d of %d ready steps with no --limit passed; the "+ + "issue-mode default must not truncate a dispatch manifest", + len(data.Items), issues) + } + if data.Truncated { + t.Error("truncated = true with no --limit passed") + } +} + // TestNextStepModeUnknownRunIsNotFound pins the taxonomy at the boundary. func TestNextStepModeUnknownRunIsNotFound(t *testing.T) { conn := newTestDB(t) diff --git a/internal/cli/registry.go b/internal/cli/registry.go new file mode 100644 index 00000000..6266d9f1 --- /dev/null +++ b/internal/cli/registry.go @@ -0,0 +1,233 @@ +package cli + +import ( + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/spf13/cobra" +) + +// `docket registry` — the STORE-WIDE view of the two registries (DKT-614). +// +// Every other registry verb resolves ONE project from the cwd: `workflow list`, +// `workflow show`'s drift check, `workflow list --orphans`, `schema list`. That +// is right for them and wrong for the question this group exists for, which is +// about the projects an operator is NOT standing in. Auto-registration only +// runs where a run activates, so the projects that fall behind the shared +// corpus are precisely the ones nobody has cd'd into lately — and every +// existing verb is scoped to somewhere they have. +// +// It is a sibling group to `project`, which is the other store-wide surface, +// rather than a flag on `workflow list`, because the report spans BOTH +// registries. A `--all-projects` on `workflow list` would answer half the +// question and leave the schema half to a second verb that does not exist. + +var registryCmd = &cobra.Command{ + Use: "registry", + Short: "Inspect the store's workflow and schema registries", + Long: `Inspect the store's workflow and schema registries. + +Registration is a ROW, not a file. Activation auto-registers the instance +config's current contents, so a project's rows track the corpus only as far as +its last activation — and nothing else ever moves them. These verbs read that +divergence across the whole store.`, +} + +var registryAuditCmd = &cobra.Command{ + Use: "audit", + Short: "Report every project's registry drift against the shared corpus", + Long: `Report, for every project in the store, which registered workflow and +schema names are BEHIND the shared corpus, and which are ORPHANED. + +Auto-registration adopts the instance config's current contents, but only as a +side effect of activating a run IN THAT PROJECT. A project no run has activated +lately goes stale against a corpus every other project already moved past, and +short of cd'ing into each checkout in turn there was no way to see it. + +Two findings, per registered NAME rather than per version: + + behind the project's highest registered version for that name is lower + than the version a file in the instance config now declares. The + next activation there adopts it; until then the project binds the + older definition. + + orphaned no file in any scanned root declares that name any more — the + residue of a rename or a deletion, since registering a new name + never retires the old one. It still binds until it is deprecated + ('docket workflow deprecate @'). + +THE CORPUS IS SCANNED ONCE, not once per project: '~/.docket/config' is shared +by every project in the store, so what "current" means is one answer. The roots +that were scanned are reported, because they are read from THIS invocation's +config — a repository's own '.docket/config/' additions belong to the checkout +you are standing in, so a name another project registered from ITS local config +directory is reported orphaned here. That is why the roots are printed rather +than assumed. + +It REPAIRS NOTHING. Adopting a bumped definition is what activation does, +inside a transaction, with the validation and collision rules that go with it.`, + Args: cobra.NoArgs, + RunE: func(cmd *cobra.Command, args []string) error { + return runRegistryAudit(cmd, args, getWriter(cmd)) + }, +} + +func runRegistryAudit(cmd *cobra.Command, _ []string, w *output.Writer) error { + conn := getDB(cmd) + + var opts engine.RegistryAuditOptions + if ref, _ := cmd.Flags().GetString("project"); strings.TrimSpace(ref) != "" { + target, err := resolveProjectRef(conn, ref) + if err != nil { + return err + } + opts.ProjectID = target.ID + } + + // THE TWO REFUSALS COME FIRST, and they are separate messages because they + // send an operator to different places. With nothing scanned, EVERY + // registered name in the store trivially has no file declaring it and every + // version trivially has no current version to be behind — so the report + // would be invented out of having looked nowhere, which is the one output + // this verb must never produce. `workflow list --orphans` refuses on the + // same two conditions, with the same codes. + index, err := engine.ScanCorpus() + if err != nil { + return cmdErr(err, output.ErrValidation) + } + if !index.Scanned() { + return cmdErr(fmt.Errorf( + "no instance-config root exists on this machine, so no registry can "+ + "be compared against anything: with nothing to scan, every "+ + "registered name would look orphaned"), output.ErrValidation) + } + if err := index.Err(); err != nil { + return cmdErr(fmt.Errorf( + "scanning the instance config for current definitions: %w", err), + output.ErrGeneral) + } + + audit, err := engine.AuditRegistries(conn, index, opts) + if err != nil { + return cmdErr(err, output.ErrGeneral) + } + + // NOT a Collection. The v2 collection envelope replaces `data` with + // {items, total, truncated}, which would drop `roots` and the store-wide + // counts — and the roots are not decoration here, they are the record of + // WHERE "current" was read from, without which a reported orphan cannot be + // judged. This is a report with a header, like `run report`, not a listing. + var message string + if !w.JSONMode { + message = renderRegistryAudit(audit) + } + w.Success(audit, message) + return nil +} + +func renderRegistryAudit(audit *engine.RegistryAudit) string { + var b strings.Builder + + fmt.Fprintf(&b, "scanned %d instance-config root(s) for current versions:\n", + len(audit.Roots)) + for _, root := range audit.Roots { + fmt.Fprintf(&b, " %s\n", root) + } + + // The columns are sized over EVERY project's findings, so the report reads + // as one table rather than as a stack of independently-aligned blocks. + nameWidth := 4 + for _, p := range audit.Projects { + for _, lag := range p.Behind { + nameWidth = max(nameWidth, len(lag.Name)) + } + for _, orphan := range p.Orphaned { + nameWidth = max(nameWidth, len(orphan.Name)) + } + } + + dirty := 0 + for _, p := range audit.Projects { + b.WriteString("\n") + fmt.Fprintf(&b, "%s\n", describeAuditedProject(p)) + if p.Clean() { + if p.Compared == 0 { + // DISTINCT FROM "up to date". A project with no rows has not + // been checked and found healthy; it has never registered + // anything, which for a project that runs workflows is itself + // the finding. + b.WriteString(" nothing registered\n") + continue + } + fmt.Fprintf(&b, " up to date (%d registered name(s))\n", p.Compared) + continue + } + dirty++ + for _, lag := range p.Behind { + fmt.Fprintf(&b, " behind %-9s %-*s registered %d, current %d\n", + lag.Kind, nameWidth, lag.Name, lag.RegisteredVersion, lag.CurrentVersion) + } + for _, orphan := range p.Orphaned { + fmt.Fprintf(&b, " orphaned %-9s %-*s registered %s%s\n", + orphan.Kind, nameWidth, orphan.Name, + joinVersions(orphan.Versions), retiredMarker(orphan)) + } + } + + b.WriteString("\n") + if audit.BehindTotal == 0 && audit.OrphanedTotal == 0 { + fmt.Fprintf(&b, "%d project(s) checked; every registered name matches the corpus.", + len(audit.Projects)) + return b.String() + } + fmt.Fprintf(&b, + "%d of %d project(s) carry findings: %d name(s) behind, %d orphaned.\n"+ + "A behind name is adopted by the next `docket run activate` in that "+ + "project; an orphaned one is retired with `docket workflow deprecate`.", + dirty, len(audit.Projects), audit.BehindTotal, audit.OrphanedTotal) + return b.String() +} + +// describeAuditedProject names a project the way `project list` does — the +// three columns that resolve on `--project` — so the two verbs' output can be +// read against each other. +func describeAuditedProject(p engine.ProjectRegistryAudit) string { + name := p.Project + if name == "" { + name = p.Identity + } + if name == "" { + name = fmt.Sprintf("project %d", p.ProjectID) + } + if p.Prefix != "" { + return fmt.Sprintf("%s (%s)", name, p.Prefix) + } + return name +} + +func joinVersions(versions []int) string { + out := make([]string, 0, len(versions)) + for _, v := range versions { + out = append(out, fmt.Sprintf("@%d", v)) + } + return strings.Join(out, " ") +} + +// retiredMarker distinguishes the orphan an operator has already dealt with +// from the one still binding. Retired orphans stay listed for `workflow list +// --orphans`' reason — a cleanup pass has to be able to see its own work. +func retiredMarker(orphan engine.RegistryOrphan) string { + if orphan.Retired { + return " [all deprecated]" + } + return "" +} + +func init() { + registryAuditCmd.Flags().String("project", "", + "Audit one project (its PREFIX, NAME, IDENTITY, or id) instead of every project") + registryCmd.AddCommand(registryAuditCmd) + rootCmd.AddCommand(registryCmd) +} diff --git a/internal/cli/registry_audit_test.go b/internal/cli/registry_audit_test.go new file mode 100644 index 00000000..5fd69489 --- /dev/null +++ b/internal/cli/registry_audit_test.go @@ -0,0 +1,252 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" + "github.com/spf13/cobra" +) + +// DKT-614: `docket registry audit` — the store-wide comparison of every +// project's registries against the shared corpus. +// +// The state it exists for: eleven projects sharing one store, seven of them +// several versions behind the corpus and missing its newest names entirely, +// and no verb anywhere that would say so without cd'ing into each checkout. + +const auditInvestigationV8 = ` +[pipeline] +name = "investigation" +version = 8 +[[step]] +name = "look" +after = [] +executor = "w" +emits = "out" +` + +const auditFindingsSchema = `{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "findings", + "type": "object", + "properties": { "risk": { "type": "string" } } +}` + +// auditCorpusRoot points DOCKET_PATH at a store whose instance config holds +// exactly the files named, keyed by their path under `config/`. +// +// THE TEMP DIR COMES FIRST, BEFORE t.Setenv — t.TempDir() reads TMPDIR, and a +// test that rewrote the environment before taking its directory would get a +// path that does not exist. +func auditCorpusRoot(t *testing.T, files map[string]string) { + t.Helper() + docketDir := filepath.Join(t.TempDir(), ".docket") + for rel, body := range files { + path := filepath.Join(docketDir, "config", rel) + testsupport.Must(t, os.MkdirAll(filepath.Dir(path), 0o755), + "creating %s", filepath.Dir(path)) + testsupport.Must(t, os.WriteFile(path, []byte(body), 0o644), "writing %s", rel) + } + t.Setenv("DOCKET_PATH", docketDir) +} + +// auditStandardCorpus is the corpus these tests compare against: one workflow +// at version 8, one schema at version 2. +func auditStandardCorpus(t *testing.T) { + t.Helper() + auditCorpusRoot(t, map[string]string{ + "workflows/investigation.toml": auditInvestigationV8, + "schemas/findings@2.json": auditFindingsSchema, + }) +} + +func auditWorkflowRow(t *testing.T, conn *sql.DB, projectID int, name string, version int) { + t.Helper() + body := "bytes for " + name + _, _, err := db.InsertWorkflow(conn, &model.Workflow{ + ProjectID: projectID, Name: name, Version: version, + SourcePath: "/nowhere/" + name + ".toml", SourceSHA256: workflow.SHA256([]byte(body)), + Body: body, Parsed: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "registering %s@%d: %v", name, version, err) +} + +func auditSchemaRow(t *testing.T, conn *sql.DB, projectID int, name string, version int) { + t.Helper() + _, _, err := db.InsertSchema(conn, &model.Schema{ + ProjectID: projectID, Name: name, Version: version, + SourcePath: "/nowhere/" + name + ".json", + SourceSHA256: workflow.SHA256([]byte(auditFindingsSchema)), + Body: auditFindingsSchema, Ordered: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "registering schema %s@%d: %v", name, version, err) +} + +func registryAuditCmdWithDB(conn *sql.DB) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("project", "", "") + return cmd +} + +// runRegistryAuditJSON runs the verb and returns both renders: the JSON +// payload a consumer parses and the text an operator reads. +func runRegistryAuditJSON(t *testing.T, cmd *cobra.Command) (engine.RegistryAudit, string) { + t.Helper() + + w, buf := bufWriter(true) + testsupport.Must(t, runRegistryAudit(cmd, nil, w), "registry audit --json") + var envelope struct { + Data engine.RegistryAudit `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + + w, human := bufWriter(false) + testsupport.Must(t, runRegistryAudit(cmd, nil, w), "registry audit") + return envelope.Data, human.String() +} + +// TestRegistryAuditNamesEveryProjectsDrift is DKT-614's acceptance criterion: +// ONE command, every project in the store, BOTH registries — the behind names +// and the orphaned ones — with no cd and no sqlite. +func TestRegistryAuditNamesEveryProjectsDrift(t *testing.T) { + conn := newTestDB(t) + auditStandardCorpus(t) + + // The first identity to register claims the default row, so both projects + // are named rather than assuming one of them is the default. + stale, err := db.EnsureProject(conn, "/repo/stale.git", "stale.git", model.NowMS()) + testsupport.Must(t, err, "creating the stale project: %v", err) + current, err := db.EnsureProject(conn, "/repo/current.git", "current.git", model.NowMS()) + testsupport.Must(t, err, "creating the current project: %v", err) + + auditWorkflowRow(t, conn, stale, "investigation", 4) + auditWorkflowRow(t, conn, stale, "security-load-bearing", 12) + auditSchemaRow(t, conn, stale, "findings", 1) + auditWorkflowRow(t, conn, current, "investigation", 8) + auditSchemaRow(t, conn, current, "findings", 2) + + audit, human := runRegistryAuditJSON(t, registryAuditCmdWithDB(conn)) + + if len(audit.Projects) != 2 { + t.Fatalf("the audit covers %d project(s), want both: %+v", + len(audit.Projects), audit.Projects) + } + if len(audit.Roots) == 0 { + t.Error("the payload names no scanned roots, so a consumer cannot tell " + + "where `current` was read from") + } + if audit.BehindTotal != 2 || audit.OrphanedTotal != 1 { + t.Errorf("totals = %d behind / %d orphaned, want 2 / 1 (a workflow AND a "+ + "schema behind, one stranded name)", audit.BehindTotal, audit.OrphanedTotal) + } + + // The human render is the whole point of the verb — the operator asked + // because they did not want to read sqlite. + for _, want := range []string{ + "stale.git", "current.git", + "investigation", "findings", "security-load-bearing", + "behind", "orphaned", "up to date", + } { + if !strings.Contains(human, want) { + t.Errorf("the render never mentions %q:\n%s", want, human) + } + } + if !strings.Contains(human, "registered 4, current 8") { + t.Errorf("the render does not state both versions of the lag:\n%s", human) + } +} + +// TestRegistryAuditNarrowsToOneProject: every project is the default, since +// that is the question the verb exists for, but --project resolves the same +// four keys `project list` prints. +func TestRegistryAuditNarrowsToOneProject(t *testing.T) { + conn := newTestDB(t) + auditStandardCorpus(t) + + first, err := db.EnsureProject(conn, "/repo/first.git", "first.git", model.NowMS()) + testsupport.Must(t, err, "creating the first project: %v", err) + other, err := db.EnsureProject(conn, "/repo/other.git", "other.git", model.NowMS()) + testsupport.Must(t, err, "creating the second project: %v", err) + auditWorkflowRow(t, conn, first, "investigation", 4) + auditWorkflowRow(t, conn, other, "investigation", 4) + + cmd := registryAuditCmdWithDB(conn) + testsupport.Must(t, cmd.Flags().Set("project", "other.git"), "setting --project") + + audit, _ := runRegistryAuditJSON(t, cmd) + if len(audit.Projects) != 1 || audit.Projects[0].ProjectID != other { + t.Fatalf("--project audited %+v, want only other.git", audit.Projects) + } +} + +// TestRegistryAuditRefusesAnUnknownProject: the flag's whole purpose is to say +// WHICH project, so a ref that names none is refused rather than silently +// widened back to the store. +func TestRegistryAuditRefusesAnUnknownProject(t *testing.T) { + conn := newTestDB(t) + auditStandardCorpus(t) + + cmd := registryAuditCmdWithDB(conn) + testsupport.Must(t, cmd.Flags().Set("project", "no-such-repo"), "setting --project") + + w, _ := bufWriter(true) + err := runRegistryAudit(cmd, nil, w) + if err == nil { + t.Fatal("an unknown --project was accepted") + } + if got := codeOf(t, err); got != output.ErrNotFound { + t.Errorf("error code = %q, want %q", got, output.ErrNotFound) + } +} + +// TestRegistryAuditRefusesWithNothingToScan: with no instance-config root, +// every registered name in the store trivially has no file declaring it and no +// current version to be behind. Rendering that as a report would be the verb +// inventing its entire result out of having looked nowhere — the refusal +// `workflow list --orphans` already makes, with the same code. +func TestRegistryAuditRefusesWithNothingToScan(t *testing.T) { + conn := newTestDB(t) + docketDir := filepath.Join(t.TempDir(), ".docket") + testsupport.Must(t, os.MkdirAll(docketDir, 0o755), "creating the store directory") + t.Setenv("DOCKET_PATH", docketDir) + auditWorkflowRow(t, conn, db.DefaultProjectID, "investigation", 4) + + w, _ := bufWriter(true) + err := runRegistryAudit(registryAuditCmdWithDB(conn), nil, w) + if err == nil { + t.Fatal("the audit reported drift with no instance-config root to scan; " + + "every registration in the store would qualify") + } + if got := codeOf(t, err); got != output.ErrValidation { + t.Errorf("error code = %q, want %q", got, output.ErrValidation) + } +} + +// TestRegistryAuditReportsACleanStore: an empty result is a CLEAN BILL OF +// HEALTH, not "nothing found" — the two readings are opposite, and the render +// must not leave an operator guessing which one they got. +func TestRegistryAuditReportsACleanStore(t *testing.T) { + conn := newTestDB(t) + auditStandardCorpus(t) + auditWorkflowRow(t, conn, db.DefaultProjectID, "investigation", 8) + + audit, human := runRegistryAuditJSON(t, registryAuditCmdWithDB(conn)) + if audit.BehindTotal != 0 || audit.OrphanedTotal != 0 { + t.Fatalf("a matching registry reports findings: %+v", audit.Projects) + } + if !strings.Contains(human, "matches the corpus") { + t.Errorf("the render does not say the store is clean:\n%s", human) + } +} diff --git a/internal/cli/registry_fanout.go b/internal/cli/registry_fanout.go new file mode 100644 index 00000000..4a3553dd --- /dev/null +++ b/internal/cli/registry_fanout.go @@ -0,0 +1,344 @@ +package cli + +import ( + "database/sql" + "errors" + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/spf13/cobra" +) + +// CROSS-PROJECT TARGETING for the three registry-WRITING verbs (DKT-615). +// +// `workflow register`, `workflow deprecate`, and `schema register` write rows +// that are project-scoped — UNIQUE(project_id, name, version) — but each of +// them picked its project from getProjectID(cmd) and nowhere else. So +// "retire this orphaned name" and "register the current corpus" meant one +// invocation per project, run from inside that project's checkout, with no +// verb anywhere that would say whether the other twelve had been done. A +// deprecate run once from one repository retired the version in project 2 and +// left the identical row active in eleven others, silently. +// +// `docket registry audit` (DKT-614) is the READ half of the same gap and is +// the shape followed here: resolve one project or loop every project, produce +// ONE report, and name the project each outcome belongs to. +// +// THREE RULES the fan-out does not get to bend: +// +// 1. THE DEFAULT IS UNCHANGED. With neither flag the verb runs exactly the +// code it ran before, against the cwd's project, and emits exactly the +// payload it emitted before. A consumer parsing `data` as a workflow row +// keeps parsing a workflow row. +// +// 2. EACH PROJECT IS JUDGED ON ITS OWN. Idempotency, conflict, and +// not-registered are decided per project by the same taxonomy the +// single-project path uses (workflowErr / schemaErr). One project's +// CONFLICT never cancels another project's success, and never hides it: +// every target appears in the report with its own outcome. +// +// 3. A PARTIAL FAILURE IS STILL A FAILURE. The report is written to stdout — +// a machine consumer needs the per-project detail precisely when something +// went wrong — and the process then exits non-zero without a second +// envelope. See reportedFailure in root.go. + +// registryOutcome is the closed vocabulary of what happened in one project. +// +// It is a WORD rather than a bool because the interesting results are not +// success-vs-failure: "unchanged" and "registered" are both successes that mean +// different things to an operator sweeping a corpus, and "already-deprecated" +// is a refusal that means the work was already done. +const ( + outcomeRegistered = "registered" + outcomeUnchanged = "unchanged" + outcomeDeprecated = "deprecated" + outcomeRestored = "restored" + outcomeAlreadyBinding = "already-binding" + outcomeAlreadyDeprecated = "already-deprecated" + outcomeConflict = "conflict" + outcomeNotFound = "not-registered" + outcomeInvalid = "invalid" + outcomeError = "error" +) + +// registryFanoutResult is one project's outcome. +type registryFanoutResult struct { + ProjectID int `json:"project_id"` + Project string `json:"project"` + Identity string `json:"identity,omitempty"` + Prefix string `json:"prefix,omitempty"` + // Outcome is one of the constants above. + Outcome string `json:"outcome"` + // Ref is the name@version as the store holds it, when a row was reached. + Ref string `json:"ref,omitempty"` + // Detail carries the failure's own message, verbatim, so the report says + // WHY this project differed rather than only that it did. + Detail string `json:"detail,omitempty"` + // Code is the error code this project's outcome would have exited with on + // its own. Empty on success; its presence is what makes a result a failure. + Code output.ErrorCode `json:"code,omitempty"` +} + +func (r registryFanoutResult) failed() bool { return r.Code != "" } + +// registryFanoutReport is the whole invocation's result. +// +// NOT an output.Collection: the v2 collection envelope replaces `data` with +// {items, total, truncated}, which would drop the counts and the scope — and +// "3 of 13 projects failed" is the line the operator ran the command to read. +// `registry audit` declines the collection envelope for the same reason. +type registryFanoutReport struct { + // Operation is the verb, as typed: "workflow register", "schema register", + // "workflow deprecate", "workflow restore". + Operation string `json:"operation"` + // Subject is what was operated on — the ref, or the source path. + Subject string `json:"subject"` + // Scope is "project" or "all-projects". + Scope string `json:"scope"` + Results []registryFanoutResult `json:"results"` + Succeeded int `json:"succeeded"` + Failed int `json:"failed"` +} + +const ( + scopeOneProject = "project" + scopeAllProjects = "all-projects" +) + +// addRegistryTargetFlags declares the two targeting flags on a registry-writing +// verb. action is the verb's own word ("Register", "Retire") so each help text +// reads as a sentence about that verb. +func addRegistryTargetFlags(cmd *cobra.Command, action string) { + cmd.Flags().String("project", "", fmt.Sprintf( + "%s in this project instead of the one the working directory resolves to "+ + "(its PREFIX, NAME, IDENTITY, or row id)", action)) + cmd.Flags().Bool("all-projects", false, fmt.Sprintf( + "%s in EVERY project in the store, reporting each project's own outcome", action)) + cmd.MarkFlagsMutuallyExclusive("project", "all-projects") +} + +// resolveRegistryTargets answers which projects this invocation writes to. +// +// The bool is FANNED-OUT, not "found something": false means neither flag was +// given and the caller must take its original single-project path, ambient +// project and original payload shape included. It is a separate return rather +// than a nil slice because "no flags" and "an empty store" are different +// answers and only one of them is normal. +// +// The mutual exclusion is checked HERE as well as by cobra's +// MarkFlagsMutuallyExclusive: cobra enforces it during flag parsing, which a +// caller constructing a command directly (every unit test in this package) +// never runs. +func resolveRegistryTargets( + cmd *cobra.Command, conn *sql.DB, +) (targets []*model.Project, fannedOut bool, err error) { + ref, _ := cmd.Flags().GetString("project") + ref = strings.TrimSpace(ref) + all, _ := cmd.Flags().GetBool("all-projects") + + switch { + case ref != "" && all: + return nil, false, cmdErr(fmt.Errorf( + "--project and --all-projects both name the targets and disagree: "+ + "pass one project's ref, or --all-projects for every project in "+ + "the store"), output.ErrValidation) + + case ref != "": + target, err := resolveProjectRef(conn, ref) + if err != nil { + return nil, false, err + } + return []*model.Project{target}, true, nil + + case all: + projects, err := db.ListProjects(conn) + if err != nil { + return nil, false, cmdErr(fmt.Errorf("listing projects: %w", err), + output.ErrGeneral) + } + if len(projects) == 0 { + // Unreachable through the ordinary store — the default row is + // created at initialization — but a fan-out that silently reported + // zero targets would look like a success that wrote nothing. + return nil, false, cmdErr(fmt.Errorf( + "--all-projects found no projects in this store"), output.ErrNotFound) + } + return projects, true, nil + + default: + return nil, false, nil + } +} + +// fanoutScope names the scope a resolved target set represents. +func fanoutScope(cmd *cobra.Command) string { + if all, _ := cmd.Flags().GetBool("all-projects"); all { + return scopeAllProjects + } + return scopeOneProject +} + +// registryFailureResult turns one project's error into its row in the report, +// using the verb's OWN error taxonomy (workflowErr / schemaErr) so a fanned-out +// refusal is classified exactly as the single-project refusal would be. +func registryFailureResult( + p *model.Project, err error, classify func(error) error, +) registryFanoutResult { + code := output.ErrGeneral + var ce *CmdError + if errors.As(classify(err), &ce) { + code = ce.Code + } + + outcome := outcomeError + switch { + case errors.Is(err, db.ErrWorkflowAlreadyDeprecated): + // The ONE sentinel workflowErr does not classify, because the + // single-project path answers it before reaching that mapper. Retiring + // twice is a CONFLICT there (db.ErrWorkflowAlreadyDeprecated's own + // header: the second caller's "I am taking this out of service" is + // wrong), and it has to be a CONFLICT here too — a fan-out that + // downgraded it would make the same store answer differently depending + // on which flag was passed. + outcome, code = outcomeAlreadyDeprecated, output.ErrConflict + case code == output.ErrConflict: + outcome = outcomeConflict + case code == output.ErrNotFound: + outcome = outcomeNotFound + case code == output.ErrValidation: + outcome = outcomeInvalid + } + + return registryFanoutResult{ + ProjectID: p.ID, Project: projectDisplayName(p), + Identity: p.Identity, Prefix: p.Prefix, + Outcome: outcome, Detail: err.Error(), Code: code, + } +} + +// registrySuccessResult is the success counterpart, so both halves of a loop +// build the same row shape from the same place. +func registrySuccessResult(p *model.Project, outcome, ref string) registryFanoutResult { + return registryFanoutResult{ + ProjectID: p.ID, Project: projectDisplayName(p), + Identity: p.Identity, Prefix: p.Prefix, + Outcome: outcome, Ref: ref, + } +} + +// finishRegistryFanout writes the report and decides the invocation's fate. +// +// THE REPORT IS ALWAYS WRITTEN, failure included: the per-project detail is +// what the caller needs most when something went wrong, and an error envelope +// carrying one flattened string would throw away twelve projects' outcomes to +// describe one. +func finishRegistryFanout(w *output.Writer, report *registryFanoutReport) error { + for _, r := range report.Results { + if r.failed() { + report.Failed++ + continue + } + report.Succeeded++ + } + + var message string + if !w.JSONMode { + message = renderRegistryFanout(report) + } + w.Success(report, message) + + if report.Failed == 0 { + return nil + } + return &reportedFailure{Code: fanoutExitCode(report)} +} + +// fanoutExitCode picks the code the process exits with when some targets +// failed. +// +// ONE CODE WHEN THEY AGREE, GENERAL_ERROR WHEN THEY DO NOT. The agreeing case +// is the one that matters: `--project X` targets exactly one project, so a +// conflict there exits 4 exactly as the same command without the flag would — +// the flag changes which project is written, not what a conflict costs. Mixed +// failures have no single honest code, and inventing a precedence between +// CONFLICT and NOT_FOUND would tell a script something the store never said; +// exit 1 sends it to the report, which holds both. +func fanoutExitCode(report *registryFanoutReport) output.ErrorCode { + var code output.ErrorCode + for _, r := range report.Results { + if !r.failed() { + continue + } + if code == "" { + code = r.Code + continue + } + if code != r.Code { + return output.ErrGeneral + } + } + if code == "" { + return output.ErrGeneral + } + return code +} + +// renderRegistryFanout is the human report: one line per project, aligned, then +// the count. +func renderRegistryFanout(report *registryFanoutReport) string { + var b strings.Builder + + scope := "every project in the store" + if report.Scope == scopeOneProject { + scope = "1 project" + } + fmt.Fprintf(&b, "%s %s — %s (%d target(s))\n\n", + report.Operation, report.Subject, scope, len(report.Results)) + + nameWidth, outcomeWidth := 4, 4 + for _, r := range report.Results { + nameWidth = max(nameWidth, len(describeFanoutProject(r))) + outcomeWidth = max(outcomeWidth, len(r.Outcome)) + } + + for _, r := range report.Results { + fmt.Fprintf(&b, " %-*s %-*s %s\n", + nameWidth, describeFanoutProject(r), + outcomeWidth, r.Outcome, + fanoutDetail(r)) + } + + b.WriteString("\n") + if report.Failed == 0 { + fmt.Fprintf(&b, "%d project(s) succeeded.", report.Succeeded) + return b.String() + } + fmt.Fprintf(&b, + "%d project(s) succeeded, %d failed. Nothing was rolled back: each "+ + "project's registry is its own, and the projects that succeeded "+ + "are done.", report.Succeeded, report.Failed) + return b.String() +} + +// describeFanoutProject names a project the way `project list` and `registry +// audit` do, so the three outputs can be read against each other. +func describeFanoutProject(r registryFanoutResult) string { + name := r.Project + if name == "" { + name = fmt.Sprintf("project %d", r.ProjectID) + } + if r.Prefix != "" { + return fmt.Sprintf("%s (%s)", name, r.Prefix) + } + return name +} + +func fanoutDetail(r registryFanoutResult) string { + if r.Detail != "" { + return r.Detail + } + return r.Ref +} diff --git a/internal/cli/registry_fanout_test.go b/internal/cli/registry_fanout_test.go new file mode 100644 index 00000000..24b8535d --- /dev/null +++ b/internal/cli/registry_fanout_test.go @@ -0,0 +1,514 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "errors" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// DKT-615: cross-project targeting for the three registry-WRITING verbs. +// +// The state it exists for, measured on 2026-08-24: `workflow deprecate +// release@7` run once from one checkout retired the version in project 2 and +// left the identical row active in ten others, with nothing in the output +// saying so. Registration had the same shape. The criterion is one invocation, +// every project, and a report that says which project got which outcome — +// because the failure mode being fixed is a partial sweep that LOOKS complete. + +// fanoutCmd builds a command carrying the targeting flags, since a test wiring +// a bare command never runs cobra's flag parsing (and therefore never runs +// MarkFlagsMutuallyExclusive either — which is why the refusal is also checked +// inside resolveRegistryTargets). +func fanoutCmd(conn *sql.DB, project string, all bool) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("project", project, "") + cmd.Flags().Bool("all-projects", all, "") + cmd.Flags().Bool("restore", false, "") + return cmd +} + +// threeProjects returns three project rows. The FIRST EnsureProject claims the +// default row, so all three are named rather than one of them being assumed. +func threeProjects(t *testing.T, conn *sql.DB) (int, int, int) { + t.Helper() + one, err := db.EnsureProject(conn, "/repo/one.git", "one.git", model.NowMS()) + testsupport.Must(t, err, "creating one.git: %v", err) + two, err := db.EnsureProject(conn, "/repo/two.git", "two.git", model.NowMS()) + testsupport.Must(t, err, "creating two.git: %v", err) + three, err := db.EnsureProject(conn, "/repo/three.git", "three.git", model.NowMS()) + testsupport.Must(t, err, "creating three.git: %v", err) + return one, two, three +} + +// fanoutReportOf parses the report out of the JSON envelope. It asserts the +// payload is the REPORT and not a single row: a caller that passed a targeting +// flag is owed every project's outcome. +func fanoutReportOf(t *testing.T, raw []byte) registryFanoutReport { + t.Helper() + var envelope struct { + Data registryFanoutReport `json:"data"` + } + if err := json.Unmarshal(raw, &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, raw) + } + if len(envelope.Data.Results) == 0 { + t.Fatalf("the report names no per-project results:\n%s", raw) + } + return envelope.Data +} + +// outcomeIn finds one project's row in the report. +func outcomeIn(t *testing.T, report registryFanoutReport, projectID int) registryFanoutResult { + t.Helper() + for _, r := range report.Results { + if r.ProjectID == projectID { + return r + } + } + t.Fatalf("project %d has no row in the report: %+v", projectID, report.Results) + return registryFanoutResult{} +} + +// reportedCodeOf asserts a partial failure surfaced as a reportedFailure — the +// report is already on stdout and must not be replaced by a single error +// envelope describing whichever failure was picked to stand for the rest. +func reportedCodeOf(t *testing.T, err error) output.ErrorCode { + t.Helper() + var rf *reportedFailure + if !errors.As(err, &rf) { + t.Fatalf("error %v is not a *reportedFailure, so the per-project report "+ + "was replaced by one flattened envelope", err) + return "" + } + return rf.Code +} + +func writeWorkflowFile(t *testing.T, src string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "wf.toml") + testsupport.Must(t, os.WriteFile(path, []byte(src), 0o644), + "writing the definition") + return path +} + +// registerAcross runs `workflow register` with the targeting flags set. +func registerAcross( + t *testing.T, conn *sql.DB, src, project string, all bool, +) (registryFanoutReport, error) { + t.Helper() + w, buf := bufWriter(true) + err := runWorkflowRegister( + fanoutCmd(conn, project, all), []string{writeWorkflowFile(t, src)}, w) + return fanoutReportOf(t, buf.Bytes()), err +} + +// TestWorkflowRegisterAllProjectsLandsInEveryRegistry is the acceptance +// criterion for register: ONE invocation, a row in every project. +func TestWorkflowRegisterAllProjectsLandsInEveryRegistry(t *testing.T) { + conn := newTestDB(t) + one, two, three := threeProjects(t, conn) + + report, err := registerAcross(t, conn, minimalWorkflow, "", true) + testsupport.Must(t, err, "register --all-projects: %v", err) + + if report.Scope != scopeAllProjects || report.Succeeded != 3 || report.Failed != 0 { + t.Fatalf("report = %+v, want 3 succeeded in scope all-projects", report) + } + for _, id := range []int{one, two, three} { + if got := outcomeIn(t, report, id).Outcome; got != outcomeRegistered { + t.Errorf("project %d reported %q, want %q", id, got, outcomeRegistered) + } + // The report is a claim about the store; the store is what settles it. + if _, err := db.GetWorkflow(conn, id, "unit", 1); err != nil { + t.Errorf("project %d has no unit@1 row after --all-projects: %v", id, err) + } + } +} + +// TestWorkflowRegisterAllProjectsIsIdempotentPerProject: re-registering the +// same bytes is a no-op success in each project, exactly as it is for one. +func TestWorkflowRegisterAllProjectsIsIdempotentPerProject(t *testing.T) { + conn := newTestDB(t) + one, _, _ := threeProjects(t, conn) + + _, err := registerAcross(t, conn, minimalWorkflow, "", true) + testsupport.Must(t, err, "first register: %v", err) + + report, err := registerAcross(t, conn, minimalWorkflow, "", true) + testsupport.Must(t, err, "second register: %v", err) + + if report.Failed != 0 || report.Succeeded != 3 { + t.Fatalf("re-registering identical bytes failed somewhere: %+v", report) + } + if got := outcomeIn(t, report, one).Outcome; got != outcomeUnchanged { + t.Errorf("outcome = %q, want %q", got, outcomeUnchanged) + } +} + +// TestWorkflowRegisterOneProjectsConflictDoesNotSwallowAnothersSuccess is the +// criterion's sharpest clause. A conflict is real, is reported against the +// project that has it, does not stop the sweep, and still costs the process a +// non-zero exit. +func TestWorkflowRegisterOneProjectsConflictDoesNotSwallowAnothersSuccess(t *testing.T) { + conn := newTestDB(t) + one, two, three := threeProjects(t, conn) + + // DIFFERENT bytes at the same name@version in one project only. + _, _, err := db.InsertWorkflow(conn, &model.Workflow{ + ProjectID: two, Name: "unit", Version: 1, + SourceSHA256: "not-the-same-hash", Body: "other bytes", Parsed: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "seeding the conflicting row: %v", err) + + report, err := registerAcross(t, conn, minimalWorkflow, "", true) + if err == nil { + t.Fatal("a conflicting project exited 0") + } + if got := reportedCodeOf(t, err); got != output.ErrConflict { + t.Errorf("exit code = %q, want %q: every failure agreed on CONFLICT", + got, output.ErrConflict) + } + + if report.Succeeded != 2 || report.Failed != 1 { + t.Fatalf("report = %+v, want 2 succeeded / 1 failed", report) + } + conflicted := outcomeIn(t, report, two) + if conflicted.Outcome != outcomeConflict || conflicted.Code != output.ErrConflict { + t.Errorf("the conflicting project reported %+v, want a CONFLICT", conflicted) + } + if conflicted.Detail == "" { + t.Error("the conflict carries no detail, so the report does not say WHY") + } + for _, id := range []int{one, three} { + if got := outcomeIn(t, report, id).Outcome; got != outcomeRegistered { + t.Errorf("project %d reported %q; one project's conflict swallowed "+ + "another's success", id, got) + } + if _, err := db.GetWorkflow(conn, id, "unit", 1); err != nil { + t.Errorf("project %d never got its row: %v", id, err) + } + } +} + +// TestWorkflowRegisterProjectTargetsExactlyOne: --project writes where it is +// told and nowhere else, including not in the ambient project. +func TestWorkflowRegisterProjectTargetsExactlyOne(t *testing.T) { + conn := newTestDB(t) + one, two, _ := threeProjects(t, conn) + + report, err := registerAcross(t, conn, minimalWorkflow, "two.git", false) + testsupport.Must(t, err, "register --project: %v", err) + + if report.Scope != scopeOneProject || len(report.Results) != 1 { + t.Fatalf("report = %+v, want exactly one target", report) + } + if report.Results[0].ProjectID != two { + t.Fatalf("--project wrote to project %d, want %d", report.Results[0].ProjectID, two) + } + if _, err := db.GetWorkflow(conn, one, "unit", 1); !errors.Is(err, db.ErrWorkflowNotFound) { + t.Errorf("--project also wrote to the ambient project: %v", err) + } +} + +// TestWorkflowRegisterValidatesAgainstEachTargetProject is the semantics call +// this issue actually turns on. `payload` resolves in the registry of the +// project being WRITTEN TO, so the same bytes are valid in a project holding +// the schema and invalid in one that does not — and the invalid project is +// refused THERE rather than the whole sweep being decided by the invoking +// project's registry. +func TestWorkflowRegisterValidatesAgainstEachTargetProject(t *testing.T) { + conn := newTestDB(t) + one, two, three := threeProjects(t, conn) + + _, _, err := db.InsertSchema(conn, &model.Schema{ + ProjectID: two, Name: "findings", Version: 1, + SourceSHA256: "sha", Body: findingsSchema, Ordered: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "seeding the schema in two.git: %v", err) + + const needsSchema = ` +[pipeline] +name = "reads" +version = 1 +[[step]] +name = "a" +after = [] +executor = "x" +emits = "k" +payload = "findings@1" +` + + report, err := registerAcross(t, conn, needsSchema, "", true) + if err == nil { + t.Fatal("projects with no findings@1 registered the definition anyway") + } + if got := reportedCodeOf(t, err); got != output.ErrValidation { + t.Errorf("exit code = %q, want %q", got, output.ErrValidation) + } + + if got := outcomeIn(t, report, two).Outcome; got != outcomeRegistered { + t.Errorf("the project holding the schema reported %q, want %q", + got, outcomeRegistered) + } + for _, id := range []int{one, three} { + got := outcomeIn(t, report, id) + if got.Outcome != outcomeInvalid || got.Code != output.ErrValidation { + t.Errorf("project %d reported %+v, want invalid: the schema its "+ + "`payload` names is not in ITS registry", id, got) + } + if _, err := db.GetWorkflow(conn, id, "reads", 1); !errors.Is(err, db.ErrWorkflowNotFound) { + t.Errorf("project %d stored a definition whose payload it cannot "+ + "resolve; it would fail at activation instead: %v", id, err) + } + } +} + +// TestWorkflowDeprecateAllProjectsRetiresEverywhere is DKT-615's originating +// case, verbatim: one deprecate, every project's identical row. +func TestWorkflowDeprecateAllProjectsRetiresEverywhere(t *testing.T) { + conn := newTestDB(t) + one, two, three := threeProjects(t, conn) + for _, id := range []int{one, two, three} { + auditWorkflowRow(t, conn, id, "release", 7) + } + + w, buf := bufWriter(true) + err := runWorkflowDeprecate(fanoutCmd(conn, "", true), []string{"release@7"}, w) + testsupport.Must(t, err, "deprecate --all-projects: %v", err) + + report := fanoutReportOf(t, buf.Bytes()) + if report.Succeeded != 3 || report.Failed != 0 { + t.Fatalf("report = %+v, want 3 retired", report) + } + for _, id := range []int{one, two, three} { + if got := outcomeIn(t, report, id).Outcome; got != outcomeDeprecated { + t.Errorf("project %d reported %q, want %q", id, got, outcomeDeprecated) + } + wf, err := db.GetWorkflow(conn, id, "release", 7) + testsupport.Must(t, err, "reading release@7 back from project %d: %v", id, err) + if !wf.Deprecated() { + t.Errorf("project %d still binds release@7 after --all-projects", id) + } + } +} + +// TestWorkflowDeprecateAllProjectsReportsMixedOutcomes: a sweep across a store +// meets projects in different states, and each one's own rule decides its own +// row — not-registered where nothing was registered, already-deprecated where +// the work was already done, retired where it was not. +func TestWorkflowDeprecateAllProjectsReportsMixedOutcomes(t *testing.T) { + conn := newTestDB(t) + fresh, retired, absent := threeProjects(t, conn) + auditWorkflowRow(t, conn, fresh, "release", 7) + auditWorkflowRow(t, conn, retired, "release", 7) + _, err := db.DeprecateWorkflow(conn, retired, "release", 7, model.NowMS()) + testsupport.Must(t, err, "pre-retiring in one project: %v", err) + + w, buf := bufWriter(true) + err = runWorkflowDeprecate(fanoutCmd(conn, "", true), []string{"release@7"}, w) + if err == nil { + t.Fatal("a sweep meeting two refusals exited 0") + } + // The two failures disagree — CONFLICT and NOT_FOUND — so no single code is + // honest and the process exits GENERAL_ERROR, pointing at the report. + if got := reportedCodeOf(t, err); got != output.ErrGeneral { + t.Errorf("exit code = %q, want %q for mixed failure codes", + got, output.ErrGeneral) + } + + report := fanoutReportOf(t, buf.Bytes()) + if report.Succeeded != 1 || report.Failed != 2 { + t.Fatalf("report = %+v, want 1 retired / 2 refused", report) + } + if got := outcomeIn(t, report, fresh).Outcome; got != outcomeDeprecated { + t.Errorf("the fresh project reported %q, want %q", got, outcomeDeprecated) + } + if got := outcomeIn(t, report, retired); got.Outcome != outcomeAlreadyDeprecated || + got.Code != output.ErrConflict { + t.Errorf("the already-retired project reported %+v, want already-deprecated "+ + "carrying CONFLICT — the single-project rule, applied there", got) + } + if got := outcomeIn(t, report, absent); got.Outcome != outcomeNotFound || + got.Code != output.ErrNotFound { + t.Errorf("the project that never registered it reported %+v, want "+ + "not-registered carrying NOT_FOUND", got) + } +} + +// TestWorkflowRestoreAllProjectsKeepsTheIdempotencyAsymmetry: restore is +// idempotent where deprecate is not, and the report shows that rather than +// smoothing it into one word. +func TestWorkflowRestoreAllProjectsKeepsTheIdempotencyAsymmetry(t *testing.T) { + conn := newTestDB(t) + binding, retired, _ := threeProjects(t, conn) + auditWorkflowRow(t, conn, binding, "release", 7) + auditWorkflowRow(t, conn, retired, "release", 7) + _, err := db.DeprecateWorkflow(conn, retired, "release", 7, model.NowMS()) + testsupport.Must(t, err, "retiring in one project: %v", err) + + cmd := fanoutCmd(conn, "", true) + testsupport.Must(t, cmd.Flags().Set("restore", "true"), "setting --restore") + + w, buf := bufWriter(true) + // The third project never registered it, so the sweep still exits non-zero. + if err := runWorkflowDeprecate(cmd, []string{"release@7"}, w); err == nil { + t.Fatal("a restore over a project with no such row exited 0") + } + + report := fanoutReportOf(t, buf.Bytes()) + if got := outcomeIn(t, report, retired).Outcome; got != outcomeRestored { + t.Errorf("the retired project reported %q, want %q", got, outcomeRestored) + } + if got := outcomeIn(t, report, binding).Outcome; got != outcomeAlreadyBinding { + t.Errorf("the never-retired project reported %q, want %q", + got, outcomeAlreadyBinding) + } + wf, err := db.GetWorkflow(conn, retired, "release", 7) + testsupport.Must(t, err, "reading release@7 back: %v", err) + if wf.Deprecated() { + t.Error("the retired project's row is still retired after --restore") + } +} + +// TestSchemaRegisterAllProjectsLandsInEveryRegistry, and its conflict half: a +// schema's frozen-bytes rule is decided per project too. +func TestSchemaRegisterAllProjectsLandsInEveryRegistry(t *testing.T) { + conn := newTestDB(t) + one, two, three := threeProjects(t, conn) + + // Different bytes at findings@1 in one project only. + _, _, err := db.InsertSchema(conn, &model.Schema{ + ProjectID: two, Name: "findings", Version: 1, + SourceSHA256: "not-the-same-hash", Body: `{"type":"object"}`, Ordered: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "seeding the conflicting schema: %v", err) + + w, buf := bufWriter(true) + err = runSchemaRegister(fanoutCmd(conn, "", true), + []string{"findings@1", writeSchema(t, findingsSchema)}, w) + if err == nil { + t.Fatal("a conflicting project exited 0") + } + if got := reportedCodeOf(t, err); got != output.ErrConflict { + t.Errorf("exit code = %q, want %q", got, output.ErrConflict) + } + + report := fanoutReportOf(t, buf.Bytes()) + if report.Succeeded != 2 || report.Failed != 1 { + t.Fatalf("report = %+v, want 2 registered / 1 conflict", report) + } + if got := outcomeIn(t, report, two).Outcome; got != outcomeConflict { + t.Errorf("the conflicting project reported %q, want %q", got, outcomeConflict) + } + for _, id := range []int{one, three} { + if _, err := db.GetSchema(conn, id, "findings", 1); err != nil { + t.Errorf("project %d never got findings@1: %v", id, err) + } + } +} + +// TestRegistryFanoutRefusesBothFlags: they both name the targets and they +// disagree. Checked in the resolver as well as by cobra, because a caller that +// builds the command directly never runs cobra's parsing. +func TestRegistryFanoutRefusesBothFlags(t *testing.T) { + conn := newTestDB(t) + threeProjects(t, conn) + + w, _ := bufWriter(true) + err := runWorkflowRegister(fanoutCmd(conn, "two.git", true), + []string{writeWorkflowFile(t, minimalWorkflow)}, w) + if err == nil { + t.Fatal("--project and --all-projects were accepted together") + } + if got := codeOf(t, err); got != output.ErrValidation { + t.Errorf("code = %q, want %q", got, output.ErrValidation) + } +} + +// TestRegistryFanoutRefusesAnUnknownProject: the flag exists to say WHICH +// project, so a ref naming none is refused rather than silently falling back to +// the ambient one — which would write to a project the operator did not name. +func TestRegistryFanoutRefusesAnUnknownProject(t *testing.T) { + conn := newTestDB(t) + threeProjects(t, conn) + + w, _ := bufWriter(true) + err := runWorkflowRegister(fanoutCmd(conn, "no-such-repo", false), + []string{writeWorkflowFile(t, minimalWorkflow)}, w) + if err == nil { + t.Fatal("an unknown --project was accepted") + } + if got := codeOf(t, err); got != output.ErrNotFound { + t.Errorf("code = %q, want %q", got, output.ErrNotFound) + } + for _, id := range []int{1, 2, 3} { + if _, err := db.GetWorkflow(conn, id, "unit", 1); !errors.Is(err, db.ErrWorkflowNotFound) { + t.Errorf("a refused --project still wrote to project %d", id) + } + } +} + +// TestRegistryWritesWithoutFlagsKeepTheirPayload: the fan-out is STRICTLY +// ADDITIVE. With neither flag the verbs emit the row they always emitted, not a +// one-element report — a consumer parsing `.data.name` keeps parsing it. +func TestRegistryWritesWithoutFlagsKeepTheirPayload(t *testing.T) { + conn := newTestDB(t) + + w, buf := bufWriter(true) + testsupport.Must(t, runWorkflowRegister(fanoutCmd(conn, "", false), + []string{writeWorkflowFile(t, minimalWorkflow)}, w), "plain register") + + var envelope struct { + Data map[string]any `json:"data"` + } + testsupport.Must(t, json.Unmarshal(buf.Bytes(), &envelope), "unmarshal: %s", buf) + if _, isReport := envelope.Data["results"]; isReport { + t.Fatalf("a flagless register emitted the fan-out report:\n%s", buf) + } + if envelope.Data["name"] != "unit" { + t.Errorf("the flagless payload is not the workflow row:\n%s", buf) + } +} + +// TestRegistryFanoutHumanReportNamesEveryProject: the operator's half. The +// whole complaint behind this issue was not knowing which projects a sweep had +// reached, so every target, its outcome, and the count have to be readable +// without --json. +func TestRegistryFanoutHumanReportNamesEveryProject(t *testing.T) { + conn := newTestDB(t) + _, two, _ := threeProjects(t, conn) + + _, _, err := db.InsertWorkflow(conn, &model.Workflow{ + ProjectID: two, Name: "unit", Version: 1, + SourceSHA256: "not-the-same-hash", Body: "other bytes", Parsed: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "seeding the conflicting row: %v", err) + + w, buf := bufWriter(false) + if err := runWorkflowRegister(fanoutCmd(conn, "", true), + []string{writeWorkflowFile(t, minimalWorkflow)}, w); err == nil { + t.Fatal("the conflicting sweep exited 0") + } + + human := buf.String() + for _, want := range []string{ + "workflow register", "unit@1", + "one.git", "two.git", "three.git", + outcomeRegistered, outcomeConflict, + "2 project(s) succeeded, 1 failed", + } { + if !strings.Contains(human, want) { + t.Errorf("the render never mentions %q:\n%s", want, human) + } + } +} diff --git a/internal/cli/root.go b/internal/cli/root.go index be818dbb..c1f30328 100644 --- a/internal/cli/root.go +++ b/internal/cli/root.go @@ -51,6 +51,24 @@ func cmdErr(err error, code output.ErrorCode) *CmdError { // Execute recognizes it explicitly and never renders it. var errSkipRun = errors.New("docket: handled in pre-run") +// reportedFailure says the command ALREADY WROTE its own structured output and +// the process must now exit non-zero without a second envelope. +// +// It exists for the cross-project registry writes (DKT-615), where the result +// is genuinely plural: eleven projects registered, two conflicted. Both of the +// alternatives lose information a caller needs. Returning a *CmdError would +// replace the per-project report with one flattened sentence about whichever +// failure was picked to represent the rest; returning nil would exit 0 on a +// command that did not do what was asked in two of thirteen places. +// +// Nothing reads its message — Execute recognizes the type and renders nothing — +// but it carries one so an unhandled path is still legible. +type reportedFailure struct{ Code output.ErrorCode } + +func (e *reportedFailure) Error() string { + return "docket: reported in the command's own output" +} + // isGuardCmd reports whether cmd is one of the `docket guard` predicates. // // Walks the parent chain rather than matching names, so a guard added later is @@ -293,7 +311,12 @@ func init() { rootCmd.PersistentFlags().BoolP("quiet", "q", false, "Suppress non-essential output") rootCmd.PersistentFlags().BoolP("watch", "w", false, "Watch for changes and refresh output") rootCmd.PersistentFlags().Duration( - "interval", 2*time.Second, "Poll interval for --watch and --follow (minimum 500ms)") + "interval", 2*time.Second, + // The UNIT IS NAMED because a bare number is the natural thing to type + // and it is refused: `--interval 2000` fails with pflag's "missing unit + // in duration", which tells an operator what is wrong but not what to + // write instead. The example does. + "Poll interval for --watch and --follow, with a unit (e.g. 2s, 500ms; minimum 500ms)") rootCmd.SilenceErrors = true rootCmd.SilenceUsage = true } @@ -431,6 +454,13 @@ func Execute() int { if errors.Is(err, errSkipRun) { return 0 } + // The command wrote its own report — a fanned-out registry write whose + // outcome differed per project. Exit non-zero, render NOTHING: a second + // envelope would make stdout two JSON documents. + var reported *reportedFailure + if errors.As(err, &reported) { + return output.ExitCodeForError(reported.Code) + } raw, _ := rootCmd.PersistentFlags().GetString("json") // An invalid --json value falls back to human error rendering, which // is the correct channel for reporting that the value was invalid. diff --git a/internal/cli/run_activate.go b/internal/cli/run_activate.go index 31c56f2e..bcd60b31 100644 --- a/internal/cli/run_activate.go +++ b/internal/cli/run_activate.go @@ -43,6 +43,18 @@ func newRunActivateCmd() *cobra.Command { Nothing executes. No gate, no action, no command runs during activation; files are read only to pin them by content hash. +Every workflow bound HERE is checked against its own registered source_path: if +the file at that path no longer hashes to the registered source_sha256, the +activation REFUSES and names both hashes, because the run would otherwise bind +and pin bytes that are not the ones at the path an operator reads. Nothing is +re-registered to fix it — that is the install path's job. Three cases warn +rather than refuse: a source that cannot be READ at all (the registered bytes +still reproduce; only their provenance is gone), drift under a binding INHERITED +from an earlier activation (a mid-run edit is deliberately a non-event for a run +already under way), and drift while registration.auto is FALSE, where binding +what is registered rather than what the corpus now says is the whole point of +the setting. + Instance config is scanned from EVERY configured root, in order. With the shared store that is ~/.docket/config/ first, then this checkout's .docket/config/ if it has one; with DOCKET_PATH or a repo-local store it is that store's config/ alone. @@ -111,6 +123,13 @@ type activateResult struct { // where every issue declared its scope carries no key at all rather than an // empty one. ScopeWarnings []engine.ScopeWarning `json:"scope_warnings,omitempty"` + // SourceWarnings is DKT-590's array: bound workflows whose registered + // source file could not be verified against `source_sha256`. Drift on a + // binding this activation MADE refuses instead, so what reaches this key + // is an unreadable source, drift under a binding inherited from an earlier + // activation, or drift while `registration.auto` is false. `omitempty`, so + // the ordinary run whose sources all verify carries no key at all. + SourceWarnings []engine.SourceWarning `json:"source_warnings,omitempty"` // Fences is §7.7 S2's array: the same data the human report renders, // carrying the RAW stored command bytes — encoding/json escapes controls // by contract and the consumer is a program, so quoting on top would @@ -208,6 +227,15 @@ func runRunActivate(cmd *cobra.Command, args []string, w *output.Writer) error { warning.IssueID) } + // DKT-590: a bound workflow whose recorded source could not be verified. + // On the warning channel because the refusing case never gets here — + // activation returns a CONFLICT for drift under a binding it just made — + // so what reaches this loop is provenance that no longer resolves, or an + // edit RA2 deliberately keeps out of a run already under way. + for _, warning := range result.SourceWarnings { + w.Warn("workflow %s: %s", warning.Workflow, warning.Reason) + } + // §7.7 S1: every harvested fenced command, verbatim, with its trust status // — so an operator sees `unmatched` commands BEFORE the run rather than // after. It renders through the escaping renderer (T18): the bytes are @@ -241,6 +269,7 @@ func runRunActivate(cmd *cobra.Command, args []string, w *output.Writer) error { FencesHarvested: result.FencesHarvested, ContextWarnings: result.ContextWarnings, ScopeWarnings: result.ScopeWarnings, + SourceWarnings: result.SourceWarnings, Fences: result.Fences, GatePreflight: result.GatePreflight, HoldPolicy: result.HoldPolicy, diff --git a/internal/cli/run_refresh_scope.go b/internal/cli/run_refresh_scope.go new file mode 100644 index 00000000..a3be2637 --- /dev/null +++ b/internal/cli/run_refresh_scope.go @@ -0,0 +1,142 @@ +package cli + +import ( + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/spf13/cobra" +) + +var runRefreshScopeCmd = &cobra.Command{ + Use: "refresh-scope RUN-N --issue DKT-M --reason R", + Short: "Make an authorized scope widen reach one issue's remaining steps", + Long: `Re-snapshot ONE issue's scope in a live run, from what the issue declares now (DKT-869). + +Activation freezes an issue's title, kind, labels and scope into the run, and +every packet renders from that snapshot — so ` + "`issue edit --scope`" + ` reaches the +scheduler's mutual-exclusion check and NOTHING ELSE. That freeze is correct and +stays. What it cost on RUN-52 was the case where the widen was authorized: the +panel rejected the work as out of scope, the operator agreed and widened it, +and the already-minted ` + "`fix@2`" + ` step still rendered the old scope, so the honest +remedy was unexecutable and the issue was abandoned mid-loop. + +This verb is that case and only that case. It copies ` + "`issues.scope_globs`" + ` — +the column ` + "`issue create|edit --scope`" + ` is the sole writer of — into the run's +snapshot for one issue, and rewrites nothing else in it: the title, kind, +labels, linked pins and description snapshot activation froze are re-encoded +byte for byte, so a mid-run relabel still cannot reroute a step and a mid-run +description edit still cannot reach a packet. + +IT CARRIES NO SCOPE OF ITS OWN, deliberately. There is no --scope here. The +only way to change what this verb will copy is to declare it on the issue, +through the one gate scope widening has always had — so a refresh can never +make real a scope nobody authorized, and a refresh with no widen behind it is +refused with nothing to copy. + +WHAT IT NEVER CHANGES: terminal steps. Their artifacts and their recorded diffs +keep the scope they ran under, and the ` + "`issue-scope-refreshed`" + ` event carries +the old scope, the new one, and the instances the change reaches — so two steps +of one run declaring two different scopes is a dated, attributable fact in the +ledger rather than drift a reader has to infer. + +It REFUSES rather than proceeding when the change could be straddled: + + - while any of the issue's steps is claimed, running, or gated (an executor + holds a packet rendered under the frozen scope, or a diff-shaped gate is + mid-saga over the artifact's paths) + - while a dispatch is open (its manifest was offered under the frozen scope) + - on a run that is planning, done, or abandoned — a planning run snapshots + the current scope at its next activation by itself, and a terminal run's + snapshot is history + - when every step of the issue in this run is terminal — nothing will render + again, so the widened scope reaches the issue's NEXT run + - when the live scope already matches the frozen one: widen it first + +--reason is required. A live run's packets changing what they declare is +something somebody will ask about later, and a trail that shows the scope +moving without saying why is a record that rewrote itself.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runRunRefreshScope(cmd, args[0], getWriter(cmd)) + }, +} + +func runRunRefreshScope(cmd *cobra.Command, ref string, w *output.Writer) error { + conn := getDB(cmd) + + runID, err := model.ParseRunID(ref) + if err != nil { + return cmdErr(err, output.ErrValidation) + } + + issueRef, _ := cmd.Flags().GetString("issue") + if issueRef == "" { + return cmdErr( + fmt.Errorf("--issue is required: the snapshot is frozen per issue, "+ + "so a refresh names the one issue whose remaining steps should "+ + "render the widened scope"), + output.ErrValidation) + } + issueID, err := issueArg(issueRef) + if err != nil { + return err + } + + reason, _ := cmd.Flags().GetString("reason") + if reason == "" { + return cmdErr( + fmt.Errorf("--reason is required to refresh a snapshotted scope; the "+ + "event trail must say why a live run's packets changed what "+ + "they declare"), + output.ErrValidation) + } + + outcome, err := engine.RefreshIssueScopeInRun( + conn, runID, issueID, reason, model.NowMS()) + if err != nil { + return runErr(err) + } + + var message string + if !w.JSONMode { + message = renderRefreshScopeOutcome(outcome) + } + w.Success(outcome, message) + return nil +} + +func renderRefreshScopeOutcome(o *engine.RefreshedScope) string { + var b strings.Builder + fmt.Fprintf(&b, "Refreshed %s's scope in %s:\n %s\n ->\n %s\n", + o.Issue, o.Run, renderScopeList(o.From), renderScopeList(o.To)) + fmt.Fprintf(&b, "%d step(s) will render it and record their diffs over it: %s\n", + len(o.Steps), strings.Join(o.Steps, ", ")) + // The half the operator must not have to infer: what did NOT move. It is + // the same division `run repin` prints, and for the same reason — an + // operator reading only the first half could reasonably think the run's + // completed work had been re-scoped underneath them. + b.WriteString( + "Steps that already recorded keep the scope they ran under (see the " + + "issue-scope-refreshed event).") + return b.String() +} + +// renderScopeList names a scope for the terminal, keeping the undeclared case +// distinct from the declared-empty one exactly as the engine's advisory does. +func renderScopeList(globs []string) string { + if len(globs) == 0 { + return "(no declared scope)" + } + return "[" + strings.Join(globs, ", ") + "]" +} + +func init() { + runRefreshScopeCmd.Flags().String("issue", "", + "The issue whose snapshotted scope is refreshed (required)") + runRefreshScopeCmd.Flags().String("reason", "", + "Why the run's frozen scope is moving (required)") + runCmd.AddCommand(runRefreshScopeCmd) +} diff --git a/internal/cli/run_refresh_scope_test.go b/internal/cli/run_refresh_scope_test.go new file mode 100644 index 00000000..8c32d38b --- /dev/null +++ b/internal/cli/run_refresh_scope_test.go @@ -0,0 +1,195 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-869 at the verb level. The engine half — that the refresh is real, that +// it copies only what the gated writer declared, and that it refuses every +// straddle — is pinned in internal/engine/dkt869_test.go. What is pinned HERE +// is the wiring an operator and a conductor actually meet, on the fixture +// RUN-52 was: an issue with a pending `fix@2` step in an activated run. + +// refreshScope drives the real verb body with a substituted Writer. +func refreshScope( + t *testing.T, conn *sql.DB, runID, issueID int, reason string, jsonMode bool, +) (string, error) { + t.Helper() + cmd := cmdWithDB(conn) + cmd.Flags().String("issue", "", "") + cmd.Flags().String("reason", "", "") + if issueID != 0 { + testsupport.Must(t, cmd.Flags().Set("issue", model.FormatID(issueID)), + "setting --issue") + } + if reason != "" { + testsupport.Must(t, cmd.Flags().Set("reason", reason), "setting --reason") + } + + w, stdout := bufWriter(jsonMode) + err := runRunRefreshScope(cmd, model.FormatRunID(runID), w) + return stdout.String(), err +} + +// TestRunRefreshScopeMakesTheWidenReachTheStep is the acceptance path: the +// operator widens, then refreshes, and the run's snapshot now carries what the +// issue declares. +func TestRunRefreshScopeMakesTheWidenReachTheStep(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, + `["cli/src/command/start.rs"]`, `["cli/src/command/start.rs"]`) + + // The authorized widen, exactly as the conductor ran it. + editWithScope(t, conn, issueID, false, + "cli/src/command/start.rs", "script/install.sh", "makefile") + + stdout, err := refreshScope(t, conn, runID, issueID, "panel agreed to widen", false) + testsupport.Must(t, err, "run refresh-scope: %v", err) + + for _, want := range []string{ + model.FormatID(issueID), model.FormatRunID(runID), + "script/install.sh", "makefile", "fix@2", + // The half an operator must not have to infer. + "already recorded keep the scope they ran under", + } { + if !strings.Contains(stdout, want) { + t.Errorf("stdout does not name %q:\n%s", want, stdout) + } + } + + if got := snapshotScopeJSON(t, conn, runID, issueID); got != + `["cli/src/command/start.rs","script/install.sh","makefile"]` { + t.Errorf("snapshot scope = %s, want the widened declaration", got) + } +} + +// TestRunRefreshScopeJSONEnvelope is the other channel, and the one that +// matters most: RUN-52's widen was typed by a CONDUCTOR. +func TestRunRefreshScopeJSONEnvelope(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, + `["internal/a/**"]`, `["internal/a/**"]`) + editWithScope(t, conn, issueID, false, "internal/a/**", "internal/b/**") + + stdout, err := refreshScope(t, conn, runID, issueID, "authorized", true) + testsupport.Must(t, err, "run refresh-scope: %v", err) + + var envelope struct { + OK bool `json:"ok"` + Data struct { + Run string `json:"run"` + Issue string `json:"issue"` + From []string `json:"from"` + To []string `json:"to"` + Steps []string `json:"steps"` + } `json:"data"` + } + testsupport.Must(t, json.Unmarshal([]byte(stdout), &envelope), "decoding the envelope") + if !envelope.OK { + t.Fatalf("envelope is not ok: %s", stdout) + } + if envelope.Data.Run != model.FormatRunID(runID) || + envelope.Data.Issue != model.FormatID(issueID) { + t.Errorf("data = %+v, want the run and issue named", envelope.Data) + } + if len(envelope.Data.From) != 1 || len(envelope.Data.To) != 2 { + t.Errorf("data = %+v, want both scopes — a caller has to be able to "+ + "report what moved", envelope.Data) + } + if len(envelope.Data.Steps) != 1 || envelope.Data.Steps[0] != "fix@2" { + t.Errorf("data.steps = %v, want the reached instance", envelope.Data.Steps) + } +} + +// TestRunRefreshScopeRequiresItsArguments: both flags are load-bearing and +// neither has a safe default. A guessed issue would move a snapshot nobody +// named; a missing reason leaves a trail that shows the scope changing with +// nothing saying why. +func TestRunRefreshScopeRequiresItsArguments(t *testing.T) { + t.Run("no --issue", func(t *testing.T) { + conn := newTestDB(t) + runID, _ := scopedIssueInLiveRun(t, conn, `["a/**"]`, `["a/**"]`) + _, err := refreshScope(t, conn, runID, 0, "authorized", false) + assertRefreshCode(t, err, output.ErrValidation, "--issue") + }) + + t.Run("no --reason", func(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, `["a/**"]`, `["a/**"]`) + editWithScope(t, conn, issueID, false, "a/**", "b/**") + _, err := refreshScope(t, conn, runID, issueID, "", false) + assertRefreshCode(t, err, output.ErrValidation, "--reason") + + // And nothing was written on the way to the refusal. + if got := snapshotScopeJSON(t, conn, runID, issueID); got != `["a/**"]` { + t.Errorf("snapshot scope = %s after a refused refresh, want it frozen", got) + } + }) +} + +// TestRunRefreshScopeWithoutAWidenIsRefused is the gate, at the verb: the only +// way to change what this verb copies is `issue edit --scope`, so a refresh +// with no widen behind it must fail rather than record a no-op ruling. +func TestRunRefreshScopeWithoutAWidenIsRefused(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, `["a/**"]`, `["a/**"]`) + + _, err := refreshScope(t, conn, runID, issueID, "authorized", false) + assertRefreshCode(t, err, output.ErrConflict, "issue edit") +} + +// TestRunRefreshScopeRefusesAClaimedStep is the straddle refusal reaching the +// operator with the right exit code — a CONFLICT is retryable once the +// executor records, and a caller keying on the code needs to see that. +func TestRunRefreshScopeRefusesAClaimedStep(t *testing.T) { + conn := newTestDB(t) + runID, issueID := scopedIssueInLiveRun(t, conn, `["a/**"]`, `["a/**"]`) + editWithScope(t, conn, issueID, false, "a/**", "b/**") + _, err := conn.Exec(`UPDATE steps SET status = ? WHERE run_id = ?`, + db.StepClaimed, runID) + testsupport.Must(t, err, "claiming the step: %v", err) + + _, refreshErr := refreshScope(t, conn, runID, issueID, "authorized", false) + assertRefreshCode(t, refreshErr, output.ErrConflict, "fix@2") +} + +func assertRefreshCode(t *testing.T, err error, want output.ErrorCode, names string) { + t.Helper() + if err == nil { + t.Fatalf("the refresh was accepted; want %v naming %q", want, names) + } + var ce *CmdError + if !asCmdError(err, &ce) { + t.Fatalf("error %v is not a CmdError", err) + } + if ce.Code != want { + t.Errorf("error code = %v, want %v: %v", ce.Code, want, err) + } + if !strings.Contains(err.Error(), names) { + t.Errorf("the refusal does not name %q:\n%s", names, err) + } +} + +func snapshotScopeJSON(t *testing.T, conn *sql.DB, runID, issueID int) string { + t.Helper() + var blob string + err := conn.QueryRow( + `SELECT issue_snapshot FROM run_issues WHERE run_id = ? AND issue_id = ?`, + runID, issueID).Scan(&blob) + testsupport.Must(t, err, "reading the snapshot: %v", err) + var frozen struct { + Scope []string `json:"scope"` + } + testsupport.Must(t, json.Unmarshal([]byte(blob), &frozen), "decoding the snapshot") + out, err := json.Marshal(frozen.Scope) + testsupport.Must(t, err, "encoding the scope: %v", err) + return string(out) +} diff --git a/internal/cli/run_repin.go b/internal/cli/run_repin.go index 28d4b9fb..ab770489 100644 --- a/internal/cli/run_repin.go +++ b/internal/cli/run_repin.go @@ -39,7 +39,41 @@ It REFUSES rather than proceeding when the transition could be straddled: terminal — nothing remains for the new agreement to govern, so a repin could only rewrite completed steps' history - when a pinned ref no longer resolves at all (NOT_FOUND: there are no - current bytes to adopt; restore the file instead) + current bytes to adopt), UNLESS you retire it with --drop/--drop-unresolvable + and no non-terminal step reads it — see below + +DELETED REFS: --drop and --drop-unresolvable. A corpus commit that DELETES a +contract leaves a pin with no bytes to adopt, and refusing the whole set for it +wedges a run that may have no step left which would ever open that file. + + --drop REF retire this one ref (repeatable) + --drop-unresolvable retire every currently drifted file pin that no + longer resolves and that no pending step reads + +Retiring means: the pin row is removed, so verify-pins stops reporting it and +the remaining steps proceed; a ` + "`run-repinned`" + ` event is recorded for it with a +NULL new_sha256 and ` + "`dropped: true`" + `, so the old sha — the agreement the +completed steps worked under — stays in the trail exactly as an ordinary repin +leaves it. Completed steps' rows, artifacts, and events are untouched here too. + +It is opt-in, never automatic: a NOT_FOUND ref that a NON-TERMINAL step's packet +closure still reaches refuses either way, naming the steps that read it, because +dropping it would only move the wedge to render time. --drop-unresolvable does +not touch refs that resolve to DIFFERENT bytes — those are the ordinary repin — +and neither flag applies to workflow or schema pins, which name registered +objects no packet closure can call unread. + +NEWLY-REQUIRED REFS: the adopted bytes bring their own closure. A corpus edit +can make an adopted contract include a fragment the run never snapshotted; +adopting the contract without the fragment would report success while every +step reading it becomes unrenderable (its packet refuses on the unpinned ref). +So a repin that adopts changed bytes also walks the packet closure those bytes +reach and PINS every file ref the run does not already hold, at its current +disk bytes, in the same transaction — each addition recorded as its own +` + "`run-repinned`" + ` event with a null old_sha256 and ` + "`added: true`" + `, naming what +requires it. A newly-required ref with no bytes on disk refuses the whole +repin up front, naming the ref and its readers: restore the file, or abandon +the run and re-plan. --reason is required. A repin moves the agreement every packet is verified against, and a trail that shows the hashes changing without saying why is a @@ -65,7 +99,12 @@ func runRunRepin(cmd *cobra.Command, ref string, w *output.Writer) error { output.ErrValidation) } - outcome, err := engine.RepinRun(conn, runID, reason, model.NowMS()) + drop, _ := cmd.Flags().GetStringArray("drop") + dropUnresolvable, _ := cmd.Flags().GetBool("drop-unresolvable") + + outcome, err := engine.RepinRunWith(conn, runID, engine.RepinOptions{ + Reason: reason, Drop: drop, DropUnresolvable: dropUnresolvable, + }, model.NowMS()) if err != nil { return runErr(err) } @@ -79,16 +118,30 @@ func runRunRepin(cmd *cobra.Command, ref string, w *output.Writer) error { } func renderRepinOutcome(o *engine.RepinOutcome) string { - if len(o.Repinned) == 0 { + if len(o.Repinned) == 0 && len(o.Dropped) == 0 && len(o.Added) == 0 { return fmt.Sprintf("%s: every pin already matches disk; nothing to repin", o.Run) } var b strings.Builder - fmt.Fprintf(&b, "Repinned %d pin(s) for %s (%d already matched):\n", - len(o.Repinned), o.Run, o.Unchanged) + fmt.Fprintf(&b, "Repinned %d pin(s) for %s (%d added, %d dropped, %d already matched):\n", + len(o.Repinned), o.Run, len(o.Added), len(o.Dropped), o.Unchanged) for _, c := range o.Repinned { fmt.Fprintf(&b, " %-9s %-40s %s -> %s\n", c.Kind, c.Ref, shortSHA(c.OldSHA256), shortSHA(c.NewSHA256)) } + // An added ref renders with the arrow pointing FROM nothing, the mirror of + // the drop below: the run held no bytes for it before, and the adopted + // bytes are why it holds these now. + for _, c := range o.Added { + fmt.Fprintf(&b, " %-9s %-40s (unpinned) -> %s (added: newly required "+ + "by the adopted bytes)\n", c.Kind, c.Ref, shortSHA(c.NewSHA256)) + } + // A dropped ref renders with the arrow pointing at nothing on purpose: the + // operator asked what happened to a pin, and "gone" is the answer, in the + // same column the new hash would have been. + for _, c := range o.Dropped { + fmt.Fprintf(&b, " %-9s %-40s %s -> (dropped: no longer resolves, unread "+ + "by pending steps)\n", c.Kind, c.Ref, shortSHA(c.OldSHA256)) + } b.WriteString( "Completed steps keep their original pins as history (see the run-repinned events).") return strings.TrimRight(b.String(), "\n") @@ -97,5 +150,11 @@ func renderRepinOutcome(o *engine.RepinOutcome) string { func init() { runRepinCmd.Flags().String("reason", "", "Why the recorded agreement is moving (required)") + runRepinCmd.Flags().StringArray("drop", nil, + "Retire this pinned ref, which no longer resolves and no pending step "+ + "reads (repeatable)") + runRepinCmd.Flags().Bool("drop-unresolvable", false, + "Retire every drifted file pin that no longer resolves and no pending "+ + "step reads") runCmd.AddCommand(runRepinCmd) } diff --git a/internal/cli/run_repin_drop_test.go b/internal/cli/run_repin_drop_test.go new file mode 100644 index 00000000..f1a9fc78 --- /dev/null +++ b/internal/cli/run_repin_drop_test.go @@ -0,0 +1,142 @@ +package cli + +import ( + "database/sql" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/pflag" +) + +// DKT-582 at the CLI seam: `--drop` and `--drop-unresolvable` have to be +// PARSED and have to REACH the engine's disposition. The behaviour itself is +// pinned in internal/engine; what can only break here is the wiring. + +// repinViaCLI runs the command's own handler over the command's OWN flag set — +// the definitions init() registers, not a test's re-declaration of them — so a +// renamed flag or a mistyped Get* call fails here. +func repinViaCLI(t *testing.T, conn *sql.DB, ref string, set map[string]string) error { + t.Helper() + cmd := cmdWithDB(conn) + cmd.Flags().AddFlagSet(runRepinCmd.Flags()) + // The flag objects are shared with the package-level command, so each call + // starts from the declared defaults rather than the previous call's values. + resetRepinFlags(t) + t.Cleanup(func() { resetRepinFlags(t) }) + for name, value := range set { + testsupport.Must(t, cmd.Flags().Set(name, value), "--%s=%s", name, value) + } + w, _ := bufWriter(false) + return runRunRepin(cmd, ref, w) +} + +// resetRepinFlags puts every `run repin` flag back to its declared default. +func resetRepinFlags(t *testing.T) { + t.Helper() + runRepinCmd.Flags().VisitAll(func(f *pflag.Flag) { + if sv, ok := f.Value.(pflag.SliceValue); ok { + testsupport.Must(t, sv.Replace(nil), "resetting --%s", f.Name) + } else { + testsupport.Must(t, f.Value.Set(f.DefValue), "resetting --%s", f.Name) + } + f.Changed = false + }) +} + +// TestRunRepinDropFlagsReachTheEngine: the SAME run, the SAME deleted ref, and +// two different refusals — NOT_FOUND without the flag (repin has no bytes to +// adopt) and CONFLICT with it (the drop was accepted, and the run then failed +// the quiescence guard for having no non-terminal step). The change in code is +// the evidence that the flag crossed the seam. +func TestRunRepinDropFlagsReachTheEngine(t *testing.T) { + conn := newTestDB(t) + run, _ := driftedRun(t, conn, model.RunActive) + + // The corpus commit these flags exist for: the pinned file is DELETED, not + // edited. + testsupport.Must(t, os.Remove(filepath.Join(os.Getenv("DOCKET_PATH"), + "config", "contracts", "synthesize-findings.md")), "deleting the contract") + + err := repinViaCLI(t, conn, run.Ref(), map[string]string{ + "reason": "cc92e38 deleted it"}) + if err == nil { + t.Fatal("repin proceeded with a pinned ref deleted and no disposition") + } + if got := codeOf(t, err); got != output.ErrNotFound { + t.Errorf("code = %s, want %s: %v", got, output.ErrNotFound, err) + } + if !strings.Contains(err.Error(), "--drop") { + t.Errorf("the refusal %q does not point at the disposition that would "+ + "resolve it", err) + } + + err = repinViaCLI(t, conn, run.Ref(), map[string]string{ + "reason": "cc92e38 deleted it", "drop-unresolvable": "true"}) + if err == nil { + t.Fatal("repin proceeded on a run with no non-terminal step") + } + if got := codeOf(t, err); got != output.ErrConflict { + t.Errorf("code = %s, want %s — the drop must have been accepted and the "+ + "refusal must come from the quiescence guard instead: %v", + got, output.ErrConflict, err) + } + + // And --drop takes a ref, repeatably, reaching the same disposition. + err = repinViaCLI(t, conn, run.Ref(), map[string]string{ + "reason": "cc92e38 deleted it", "drop": "contracts/synthesize-findings.md"}) + if err == nil { + t.Fatal("repin proceeded on a run with no non-terminal step") + } + if got := codeOf(t, err); got != output.ErrConflict { + t.Errorf("code = %s, want %s: %v", got, output.ErrConflict, err) + } +} + +// TestRunRepinHelpDocumentsTheDisposition: the Long text is where the refusal +// semantics are stated, and DKT-582 changed them — NOT_FOUND is now refused +// UNLESS the ref is unread and covered by a flag. +func TestRunRepinHelpDocumentsTheDisposition(t *testing.T) { + for _, want := range []string{ + "--drop REF", "--drop-unresolvable", "UNLESS", "NULL new_sha256", + } { + if !strings.Contains(runRepinCmd.Long, want) { + t.Errorf("`run repin --help` does not mention %q", want) + } + } + for _, name := range []string{"drop", "drop-unresolvable"} { + if runRepinCmd.Flags().Lookup(name) == nil { + t.Errorf("`run repin` declares no --%s flag", name) + } + } +} + +// TestRenderRepinOutcomeStatesDrops: a dropped pin is a different fact from a +// moved one, and the human output has to say so rather than silently omitting +// a ref the operator asked about. +func TestRenderRepinOutcomeStatesDrops(t *testing.T) { + got := renderRepinOutcome(&engine.RepinOutcome{ + Run: "RUN-42", + Repinned: []engine.RepinChange{{ + Kind: "file", Ref: "contracts/implement.md", + OldSHA256: "aaaaaaaaaaaa", NewSHA256: "bbbbbbbbbbbb", + }}, + Dropped: []engine.RepinChange{{ + Kind: "file", Ref: "contracts/test-infra.md", + OldSHA256: "cccccccccccc", Dropped: true, + }}, + Unchanged: 3, + }) + for _, want := range []string{ + "contracts/implement.md", "contracts/test-infra.md", "dropped", "1 dropped", + } { + if !strings.Contains(got, want) { + t.Errorf("the outcome does not state %q:\n%s", want, got) + } + } +} diff --git a/internal/cli/run_report.go b/internal/cli/run_report.go index 682f39b5..975f67b3 100644 --- a/internal/cli/run_report.go +++ b/internal/cli/run_report.go @@ -2,6 +2,7 @@ package cli import ( "fmt" + "sort" "strings" "github.com/charmbracelet/lipgloss" @@ -37,6 +38,13 @@ It works on a run in ANY status. A ` + "`planning`" + ` run reports zeros; an ab one reports the trail up to abandonment. A report that refused on a non-terminal run would be useless during exactly the run you want to inspect. +STEP METADATA IS REPORTED TWICE, on purpose. ` + "`Metadata`" + ` rolls every key up +to its distinct values with counts — a run-level answer — and ` + "`Step metadata`" + ` +prints each step's whole bag beside the status that step ended in. Grouping by +key is what discards which values two keys took TOGETHER on one step, so a bag +whose keys are a request and its resolution is readable only in the second +section. Core reads no key in either. + THE BUDGET NUMBERS ARE BARE. There is no currency and no unit: what they count is the workflow's business. The report publishes the cap, where the cap came from, the declared-cost floor, reported usage per unit, max(reported, floor), @@ -205,6 +213,14 @@ func writeBudgetLines(b *strings.Builder, budget engine.RunBudgetReport, line fu if budget.BreachReason != "" { line("Breach:", exec.Render(budget.BreachReason)) } + // DKT-584: on a run whose panels cast, the budget section itself says the + // seats' measured spend is excluded from the numbers above and where it + // went instead — the silent omission is the bug. The note is core's own + // constant (engine.VoteUsageExcludedNote), not stored text, so it is not + // escaped: escaping it would quote a constant. + if budget.VoteUsageNote != "" { + line("Vote usage:", budget.VoteUsageNote) + } } // writeReportSections emits R3 through R7. Shared for the same reason @@ -219,6 +235,13 @@ func writeReportSections( } stepLabel := stepLabeler(r.Attempts) + // DKT-594's two pin sections, ahead of the step rollup because both are + // statements about the AGREEMENT everything below ran under. A reader who + // takes a finding out of this document needs to know the corpus moved + // before they read the finding, not after. + writePinnedWorkflows(r.PinnedWorkflows, header, line) + writePinEpochs(r.PinEpochs, r.Attempts, stepLabel, header, line) + if len(r.Steps) > 0 { header("Steps") for _, sc := range r.Steps { @@ -236,7 +259,7 @@ func writeReportSections( if len(retried) > 0 { header("Attempts") for _, a := range retried { - line(stepLabel(a), fmt.Sprintf("%d", a.Attempts)) + line(stepLabel(a)+":", fmt.Sprintf("%d", a.Attempts)) } } @@ -289,7 +312,7 @@ func writeReportSections( if resolved := stepResolution(a, dispositions); resolved != "" { detail += " — " + resolved } - line(stepLabel(a), detail) + line(stepLabel(a)+":", detail) } } } @@ -347,6 +370,11 @@ func writeReportSections( } writeMetadataRollup(r.Metadata, "Metadata", header, line) + // The bags THE ROLLUP JUST COLLAPSED, one step per line (DKT-868) — + // immediately below the rollup for the same reason "Step usage" sits below + // the budget's per-unit totals: the rollup is the headline and this is the + // detail behind it. + writeStepMetadata(r.Attempts, stepLabel, header, line) // The casts' own claims (DKT-71): the one spend the usage ledger cannot // attribute, rolled up the same opaque way. writeMetadataRollup(r.VoteMetadata, "Vote metadata", header, line) @@ -372,8 +400,30 @@ func writeReportSections( detail += fmt.Sprintf( " — %d reported NOTHING, so their spend is missing from this "+ "run's totals, not zero", c.Silent()) + // The seating path(s) of the missing seats, ON the coverage + // line (DKT-733): "which seat path fails to report" is the + // diagnosis question, and the count alone could not answer it. + // The paths are core's closed vocabulary, not escaped. + if paths := silentSeatPathCounts(r.SilentVoteSeats); paths != "" { + detail += " (" + paths + ")" + } } line("Coverage:", detail) + // The identity behind the count, one aimable line per seat: the + // proposal is what `vote backfill-usage` takes. Voter and role are + // stored, caster-supplied text on its way to a terminal (R11). + for _, s := range r.SilentVoteSeats { + who := exec.Render(s.Voter) + if s.Role != "" { + who += " as " + exec.Render(s.Role) + } + line("Silent:", fmt.Sprintf("%s seat %s (%s)", s.Proposal, who, s.Path)) + } + if len(r.SilentVoteSeats) > 0 { + line("Backfill:", + "`docket vote backfill-usage ` records these "+ + "seats' spend after the fact") + } } } @@ -391,6 +441,30 @@ func writeReportSections( } } +// silentSeatPathCounts renders the silent seats' seating paths with counts — +// `conversational-gate: 8, vote-step: 4` — for the coverage line (DKT-733). +// Sorted by path name: a total order, so two renders of one run are +// byte-identical (R9). Empty when there is nothing to attribute. +func silentSeatPathCounts(seats []engine.SilentVoteSeat) string { + if len(seats) == 0 { + return "" + } + counts := make(map[string]int, 2) + for _, s := range seats { + counts[s.Path]++ + } + paths := make([]string, 0, len(counts)) + for p := range counts { + paths = append(paths, p) + } + sort.Strings(paths) + parts := make([]string, 0, len(paths)) + for _, p := range paths { + parts = append(parts, fmt.Sprintf("%s: %d", p, counts[p])) + } + return "silent by seating path — " + strings.Join(parts, ", ") +} + // stepLabeler decides how a step-attempt line NAMES its step (DKT-405). // // An instance label is unique within an issue and repeats across them: every @@ -413,8 +487,14 @@ func writeReportSections( // are stored text on their way to a terminal and must go through exec.Render // (R11), and `"HRN-300 verify@2":` keeps the label reading as one name where // `"HRN-300" "verify@2":` reads as two columns that happen to be adjacent. +// +// It returns the NAME WITHOUT the trailing colon, and the two sections that +// print it as a key add one. DKT-594's epoch section lists several of these +// inside one line's value, where a colon per name would read as a broken +// key-value pair — and a labeler that could only produce keys would have forced +// that section to re-derive the naming rule and be free to disagree with it. func stepLabeler(attempts []engine.StepAttempt) func(engine.StepAttempt) string { - bare := func(a engine.StepAttempt) string { return exec.Render(a.Instance) + ":" } + bare := func(a engine.StepAttempt) string { return exec.Render(a.Instance) } issues := make(map[string]struct{}, 2) for _, a := range attempts { @@ -431,7 +511,112 @@ func stepLabeler(attempts []engine.StepAttempt) func(engine.StepAttempt) string if a.Issue == "" { return bare(a) } - return exec.Render(a.Issue+" "+a.Instance) + ":" + return exec.Render(a.Issue + " " + a.Instance) + } +} + +// writePinnedWorkflows is DKT-594's staleness section: per pinned workflow, how +// far the registry has moved past the version this run is expanded from. +// +// IT PRINTS THE UP-TO-DATE ROWS TOO. The section exists because every analyst +// on RUN-32 went to git to find out whether the pinned `ui-change@8` was +// current before trusting a finding from the run — and "the report did not +// mention it" is not an answer to that question, it is the same absence that +// sent them to git. A line saying `current` is the answer; a missing line is +// indistinguishable from a report that never looked. +func writePinnedWorkflows( + rows []engine.PinnedWorkflowStaleness, + header func(title string), line func(k, v string), +) { + if len(rows) == 0 { + return + } + header("Pinned workflows") + for _, w := range rows { + // The name is a workflow author's opaque string on its way to a + // terminal, so both refs go through the escaper (R11). + current := exec.Render(fmt.Sprintf("%s@%d", w.Name, w.CurrentVersion)) + var detail string + switch { + case w.CurrentVersion == 0: + // Every version of the name is retired, or the name is gone from + // this project's registry. Not "current" — there is nothing to be + // current WITH, and a run pinning a name nothing can bind any more + // is a fact a reader must not have to infer from a silence. + detail = "0 behind — no registered version of this name still binds" + case w.Behind == 0 && w.CurrentVersion != w.PinnedVersion: + // The pinned version sits ABOVE the binding head: its own version, + // or every version over it, was retired after this run froze. + detail = "0 behind — the highest version that still binds is " + current + case w.Behind == 0: + detail = "current" + default: + detail = fmt.Sprintf("%d version(s) behind — current is %s", + w.Behind, current) + } + line(exec.Render(w.Ref)+":", detail) + } +} + +// writePinEpochs is DKT-594's second section: the run's pin-agreement timeline, +// and which steps' recorded work ran under each agreement. +// +// PRESENT ONLY ON A RUN THAT REPINNED — the engine leaves PinEpochs empty +// otherwise, and a run whose agreement never moved needs no reconciliation: the +// `pins` table already states the bytes every one of its steps read. +// +// The steps ride on the epoch's own line rather than as a `pin_epoch` column on +// every step line. The question this answers is "which side of the repin was +// this step on", which is a partition, and a partition is read by looking at +// the two groups — RUN-39's post-mortem was reconstructing exactly this +// grouping by hand from event seqs and step ids. `--json` carries the inverse +// (each attempt's own `pin_epoch`) for a consumer joining the other way. +func writePinEpochs( + epochs []engine.PinEpoch, attempts []engine.StepAttempt, + stepLabel func(engine.StepAttempt) string, + header func(title string), line func(k, v string), +) { + if len(epochs) < 2 { + return + } + header("Pin epochs") + + ran := make(map[int][]string, len(epochs)) + for _, a := range attempts { + if a.PinEpoch > 0 { + ran[a.PinEpoch] = append(ran[a.PinEpoch], stepLabel(a)) + } + } + + for _, e := range epochs { + detail := fmt.Sprintf("%s at seq %d", e.Origin, e.FromSeq) + if e.FromSeq == 0 { + // The activation event was pruned away. Saying "at seq 0" would + // name a sequence number that never existed. + detail = e.Origin + } + for _, c := range e.Changes { + switch { + case c.Dropped: + detail += fmt.Sprintf(" — %s %s dropped (was %s)", + c.Kind, exec.Render(c.Ref), shortSHA(c.OldSHA256)) + default: + detail += fmt.Sprintf(" — %s %s %s->%s", + c.Kind, exec.Render(c.Ref), + shortSHA(c.OldSHA256), shortSHA(c.NewSHA256)) + } + } + if e.Reason != "" { + detail += " — " + exec.Render(reasonHead(e.Reason, 72)) + } + line(fmt.Sprintf("Epoch %d:", e.Epoch), detail) + // A step list per epoch, and NOTHING when an epoch ran no step. An + // empty list would read as "these steps ran and produced nothing"; + // the absence says the agreement governed no recorded work, which is + // the ordinary shape of a repin performed to unwedge a run. + if steps := ran[e.Epoch]; len(steps) > 0 { + line(fmt.Sprintf(" ran under %d:", e.Epoch), strings.Join(steps, ", ")) + } } } @@ -515,6 +700,75 @@ func writeMetadataRollup( } } +// writeStepMetadata is DKT-868's section: each step's whole bag on ONE LINE, +// beside the status that step ended in. +// +// WHY A SECOND SECTION RATHER THAN A RESHAPED FIRST ONE. The rollup above +// answers a run-level question — which values did this key take, and how often +// — and is the right shape for it. What it cannot answer is which values two +// keys took TOGETHER on one step, because grouping by key is exactly what +// discards the pairing. RUN-51's rollup published a key whose partner key never +// showed the value it resolved to: a real mismatch, on one step, that the +// document could not name. Recovering it meant `docket step show` per step, and +// the audit that found this ran ~90 of them across 19 runs. Deleting the rollup +// to fix that would trade a run-level answer for a per-step one; both questions +// are real, so the document carries both. +// +// THE STATUS RIDES ON THE LINE because the bag's completeness depends on it. A +// dispatcher's `step claim --metadata` (DKT-592) lands before the work runs, and +// a step that then failed or was reaped carries only that half — so a bag with +// a request and no resolution is a FINDING on a failed step and a defect on a +// done one, and a line that did not say which invited the wrong reading. In the +// rollup those two rows were indistinguishable, which is why drift concentrated +// in failures read as no drift at all. +// +// NO KEY IS NAMED HERE (docs/design/genericity.md, R7). The section prints +// whatever keys a bag holds, sorted, and never compares two of them: `a=1, b=2` +// on one line is all core does, and what that pair MEANS is the workflow +// author's business. +func writeStepMetadata( + attempts []engine.StepAttempt, stepLabel func(engine.StepAttempt) string, + header func(title string), line func(k, v string), +) { + var rows []engine.StepAttempt + for _, a := range attempts { + if len(a.Metadata) > 0 || a.MetadataUnreadable { + rows = append(rows, a) + } + } + if len(rows) == 0 { + return + } + header("Step metadata") + for _, a := range rows { + detail := a.Status + if a.MetadataUnreadable { + // Said out loud rather than skipped. An absent bag reads as "the + // dispatcher recorded nothing", which is the comfortable claim and + // the wrong one — the row exists and its bytes do not decode. + line(stepLabel(a)+":", detail+" — metadata is stored but does not "+ + "decode as an object; `docket step show` has the raw bytes") + continue + } + // Sorted keys, so two renders of one run are byte-identical (R9) — a + // bare map range would emit a different document per invocation. + keys := make([]string, 0, len(a.Metadata)) + for k := range a.Metadata { + keys = append(keys, k) + } + sort.Strings(keys) + parts := make([]string, 0, len(keys)) + for _, k := range keys { + // Both halves are a workflow author's opaque strings on their way to + // a TERMINAL (R11), and the value goes through db's own renderer so + // this section and the rollup above spell one value identically. + parts = append(parts, fmt.Sprintf("%s=%s", + exec.Render(k), exec.Render(db.RenderMetadataValue(a.Metadata[k])))) + } + line(stepLabel(a)+":", detail+" — "+strings.Join(parts, ", ")) + } +} + func writeVerdicts( counts []db.VerdictCount, title string, header func(title string), line func(k, v string), @@ -559,7 +813,35 @@ func writeVerdicts( summary += fmt.Sprintf(" — %d of %d ran a stub", c.Stub, total) } } - line(exec.Render(c.Name)+":", summary) + // THE PRE MARKER RIDES ON THE NAME (DKT-862), because a pre-gate's + // numbers are not the same KIND of fact as a blocking gate's: §11.1 + // runs it at claim as an input to the step, and PG4 keeps it out of the + // saga's verdict, so `fail 1` here did not route anything. This section + // rendered the two identically, and on RUN-61 a conductor reading it + // nearly reported a fix round as burned on an advisory failure. + // + // `[pre]` is `step show`'s spelling, off the same `pre` column, so an + // operator moving between the two surfaces reads one marker and not + // two vocabularies for one fact. + name := exec.Render(c.Name) + // The four verdicts partition, so their sum is the row count `Pre` + // overlaps — INCLUDING `skipped`, which pre-gates record whenever the + // tree could not be bound and which is therefore the most common + // pre-gate row of all. + if total := c.Pass + c.Fail + c.Unmatched + c.Skipped; c.Pre > 0 { + if c.Pre >= total { + name += " [pre]" + } else { + // A name that ran BOTH ways in one run — the same gate declared + // `pre` by one workflow and blocking by another — cannot take + // the marker without claiming the whole tally was advisory. The + // ratio says which part was. + summary += fmt.Sprintf( + " — %d of %d ran as a pre-gate (advisory, routed nothing)", + c.Pre, total) + } + } + line(name+":", summary) } } diff --git a/internal/cli/run_report_metadata_test.go b/internal/cli/run_report_metadata_test.go new file mode 100644 index 00000000..03070a56 --- /dev/null +++ b/internal/cli/run_report_metadata_test.go @@ -0,0 +1,179 @@ +package cli + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// DKT-868 — the rendered document, which is what an operator who never passes +// `--json` has. +// +// The "Metadata" rollup published RUN-51's anomaly and could not name it: an +// `effort_resolved` value of `low` under a key whose partner `effort_requested` +// never showed one. Grouping by key is what discards the pairing, so the only +// remaining reader was `docket step show`, one step at a time. + +// tieredRunReport is the RUN-51 shape, minimally: three steps, one of which was +// served at a tier other than the one it was dispatched at, and a fourth that +// failed before it could report what it resolved to. +func tieredRunReport() *engine.RunReport { + return &engine.RunReport{ + Run: &model.Run{ID: 51, Status: model.RunDone}, + Steps: []model.StatusCount{ + {Status: db.StepDone, Count: 3}, + {Status: db.StepFailedRouted, Count: 1}, + }, + Attempts: []engine.StepAttempt{ + { + Step: "STEP-1", Instance: "implement@0", Status: db.StepDone, + Attempts: 1, + Metadata: map[string]any{ + "effort_requested": "high", "effort_resolved": "high", + }, + }, + { + Step: "STEP-2", Instance: "review@0", Status: db.StepDone, + Attempts: 1, + Metadata: map[string]any{ + "effort_requested": "high", "effort_resolved": "low", + }, + }, + { + Step: "STEP-3", Instance: "reconcile@0", Status: db.StepDone, + Attempts: 1, + Metadata: map[string]any{ + "effort_requested": "high", "effort_resolved": "high", + }, + }, + { + Step: "STEP-4", Instance: "verify@0", Status: db.StepFailedRouted, + Attempts: 1, + Metadata: map[string]any{"effort_requested": "high"}, + }, + }, + // The rollup as it always rendered: the anomaly, unattributable. + Metadata: []db.MetadataKeyRollup{ + {Key: "effort_requested", Values: []db.MetadataValueCount{ + {Value: "high", Count: 4}}}, + {Key: "effort_resolved", Values: []db.MetadataValueCount{ + {Value: "high", Count: 2}, {Value: "low", Count: 1}}}, + }, + } +} + +// TestReportPairsEachStepsKeysOnOneLine is the acceptance criterion: the two +// keys the rollup separated appear together, on the line that names the step. +func TestReportPairsEachStepsKeysOnOneLine(t *testing.T) { + out := renderPlainRunReport(tieredRunReport()) + + if !strings.Contains(out, "Step metadata") { + t.Fatalf("the report has no per-step metadata section:\n%s", out) + } + + var line string + for _, l := range strings.Split(out, "\n") { + if strings.Contains(l, "review@0") && strings.Contains(l, "effort_resolved") { + line = l + } + } + if line == "" { + t.Fatalf("no line pairs review@0 with its bag; the drifted step is "+ + "still recoverable only by `step show`:\n%s", out) + } + for _, needle := range []string{"effort_requested", "high", "effort_resolved", "low"} { + if !strings.Contains(line, needle) { + t.Errorf("the review@0 line %q never says %q", line, needle) + } + } + // The rollup is still there. Both questions are real — "which values did + // this key take across the run" and "which values did two keys take on one + // step" — and the fix adds the second reader rather than trading one for + // the other. + if !strings.Contains(out, "\nMetadata\n") { + t.Errorf("the key rollup is gone from the document:\n%s", out) + } +} + +// TestFailedStepsHalfBagIsAttributable is the issue's stated consequence: a +// step that failed before reporting a resolution must not read as one whose +// tiers agreed. +func TestFailedStepsHalfBagIsAttributable(t *testing.T) { + out := renderPlainRunReport(tieredRunReport()) + + var line string + for _, l := range strings.Split(out, "\n") { + if strings.Contains(l, "verify@0") && strings.Contains(l, "effort_requested") { + line = l + } + } + if line == "" { + t.Fatalf("the failed step's dispatch bag is in no line of the "+ + "document:\n%s", out) + } + if !strings.Contains(line, db.StepFailedRouted) { + t.Errorf("the verify@0 line %q does not say the step failed, so its "+ + "missing resolution reads as agreement", line) + } + if strings.Contains(line, "effort_resolved") { + t.Errorf("the verify@0 line %q invents a resolution the step never "+ + "reported", line) + } +} + +// TestStepMetadataLineIsDeterministic is R9 through a map-valued field: two +// renders of one report are byte-identical. A bare range over the bag would +// emit a different document per invocation. +func TestStepMetadataLineIsDeterministic(t *testing.T) { + first := renderPlainRunReport(tieredRunReport()) + for i := 0; i < 20; i++ { + if again := renderPlainRunReport(tieredRunReport()); again != first { + t.Fatalf("render %d differs from the first:\n%s\n---\n%s", + i, first, again) + } + } +} + +// TestReportWithoutStepBagsIsUnchanged is the regression half: a run whose +// steps carried no metadata gains no section, no header, and no blank line. +func TestReportWithoutStepBagsIsUnchanged(t *testing.T) { + report := tieredRunReport() + for i := range report.Attempts { + report.Attempts[i].Metadata = nil + } + report.Metadata = nil + + out := renderPlainRunReport(report) + if strings.Contains(out, "Step metadata") { + t.Errorf("a run with no step bags grew a per-step metadata "+ + "section:\n%s", out) + } +} + +// TestUnreadableBagSaysSoInTheDocument keeps R10's tolerance from reading as an +// absence: bytes that do not decode are reported as such, not skipped into a +// silence a reader would take for "the dispatcher recorded nothing". +func TestUnreadableBagSaysSoInTheDocument(t *testing.T) { + report := tieredRunReport() + report.Attempts[0].Metadata = nil + report.Attempts[0].MetadataUnreadable = true + + out := renderPlainRunReport(report) + + var line string + for _, l := range strings.Split(out, "\n") { + if strings.Contains(l, "implement@0") && strings.Contains(l, "decode") { + line = l + } + } + if line == "" { + t.Fatalf("a stored bag that does not decode is reported nowhere:\n%s", out) + } + if !strings.Contains(line, "step show") { + t.Errorf("the line %q says the bag is unreadable and not where the raw "+ + "bytes are", line) + } +} diff --git a/internal/cli/run_report_pins_test.go b/internal/cli/run_report_pins_test.go new file mode 100644 index 00000000..25b2b2c6 --- /dev/null +++ b/internal/cli/run_report_pins_test.go @@ -0,0 +1,155 @@ +package cli + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// DKT-594 — the RENDERED document, which is what a post-mortem reader has. +// +// RUN-32's readers each went to git to find out that `ui-change@8` was five +// registered versions behind before trusting a finding; RUN-39's read the +// repin's event seqs against step ids by hand to learn which contract bytes a +// completed step had actually consumed. Neither fact was in this document. + +// staleRunReport is the RUN-32/RUN-39 shape: two pinned workflows, one of them +// well behind the corpus, and a run whose agreement moved mid-flight with steps +// on both sides of the move. +func staleRunReport() *engine.RunReport { + return &engine.RunReport{ + Run: &model.Run{ID: 39, Status: model.RunActive}, + PinnedWorkflows: []engine.PinnedWorkflowStaleness{ + {Ref: "standard-change@2", Name: "standard-change", + PinnedVersion: 2, CurrentVersion: 2, Behind: 0}, + {Ref: "ui-change@8", Name: "ui-change", + PinnedVersion: 8, CurrentVersion: 13, Behind: 5}, + }, + PinEpochs: []engine.PinEpoch{ + {Epoch: 1, FromSeq: 5301, Origin: engine.PinEpochActivation}, + { + Epoch: 2, FromSeq: 5375, Origin: engine.PinEpochRepin, + Reason: "corpus install 2026-08-23", + Changes: []engine.RepinChange{{ + Kind: db.PinKindFile, Ref: "contracts/implement.md", + OldSHA256: "aaaaaaaaaaaaaaaaaaaa", NewSHA256: "bbbbbbbbbbbbbbbbbbbb", + }}, + }, + }, + Attempts: []engine.StepAttempt{ + { + Step: "STEP-1350", Instance: "implement@0", Issue: "DKT-1", + Status: db.StepDone, Attempts: 1, PinEpoch: 1, + }, + { + Step: "STEP-1353", Instance: "verify@0", Issue: "DKT-2", + Status: db.StepDone, Attempts: 1, PinEpoch: 2, + }, + }, + } +} + +// TestReportPublishesTheStalenessCount is criterion 1: the subtraction is in +// the document, naming the version the corpus has reached. +func TestReportPublishesTheStalenessCount(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(staleRunReport()), "Pinned workflows") + if len(lines) != 2 { + t.Fatalf("Pinned workflows printed %d rows, want one per pinned workflow: %v", + len(lines), lines) + } + if !linesContain(lines, `"ui-change@8"`) || !linesContain(lines, "5 version(s) behind") { + t.Errorf("no staleness count for the drifted pin:\n %s", strings.Join(lines, "\n ")) + } + if !linesContain(lines, `"ui-change@13"`) { + t.Errorf("the staleness line does not name the version the corpus reached:\n %s", + strings.Join(lines, "\n ")) + } +} + +// TestAnUpToDatePinStillGetsALine is the point of the section, not an edge +// case: "the report did not mention it" is the same absence that sent RUN-32's +// readers to git. +func TestAnUpToDatePinStillGetsALine(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(staleRunReport()), "Pinned workflows") + for _, l := range lines { + if strings.Contains(l, `"standard-change@2"`) { + if !strings.Contains(l, "current") { + t.Errorf("the up-to-date pin's line does not say so:\n %s", l) + } + return + } + } + t.Errorf("the up-to-date pin has no line at all:\n %s", strings.Join(lines, "\n ")) +} + +// TestReportPartitionsStepsByPinEpoch is criterion 2: the grouping RUN-39's +// post-mortem assembled by hand, rendered. +func TestReportPartitionsStepsByPinEpoch(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(staleRunReport()), "Pin epochs") + + for _, want := range []string{ + "Epoch 1:", "activation at seq 5301", + "Epoch 2:", "repin at seq 5375", + "contracts/implement.md", "corpus install 2026-08-23", + } { + if !linesContain(lines, want) { + t.Errorf("the epoch timeline is missing %q:\n %s", + want, strings.Join(lines, "\n ")) + } + } + + // The partition itself: each step is listed under the agreement its + // recorded work ran under, and under no other. + for _, tc := range []struct{ epoch, step, other string }{ + {"ran under 1:", "DKT-1 implement@0", "DKT-2 verify@0"}, + {"ran under 2:", "DKT-2 verify@0", "DKT-1 implement@0"}, + } { + var found bool + for _, l := range lines { + if !strings.HasPrefix(l, tc.epoch) { + continue + } + found = true + if !strings.Contains(l, tc.step) { + t.Errorf("%s does not list %s:\n %s", tc.epoch, tc.step, l) + } + if strings.Contains(l, tc.other) { + t.Errorf("%s also lists %s, which ran under the other agreement:\n %s", + tc.epoch, tc.other, l) + } + } + if !found { + t.Errorf("no %q line:\n %s", tc.epoch, strings.Join(lines, "\n ")) + } + } +} + +// TestARunWithOneAgreementHasNoEpochSection is the falsifier for the section's +// presence: on a run that never repinned the partition is a single group, the +// pins table already states it, and a section per report would be noise. +func TestARunWithOneAgreementHasNoEpochSection(t *testing.T) { + r := staleRunReport() + r.PinEpochs = nil + for i := range r.Attempts { + r.Attempts[i].PinEpoch = 0 + } + if out := renderPlainRunReport(r); strings.Contains(out, "Pin epochs") { + t.Errorf("a run that never repinned rendered an epoch section:\n%s", out) + } +} + +// TestPlanningRunRendersNoPinSections: a run that never activated pins nothing, +// and a section of zeros would report an agreement it does not have. +func TestPlanningRunRendersNoPinSections(t *testing.T) { + out := renderPlainRunReport(&engine.RunReport{ + Run: &model.Run{ID: 40, Status: model.RunPlanning}, + }) + for _, section := range []string{"Pinned workflows", "Pin epochs"} { + if strings.Contains(out, section) { + t.Errorf("a planning run rendered %q:\n%s", section, out) + } + } +} diff --git a/internal/cli/run_report_render_test.go b/internal/cli/run_report_render_test.go index aca90fbc..3b70e27d 100644 --- a/internal/cli/run_report_render_test.go +++ b/internal/cli/run_report_render_test.go @@ -294,3 +294,181 @@ func linesContain(lines []string, needle string) bool { } return false } + +// DKT-733 — RUN-51's coverage line said 12 of 57 seats reported nothing and +// named neither the seats nor the seating paths, so the gap could not be +// diagnosed or back-filled from the report. + +// silentSeatsRunReport is the RUN-51 shape, minimally: casts on both seating +// paths, some of them silent. +func silentSeatsRunReport() *engine.RunReport { + return &engine.RunReport{ + Run: &model.Run{ID: 51, Status: model.RunDone}, + VoteUsageCoverage: db.VoteUsageCoverage{Casts: 5, Reported: 2}, + SilentVoteSeats: []engine.SilentVoteSeat{ + {Proposal: "DKT-V218", Voter: "sec-arch", Role: "judge", + Path: engine.SeatPathConversationalGate}, + {Proposal: "DKT-V219", Voter: "sec-crypto", + Path: engine.SeatPathConversationalGate}, + {Proposal: "DKT-V220", Voter: "verify-seat", Role: "verifier", + Path: engine.SeatPathVoteStep}, + }, + } +} + +// TestCoverageLineNamesSilentSeatPaths is DKT-733's AC3: the coverage line +// itself names the seating path(s) of the missing seats, with counts. +func TestCoverageLineNamesSilentSeatPaths(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(silentSeatsRunReport()), "Vote usage") + + var coverage string + for _, l := range lines { + if strings.HasPrefix(l, "Coverage:") { + coverage = l + } + } + if coverage == "" { + t.Fatalf("no coverage line rendered:\n%v", lines) + } + if !strings.Contains(coverage, "3 reported NOTHING") { + t.Errorf("the coverage line lost its silent count:\n %s", coverage) + } + if !strings.Contains(coverage, "conversational-gate: 2, vote-step: 1") { + t.Errorf("the coverage line does not name the seating paths of the "+ + "missing seats:\n %s", coverage) + } +} + +// TestSilentSeatsAreEnumeratedWithABackfillPointer is DKT-733's AC1 rendered: +// each silent seat is one line naming its proposal (the backfill verb's +// argument), its voter, and its path — and the verb itself is named once. +func TestSilentSeatsAreEnumeratedWithABackfillPointer(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(silentSeatsRunReport()), "Vote usage") + + for _, needle := range []string{ + `DKT-V218 seat "sec-arch" as "judge" (conversational-gate)`, + `DKT-V219 seat "sec-crypto" (conversational-gate)`, + `DKT-V220 seat "verify-seat" as "verifier" (vote-step)`, + "vote backfill-usage", + } { + if !linesContain(lines, needle) { + t.Errorf("the Vote usage section never says %q:\n%v", needle, lines) + } + } +} + +// TestFullyReportedRunListsNoSilentSeats: a run whose every seat reported +// renders the coverage line alone — no Silent rows, no backfill pointer. +func TestFullyReportedRunListsNoSilentSeats(t *testing.T) { + r := silentSeatsRunReport() + r.VoteUsageCoverage = db.VoteUsageCoverage{Casts: 5, Reported: 5} + r.SilentVoteSeats = nil + + lines := sectionLines(t, renderPlainRunReport(r), "Vote usage") + if !linesContain(lines, "5 of 5 seat(s) reported spend") { + t.Fatalf("no coverage line on a fully-reported run:\n%v", lines) + } + for _, forbidden := range []string{"Silent:", "Backfill:", "NOTHING"} { + if linesContain(lines, forbidden) { + t.Errorf("a fully-reported run still renders %q:\n%v", forbidden, lines) + } + } +} + +// --------------------------------------------------------------------------- +// DKT-862 — the Gates tally distinguishes a pre-gate from a blocking one +// --------------------------------------------------------------------------- + +// The section rendered `ac-commands: pass 0, fail 1` whether the failure +// BLOCKED the step or was an advisory input to it. A `pre = true` gate never +// routes — §11.1 runs it at claim, PG4 keeps it out of the saga's verdict — so +// on RUN-61 three of them failed and every step routed on its executor's own +// verdict anyway. Only `step show` carried the marker, and a conductor reading +// this document nearly reported a fix round as burned on a gate artifact. + +// mixedGateReport is RUN-61's shape: one advisory gate, one blocking gate, and +// one name that ran BOTH ways across the run's workflows. +func mixedGateReport() *engine.RunReport { + return &engine.RunReport{ + Run: &model.Run{ID: 61, Status: model.RunDone}, + Gates: []db.VerdictCount{ + // The pre-gate RUN-61 kept failing: every row advisory. + {Name: "ac-commands", Fail: 1, Pre: 1}, + // A blocking gate that failed identically — the row whose bytes + // were indistinguishable from the one above. + {Name: "build", Pass: 2, Fail: 1}, + // One name, both declarations: `design-qa` declares render-verify + // `pre`, another workflow gates on it. + {Name: "render-verify", Pass: 1, Fail: 1, Pre: 1}, + }, + } +} + +// TestPreGateTallyCarriesTheMarker is DKT-862's AC1. +// +// The marker sits OUTSIDE the rendered name's quotes: `exec.Render` escapes a +// stored string, and the marker is core's own word about it — quoting it would +// claim a workflow author had written it. +func TestPreGateTallyCarriesTheMarker(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(mixedGateReport()), "Gates") + + if !linesContain(lines, `"ac-commands" [pre]: pass 0, fail 1`) { + t.Errorf("the advisory gate's tally carries no [pre] marker, so it "+ + "reads as a gate that blocked the step:\n%v", lines) + } +} + +// TestBlockingGateTallyIsUnmarked is the other half of AC1, and the half that +// makes the marker mean something: a gate that DID route must not carry it. +func TestBlockingGateTallyIsUnmarked(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(mixedGateReport()), "Gates") + + for _, l := range lines { + if strings.Contains(l, `"build"`) && strings.Contains(l, "[pre]") { + t.Errorf("a blocking gate's tally carries the advisory marker:\n %s", l) + } + } + if !linesContain(lines, `"build":`) || !linesContain(lines, "pass 2, fail 1") { + t.Errorf("the blocking gate's line changed:\n%v", lines) + } +} + +// TestSplitGateNameReportsTheRatioRatherThanTheMarker: one gate name declared +// `pre` by one workflow and blocking by another cannot take the marker without +// claiming the whole tally was advisory, so the ratio says which part was. +func TestSplitGateNameReportsTheRatioRatherThanTheMarker(t *testing.T) { + lines := sectionLines(t, renderPlainRunReport(mixedGateReport()), "Gates") + + for _, l := range lines { + if !strings.Contains(l, `"render-verify"`) { + continue + } + if strings.Contains(l, "[pre]") { + t.Errorf("a name that also ran as a blocking gate takes the "+ + "whole-tally marker:\n %s", l) + } + if !strings.Contains(l, "1 of 2 ran as a pre-gate") { + t.Errorf("a split gate name says nothing about which rows "+ + "routed:\n %s", l) + } + return + } + t.Fatalf("render-verify is not in the Gates section at all:\n%v", lines) +} + +// TestActionsTallyIsUnaffected: `action_results` has no `pre` column, and the +// shared rollup body must render actions exactly as before rather than growing +// a marker off an honest zero. +func TestActionsTallyIsUnaffected(t *testing.T) { + r := mixedGateReport() + r.Actions = []db.VerdictCount{{Name: "reconcile", Pass: 1}} + + lines := sectionLines(t, renderPlainRunReport(r), "Actions") + if !linesContain(lines, `"reconcile": pass 1, fail 0`) { + t.Errorf("the Actions section changed:\n%v", lines) + } + if linesContain(lines, "[pre]") || linesContain(lines, "pre-gate") { + t.Errorf("an action tally grew a pre marker off a column its table "+ + "does not have:\n%v", lines) + } +} diff --git a/internal/cli/run_verify_pins.go b/internal/cli/run_verify_pins.go index f64e7385..01948350 100644 --- a/internal/cli/run_verify_pins.go +++ b/internal/cli/run_verify_pins.go @@ -27,8 +27,17 @@ Each pin gets one verdict: refuses (exit 4), so the run is already blocked on it missing the run depends on the ref and it no longer resolves at all -Exit 4 if any pin changed, exit 2 if any is missing and none changed, 0 when -every pin is sound. +THEN THE PIN SET IS CHECKED FOR CLOSURE (DKT-821). Every pin matching disk does +not make a pin set healthy: the pinned bytes themselves reference files, and a +reference the run never pinned makes every step whose packet resolves it +unclaimable. Those are reported separately, as ` + "`references`" + `: + + unpinned-reference a pinned file's packet_includes (or a pending step's + packet entry) names a ref this run does not pin — the + claim refuses (exit 3) whether or not the file is there + +Exit 4 if any pin changed, exit 2 if any is missing and none changed, exit 3 if +the pins are all sound but the pin set is not closed, 0 when both halves pass. WHY THIS VERB EXISTS. The verbs that check pins each check only the pins THEY read: step render verifies its template and its own step's packet files, and a @@ -66,10 +75,27 @@ func runRunVerifyPins(cmd *cobra.Command, ref string, w *output.Writer) error { remedy := fmt.Sprintf("; restore the file(s), or `docket run repin %s "+ "--reason ...` adopts current disk bytes for the remaining steps "+ "(DKT-408)", report.Run) - if report.Changed == 0 { + switch { + case report.Changed > 0: + // Drift wins the disposition: it is the half a repin can fix, and + // its remedy is the one to try first. + case report.Missing > 0: code = output.ErrNotFound // A missing ref has no current bytes to adopt; repin would refuse. remedy = "; restore the file(s) or abandon the run" + default: + // ONLY THE CLOSURE IS OPEN. Nothing drifted, so there is nothing on + // disk to restore and repin has no changed bytes to adopt — it + // would report a clean no-op on exactly this run. The pin set froze + // without the ref, and only a new activation pins one. The code is + // the one the blocked claims themselves return, which is this + // verb's rule: a caller that handles the refusal handles the + // prediction of it. + code = output.ErrValidation + remedy = "; the pin set froze without the ref(s), so the step(s) " + + "reading them refuse at claim (VALIDATION_ERROR) — start a new " + + "run to pin them, or take the reference back out of the " + + "referencing file" } return cmdErr(fmt.Errorf("%s: %s%s", report.Run, engine.PinReportReason(report), remedy), code) @@ -85,7 +111,11 @@ func runRunVerifyPins(cmd *cobra.Command, ref string, w *output.Writer) error { func renderPinReport(r *engine.PinReport) string { var b strings.Builder - fmt.Fprintf(&b, "%s: %d pin(s), all sound", r.Run, len(r.Pins)) + // BOTH HALVES ARE STATED on success, because a reader who was told only + // "all sound" cannot tell whether the closure was checked at all — which is + // precisely the false confidence DKT-821 was. + fmt.Fprintf(&b, "%s: %d pin(s), all sound; every ref they reference is pinned", + r.Run, len(r.Pins)) return b.String() } diff --git a/internal/cli/run_verify_pins_test.go b/internal/cli/run_verify_pins_test.go new file mode 100644 index 00000000..41a8ea6b --- /dev/null +++ b/internal/cli/run_verify_pins_test.go @@ -0,0 +1,138 @@ +package cli + +import ( + "database/sql" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-821 — `verify-pins` answered exit 0 on RUN-59 while every remaining judge +// step was unclaimable: the pins all matched disk, and the bytes they pinned +// referenced two fragments the run never pinned. The exit code is the part of +// this verb's answer a conductor's pre-dispatch check cannot miss, so it is +// asserted here rather than only in the engine. + +const dkt821CLIWorkflow = ` +[pipeline] +name = "dkt821-cli" +version = 1 + +[match] +kind = ["feature"] + +[[step]] +name = "judge" +executor = "judge" +emits = "change-summary" +after = [] +packet = ["contracts/judge.md"] +` + +// verifyPinsFixture activates a run against a config tree whose one contract +// declares `includes` as its `packet_includes`, and returns the config dir with +// the run ref. +// +// `fragmentAtActivation` is the whole variable: activation pins the closure it +// can SEE, so a `packet_includes` naming a file no root holds yet is pinned by +// nothing and the run activates clean — the pin set froze open, with no repin +// anywhere in the story. +func verifyPinsFixture( + t *testing.T, includes string, fragmentAtActivation bool, +) (*sql.DB, string, string) { + t.Helper() + // THE TEMP DIR COMES FIRST, BEFORE t.Setenv — `t.TempDir()` reads TMPDIR. + root := t.TempDir() + configDir := filepath.Join(root, ".docket", "config") + testsupport.Must(t, os.MkdirAll(configDir, 0o755), "creating the config dir") + t.Setenv("DOCKET_PATH", filepath.Join(root, ".docket")) + + write := func(rel, body string) { + full := filepath.Join(configDir, rel) + testsupport.Must(t, os.MkdirAll(filepath.Dir(full), 0o755), "mkdir %s", rel) + testsupport.Must(t, os.WriteFile(full, []byte(body), 0o644), "writing %s", rel) + } + write("workflows/dkt821-cli.toml", dkt821CLIWorkflow) + write("contracts/judge.md", includes+"the judge contract\n") + write("policy.toml", "opaque = \"instance policy\"\n") + if fragmentAtActivation { + write("fragments/ladder.md", "the laziness ladder\n") + } + + conn := newTestDB(t) + issueID := createIssue(t, conn, "closure", model.StatusBacklog, model.PriorityNone) + run, err := db.InsertRun(conn, 1, "test run", 0, model.NowMS()) + testsupport.Must(t, err, "InsertRun: %v", err) + testsupport.Must(t, db.AddRunIssue(conn, run.ID, issueID), "AddRunIssue") + _, err = engine.Activate(conn, run.ID, engine.ActivateOptions{NowMS: model.NowMS()}) + testsupport.Must(t, err, "Activate: %v", err) + + return conn, configDir, run.Ref() +} + +// TestVerifyPinsExitsNonZeroOnAnUnpinnedReference is AC1's exit code: the pins +// all match, so the old verb said exit 0 — a caller's only unmissable signal +// pointed the wrong way. +func TestVerifyPinsExitsNonZeroOnAnUnpinnedReference(t *testing.T) { + conn, configDir, runRef := verifyPinsFixture(t, + "---\npacket_includes:\n - fragments/ladder.md\n---\n", false) + // The fragment lands after the pin set froze — RUN-59's state exactly: the + // file is right there, and the run does not pin it. + testsupport.Must(t, os.MkdirAll(filepath.Join(configDir, "fragments"), 0o755), + "mkdir fragments") + testsupport.Must(t, os.WriteFile( + filepath.Join(configDir, "fragments/ladder.md"), + []byte("the laziness ladder\n"), 0o644), "writing the fragment") + + w, _ := bufWriter(false) + err := runRunVerifyPins(cmdWithDB(conn), runRef, w) + if err == nil { + t.Fatal("verify-pins reported clean on a run whose pinned contract " + + "references a fragment the run does not pin") + } + cerr, ok := err.(*CmdError) + if !ok { + t.Fatalf("err = %T, want *CmdError", err) + } + // The code the BLOCKED CLAIMS themselves return, which is this verb's rule: + // a caller that handles the refusal handles the prediction of it. + if cerr.Code != output.ErrValidation { + t.Errorf("code = %q, want %q", cerr.Code, output.ErrValidation) + } + for _, want := range []string{ + "contracts/judge.md", "fragments/ladder.md", "does not pin", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to say %q", err.Error(), want) + } + } + // The remedy must not send anyone to repin: nothing drifted, so a repin on + // this run is a clean no-op and adopts nothing. + if strings.Contains(err.Error(), "docket run repin") { + t.Errorf("err = %q names repin, which cannot fix a run with no drift "+ + "to adopt", err.Error()) + } +} + +// TestVerifyPinsExitsZeroOnAClosedPinSet is AC3, at the exit code that matters. +func TestVerifyPinsExitsZeroOnAClosedPinSet(t *testing.T) { + conn, _, runRef := verifyPinsFixture(t, "", false) + + w, buf := bufWriter(false) + if err := runRunVerifyPins(cmdWithDB(conn), runRef, w); err != nil { + t.Fatalf("verify-pins refuses a fully closed run: %v", err) + } + if got := buf.String(); !strings.Contains(got, "all sound") || + !strings.Contains(got, "every ref they reference is pinned") { + t.Errorf("output = %q, want it to state BOTH halves — a reader told "+ + "only \"all sound\" cannot tell whether the closure was checked", + got) + } +} diff --git a/internal/cli/schema_register.go b/internal/cli/schema_register.go index dae3adc5..c2c66705 100644 --- a/internal/cli/schema_register.go +++ b/internal/cli/schema_register.go @@ -1,6 +1,7 @@ package cli import ( + "database/sql" "encoding/json" "fmt" "os" @@ -29,7 +30,16 @@ is no second list to disagree with it. Registering the same bytes at an existing name@version again is a success and changes nothing. Registering DIFFERENT bytes is a CONFLICT: a schema decides whether a worker's payload is accepted, so a mutable one would change a run's -acceptance criteria mid-flight. Bump the version.`, +acceptance criteria mid-flight. Bump the version. + +A registry is PER PROJECT. By default the schema lands in the project the +working directory resolves to; --project registers it into one other project, +and --all-projects registers it into every project in the store. Both report +each project's own outcome, and the frozen-bytes rule is decided per project — +a conflict in one project neither cancels nor hides another project's +registration. Registering schemas store-wide is the step that comes BEFORE +sweeping a workflow corpus, since a workflow's 'payload' references resolve +against the registry of the project it is being registered into.`, Args: cobra.ExactArgs(2), RunE: func(cmd *cobra.Command, args []string) error { return runSchemaRegister(cmd, args, getWriter(cmd)) @@ -73,12 +83,47 @@ func runSchemaRegister(cmd *cobra.Command, args []string, w *output.Writer) erro return cmdErr(fmt.Errorf("encoding the ordered index: %w", err), output.ErrGeneral) } - stored, created, err := db.InsertSchema(conn, &model.Schema{ - // The registration lands in the INVOKING project's registry (DKT-20): + // The DOCUMENT is judged before the targets are: bytes that do not compile + // are wrong in every project, and naming a project in that refusal would + // suggest the document might have been right somewhere else. + targets, fannedOut, err := resolveRegistryTargets(cmd, conn) + if err != nil { + return err + } + if fannedOut { + return registerSchemaAcrossProjects( + cmd, w, conn, name, version, path, src, string(ordered), targets) + } + + stored, created, err := db.InsertSchema( + conn, schemaRow(name, version, path, src, string(ordered), getProjectID(cmd)), + model.NowMS()) + if err != nil { + return schemaErr(err) + } + + message := fmt.Sprintf("Registered %s", stored.Ref()) + if fields := stored.OrderedFields(); len(fields) > 0 { + message += fmt.Sprintf(" (ordered: %s)", joinFields(fields)) + } + if !created { + message = fmt.Sprintf("%s is already registered with these bytes", stored.Ref()) + } + w.Success(stored, message) + return nil +} + +// schemaRow builds the row to insert. Shared by the single-project path and the +// fan-out so the two can never store different columns for the same bytes. +func schemaRow( + name string, version int, path string, src []byte, ordered string, projectID int, +) *model.Schema { + return &model.Schema{ + // The registration lands in a NAMED project's registry (DKT-20): // without this the insert falls through to the column's DEFAULT 1, so // every project's registrations pile up under whichever repo holds the // default row — invisible to the very project that registered them. - ProjectID: getProjectID(cmd), + ProjectID: projectID, Name: name, Version: version, // source_path is provenance only — it records where the bytes came from @@ -86,23 +131,49 @@ func runSchemaRegister(cmd *cobra.Command, args []string, w *output.Writer) erro SourcePath: path, SourceSHA256: workflow.SHA256(src), Body: string(src), - Ordered: string(ordered), - }, model.NowMS()) - if err != nil { - return schemaErr(err) + Ordered: ordered, } +} - message := fmt.Sprintf("Registered %s", stored.Ref()) - if fields := stored.OrderedFields(); len(fields) > 0 { - message += fmt.Sprintf(" (ordered: %s)", joinFields(fields)) +// registerSchemaAcrossProjects is the --project / --all-projects path. +// +// A schema has no environment to validate against — the document compiled +// once, above, and compiling is a pure function of the bytes — so the loop is +// insert-only. A conflicting project is recorded and the sweep continues, for +// the reason the flag exists: one command is supposed to cover the store, and +// stopping at the first conflict would leave the operator to work out by hand +// which projects it had reached. +func registerSchemaAcrossProjects( + cmd *cobra.Command, w *output.Writer, conn *sql.DB, + name string, version int, path string, src []byte, ordered string, + targets []*model.Project, +) error { + report := ®istryFanoutReport{ + Operation: "schema register", + Subject: fmt.Sprintf("%s@%d", name, version), + Scope: fanoutScope(cmd), } - if !created { - message = fmt.Sprintf("%s is already registered with these bytes", stored.Ref()) + + for _, target := range targets { + stored, created, err := db.InsertSchema( + conn, schemaRow(name, version, path, src, ordered, target.ID), model.NowMS()) + if err != nil { + report.Results = append(report.Results, + registryFailureResult(target, err, schemaErr)) + continue + } + outcome := outcomeUnchanged + if created { + outcome = outcomeRegistered + } + report.Results = append(report.Results, + registrySuccessResult(target, outcome, stored.Ref())) } - w.Success(stored, message) - return nil + + return finishRegistryFanout(w, report) } func init() { + addRegistryTargetFlags(schemaRegisterCmd, "Register") schemaCmd.AddCommand(schemaRegisterCmd) } diff --git a/internal/cli/step.go b/internal/cli/step.go index d8fe76fe..e853ef9f 100644 --- a/internal/cli/step.go +++ b/internal/cli/step.go @@ -117,7 +117,19 @@ Human and vote steps are NOT claimable: they are gates, not work. A claim against one is refused naming its class. With --render, the assembled work packet is returned instead of the bundle, in -the same atomic call.`, +the same atomic call. + +With --metadata, an opaque KV bag is merged onto the step's own IN THE SAME +COMMIT as the claim. Facts a dispatcher already knows when it hands the step +out belong here rather than at completion: a step that fails or crashes never +completes, and a bag deferred to completion is a bag those steps never record. + +With --cost-multiplier, this claim accrues the step's declared expected_cost +scaled by the given factor, and the run's budget cap is checked against the +scaled cost before the claim commits. It is the dispatcher's declaration that +it resolved the step to a variant priced differently from what the definition +guessed — an escalated retry costs what it costs, at the cap, the moment it is +claimed.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { w := getWriter(cmd) @@ -145,31 +157,52 @@ the same atomic call.`, ttlMS = ttl.Milliseconds() } + metadata, _ := cmd.Flags().GetString("metadata") + + // DKT-867: the dispatcher's variant scaling. `Changed` distinguishes an + // explicit value from the unset default, so an explicit 0 — a free + // claim, which would erase an accrual the workflow author declared — + // is refusable rather than indistinguishable from "not passed". + var costMultiplier float64 + if cmd.Flags().Changed("cost-multiplier") { + costMultiplier, _ = cmd.Flags().GetFloat64("cost-multiplier") + if costMultiplier <= 0 { + return cmdErr(fmt.Errorf( + "--cost-multiplier must be positive, got %g", costMultiplier), + output.ErrValidation) + } + } + label := stepLabel(id) - // The gate-bearing claim: a step's `pre = true` gates run HERE, at - // claim, with their results in the returned bundle (§11.1, §7.6). A - // step with no pre-gates takes the identical single-transaction path. - result, err := engine.NewEngine().ClaimStepWithGates(conn, id, engine.ClaimOptions{ - Owner: owner, TTLOverride: ttlMS, NowMS: model.NowMS(), - }) - if err != nil { - return stepErr(err, label) + claimOpts := engine.ClaimOptions{ + Owner: owner, TTLOverride: ttlMS, Metadata: metadata, + CostMultiplier: costMultiplier, NowMS: model.NowMS(), } render, _ := cmd.Flags().GetBool("render") - templatePath, _ := cmd.Flags().GetString("template") if render { // `claim --render` returns the packet instead of the bundle, in the - // same call (§6.11). The claim has already committed, so this is a - // pure formatting step over what it returned. + // same call (§6.11) — and the render is part of the claim SAGA + // (DKT-804): a packet that cannot render refuses BEFORE the lease + // is taken, so a validation failure leaves the step claimable + // rather than stranding it claimed with no token issued. + templatePath, _ := cmd.Flags().GetString("template") executor, _ := cmd.Flags().GetString("executor") - packet, err := engine.RenderStepAs(conn, id, templatePath, executor, model.NowMS()) + result, packet, err := engine.NewEngine().ClaimStepRendered( + conn, id, claimOpts, templatePath, executor) if err != nil { return stepErr(err, label) } return emitClaim(w, result, packet) } + // The gate-bearing claim: a step's `pre = true` gates run HERE, at + // claim, with their results in the returned bundle (§11.1, §7.6). A + // step with no pre-gates takes the identical single-transaction path. + result, err := engine.NewEngine().ClaimStepWithGates(conn, id, claimOpts) + if err != nil { + return stepErr(err, label) + } return emitClaim(w, result, nil) }, } @@ -263,11 +296,16 @@ can, and this verb is the channel for what it observed. TOKEN-FREE, like approve and resolve: the authority is repository access plus the recorded assertion (--reason, required) that the holder is dead. Every -consequence is the expiry reap's own — the same lease-reaped event (with -data.forced and the reason), the same write-class headroom hold awaiting +scheduling consequence is the expiry reap's own — the same lease-reaped event +(with data.forced and the reason), the same write-class headroom hold awaiting --ack-reap, the same return of the step to the pool. Reaping a holder that is in fact alive carries exactly the risks a lease expiry does; assert liveness, -do not assume it.`, +do not assume it. + +Unlike an expiry, a forced reap does not consume the step's attempt budget: a +holder asserted dead is not an executor that failed, so the reaped attempt +does not count against max_attempts. The attempt number itself stands, and the +dead attempt's usage remains back-fillable against it.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { w := getWriter(cmd) @@ -345,6 +383,14 @@ interprets, converts, or routes on them. They sum per unit in ` + "`run report`" the one named by ` + "`docket config budget.unit`" + `, if any, is the one the run's cap counts. +A GATE THAT FAILED IS NAMED ON STDOUT. The gates run inside this verb, so when +one of them does not pass, the line printed here is the only report the caller +gets: ` + "`gate self-hygiene failed (exit 2); STEP-N parked waiting-human`" + `, with +the failure glyph rather than the success checkmark. --json carries the same +rows as ` + "`failed_gates`" + ` beside the step, and the envelope stays a success +envelope — the RECORDING succeeded; its subject did not. A record whose gates +all passed prints the one-line success it always has. + --gap-file (repeatable) records an out-of-scope problem the work surfaced: an auxiliary artifact of kind ` + "`gap`" + ` beside the step's declared emit, plus a backlog issue related to the step's own — same transaction, so the residue @@ -426,15 +472,113 @@ open.`, } } - verb := "Completed" - if len(gapIssues) > 0 { - verb = fmt.Sprintf("Completed, gaps filed as %s:", - strings.Join(gapIssues, ", ")) - } - return emitStepState(w, conn, id, verb) + return emitRecordState(w, conn, id, gapIssues) }, } +// emitRecordState prints what a completion actually did. +// +// A GATE THAT FAILED IS NAMED ON STDOUT (DKT-982). The completion gates ran +// synchronously inside the saga, and where one of them did not pass, this line +// is the only thing the executor that recorded ever reads. It used to be +// `✔ Completed STEP-N (waiting-human)` — no gate, no verdict, and a success +// glyph on a parked step — so RUN-63's executor answered "Record succeeded" +// over a real cargo-fmt failure and reported the whole wave green; the +// conductor found the park later, from engine reads it should not have needed. +// +// A record whose gates all passed prints exactly the line it always has. +func emitRecordState( + w *output.Writer, conn *sql.DB, id int, gapIssues []string, +) error { + failed, err := engine.FailedGates(conn, id) + if err != nil { + return stepErr(err, stepLabel(id)) + } + if len(failed) > 0 { + return emitRecordGateFailure(w, conn, id, failed, gapIssues) + } + + verb := "Completed" + if len(gapIssues) > 0 { + verb = fmt.Sprintf("Completed, gaps filed as %s:", + strings.Join(gapIssues, ", ")) + } + return emitStepState(w, conn, id, verb) +} + +// emitRecordGateFailure prints a completion whose gates did not all pass. +// +// The line names every failed gate and its exit code, says what became of the +// step, and points at the verb that holds the captured output — and it goes out +// through Writer.Outcome, so the checkmark that made this a silent failure is +// not on it. The JSON envelope stays a success envelope (the RECORDING +// succeeded) and grows `failed_gates` beside the row, the same way DKT-414's +// advisory rides beside it: a machine consumer gets the same fact as the human +// one instead of having to run `step gates` to learn it. +func emitRecordGateFailure( + w *output.Writer, conn *sql.DB, id int, + failed []engine.FailedGate, gapIssues []string, +) error { + view, err := engine.LoadStepView(conn, id, model.NowMS()) + if err != nil { + return stepErr(err, stepLabel(id)) + } + + step := view.Row.Step + noun := "gate" + if len(failed) > 1 { + noun = "gates" + } + clauses := make([]string, 0, len(failed)) + for _, g := range failed { + clauses = append(clauses, failedGateClause(g)) + } + + message := fmt.Sprintf("%s %s; %s", noun, strings.Join(clauses, ", "), + recordedStepClause(step, view.Row.Status)) + if len(gapIssues) > 0 { + message += fmt.Sprintf("; gaps filed as %s", strings.Join(gapIssues, ", ")) + } + message += fmt.Sprintf(" — `docket step gates %s` has the captured output", step) + + w.Outcome(recordedStepPayload{StepRow: view.Row, FailedGates: failed}, message) + return nil +} + +// recordedStepPayload is `step record`'s JSON shape when a gate did not pass: +// the row every completion has always emitted, plus the gates that failed. The +// embedded StepRow marshals inline, so this is additive — a consumer reading +// the row's fields reads them unchanged. +type recordedStepPayload struct { + model.StepRow + FailedGates []engine.FailedGate `json:"failed_gates"` +} + +// failedGateClause renders one failed gate: `self-hygiene failed (exit 2)`. +// +// A gate that never ran says so instead of borrowing an exit code it does not +// have — `unmatched (no exit)`, never `exit 0`, which reads as a pass (T11). +func failedGateClause(g engine.FailedGate) string { + verdict := g.Verdict + if verdict == db.GateVerdictFail { + verdict = "failed" + } + if g.Exit == nil { + return fmt.Sprintf("%s %s (no exit)", g.Gate, verdict) + } + return fmt.Sprintf("%s %s (exit %d)", g.Gate, verdict, *g.Exit) +} + +// recordedStepClause says what became of the step. A park is named as one: +// "waiting-human" is the status, and "parked" is what it MEANS to the executor +// reading the line, which is the half that was missing. +func recordedStepClause(step, status string) string { + if status == db.StepWaitingHuman { + return fmt.Sprintf("%s parked waiting-human", step) + } + return fmt.Sprintf("%s recorded (%s)", step, status) +} + var stepFailCmd = &cobra.Command{ Use: "fail STEP-N", Short: "Report a step as failed, consuming an attempt", @@ -572,7 +716,40 @@ var stepResolveCmd = &cobra.Command{ fix+review round. Distinct from retry: retry re-runs the check that reported the problem, this schedules another round of work ON the problem (DKT-237) - override-pass pass the step as though its gates had allowed it + override-pass pass the step as though its gates had allowed it. With + --batch, ALSO record one run-scoped grant per failed gate + (gate name + exit + reason — the failure signature), so later + steps in THIS run failing the same gate with the same + signature auto-pass at routing instead of parking (DKT-546). + The gates still run and a failure with a different signature + still parks; every auto-pass is event-logged against its + grant, and the grant dies with the run — a new run re-asks + +OVERRIDE-PASS NEVER EVALUATES THE STEP'S THRESHOLD (DKT-470): it records a +generic ` + "`pass`" + `, so a step the threshold interposes is skipped unconditionally, +whatever the (unevaluated) payload would have decided. On a step with such +interposed step(s) the resolution is therefore REFUSED before anything commits +unless ` + "`--drop-interposed`" + ` acknowledges the skip (DKT-861); resolve the +interposed step(s) directly if their condition should still apply. A step +whose threshold interposes nothing resolves without the flag, exactly as +always. + +--batch IS A STANDING AUTHORIZATION, NOT A ONE-STEP ONE. It covers every later +step of the run whose same gate fails the same way — INCLUDING STEPS THAT DO +NOT EXIST WHEN YOU GRANT IT, because a fix round mints new steps inside the +same run (DKT-734). Granting on a parked ` + "`fix@7`" + ` therefore covers ` + "`fix@8`" + ` and +` + "`fix@9`" + ` if later rounds fail the same gate the same way. Read the reach before +granting: one ruling can auto-pass N steps, and it will not re-ask. + +What it does NOT do is make the gate advisory. Every round still runs it, and +a round that fails with a DIFFERENT exit or reason parks for a fresh decision +no matter how many rounds the grant has already covered. + +To audit what one grant actually spent: ` + "`docket events list`" + ` shows the ruling +once as ` + "`gate-override-granted`" + ` with ` + "`detail=GATE#ID`" + `, and each application as +` + "`step-batch-overridden`" + ` with ` + "`detail=ID`" + ` on the step it passed. Same id both +sides — N ` + "`step-batch-overridden`" + ` events against ONE ` + "`gate-override-granted`" + ` is +one authorization spent N times, not N authorizations. ` + "`retry`" + ` resets the STEP's attempt budget. That is a different counter from the issue-level attempt trail, which is monotonic and never reset. It also @@ -605,7 +782,24 @@ never arrives. A HELD CLUSTER parked by a vote that did not pass is answered with ` + "`step approve`" + ` / ` + "`step reject`" + ` instead; ` + "`retry`" + ` refuses there for the same reason it refuses on a rejected hold, because the same tally would be read -again.`, +again. + +A ` + "`type=\"vote\"`" + ` step whose OWN proposal has been decided — approved or +rejected — refuses ` + "`retry`" + ` for that same reason. The proposal is keyed to +the step instance, so returning the step to the pool re-reads the identical +tally and routes to the identical place; no second ballot opens and no cast +changes. Reach for ` + "`fix-round`" + `, which authorizes another round of work on +the reported problem and mints a fresh vote at a new ordinal — a NEW proposal — +or for ` + "`override-pass`" + `, ` + "`skip`" + `, or ` + "`abandon-issue`" + `. ` + "`retry`" + ` +stays available on a vote whose proposal is still open, which is the case R11's +exception exists for: a quorum that never arrives. + +THE -m NOTE IS THE STEERING CHANNEL. A note on ` + "`retry`" + ` or ` + "`rerun-gates`" + ` +renders in the same step's next packet as its == RESOLUTION section; a note on +` + "`fix-round`" + ` is stamped onto every step the new round mints, so the remedy +you authorized the round for renders in each of its packets. Steering left as +an issue comment NEVER reaches a packet, and a mid-run description edit does +not either — packets read the description snapshot frozen at activation.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runStepResolve(cmd, args, getWriter(cmd)) @@ -626,22 +820,36 @@ func runStepResolve(cmd *cobra.Command, args []string, w *output.Writer) error { output.ErrValidation) } note, _ := cmd.Flags().GetString("note") + batch, _ := cmd.Flags().GetBool("batch") + dropInterposed, _ := cmd.Flags().GetBool("drop-interposed") label := stepLabel(id) e := engine.NewEngine() - // DKT-470: override-pass records a generic "pass" and never evaluates the - // step's threshold, so an interposed step it names is skipped no matter - // what the (unevaluated) payload would have decided. Named BEFORE the - // resolution commits, so the warning is the blast radius the operator is - // approving rather than a report of damage already done. - if as == engine.ResolveOverridePass { + // DKT-470/DKT-861: override-pass records a generic "pass" and never + // evaluates the step's threshold, so an interposed step it names is + // skipped no matter what the (unevaluated) payload would have decided. + // Without --drop-interposed the ENGINE refuses such a resolution before + // anything commits, carrying this same sentence; with it, the warning is + // still printed BEFORE the resolution commits, so it names the blast + // radius the operator acknowledged rather than reporting damage already + // done. + if as == engine.ResolveOverridePass && dropInterposed { for _, warning := range engine.OverridePassSkipsInterposedTargets(conn, id) { w.Warn("%s", warning) } } - if err := e.ResolveStep(conn, id, as, note, model.NowMS()); err != nil { + resolve := e.ResolveStep + if batch { + resolve = e.ResolveStepBatch + } + if dropInterposed { + resolve = func(conn *sql.DB, stepID int, as, note string, nowMS int64) error { + return e.ResolveStepDropInterposed(conn, stepID, as, note, batch, nowMS) + } + } + if err := resolve(conn, id, as, note, model.NowMS()); err != nil { return stepErr(err, label) } @@ -865,6 +1073,16 @@ var stepRenderCmd = &cobra.Command{ Short: "Render a step's context bundle into a work packet", Long: `Render a step's context bundle through a template. +THE PACKET DRAWS FROM EXACTLY THESE SURFACES: the step row and its pinned +workflow definition, the issue's description and title/kind/labels/scope AS +SNAPSHOTTED AT ACTIVATION, the recorded input artifacts the step declares, the +pin list, the step's declared packet files, and the step's own routing record +(rendered as == RESOLUTION). Issue COMMENTS are never rendered, and a mid-run +edit to the issue's description never renders either — the packet reads the +activation-time snapshot. To steer a step's next execution, put the words in +` + "`step resolve ... -m`" + `: a retry note renders on the same step, and a +fix-round note renders in every packet of the round it authorizes. + Without --template the shipped default is used; it ships in the binary, so it cannot drift. With --template F, if the run PINNED that path, the file's bytes are verified against the pin and a mismatch is refused with CONFLICT naming @@ -1162,6 +1380,14 @@ func init() { stepClaimCmd.Flags().String("executor", "", "Resolved executor hint for --render's {executor} substitution "+ "(default: the step's declared hint)") + stepClaimCmd.Flags().String("metadata", "", + "Opaque metadata JSON, merged onto the step's own IN THE CLAIM: what "+ + "the dispatcher knows now, recorded before the work can fail") + stepClaimCmd.Flags().Float64("cost-multiplier", 0, + "Scale this claim's budget accrual: the step's declared expected_cost "+ + "times this, checked at the cap and recorded with the claim — for a "+ + "dispatcher that resolved the step to a pricier (or cheaper) variant "+ + "than the definition priced (default: 1, the declared cost verbatim)") stepCompleteCmd.Flags().String("artifact-file", "", "File holding the artifact body (required)") stepCompleteCmd.Flags().String("payload-file", "", "File holding the JSON payload") @@ -1183,8 +1409,20 @@ func init() { stepRejectCmd.Flags().String("note", "", "Why the gate was rejected") stepResolveCmd.Flags().String("as", "", - "retry | skip | abandon-issue | override-pass | fix-round") + "retry | rerun-gates | skip | abandon-issue | override-pass | fix-round") stepResolveCmd.Flags().String("note", "", "Why the step was resolved this way") + stepResolveCmd.Flags().Bool("batch", false, + "With --as override-pass: also record one run-scoped grant per failed "+ + "gate, so later steps in THIS run failing the same gate with the "+ + "same exit and reason auto-pass instead of parking (DKT-546) — "+ + "including steps later fix rounds mint, which do not exist when "+ + "you grant it (DKT-734)") + stepResolveCmd.Flags().Bool("drop-interposed", false, + "With --as override-pass on a step whose threshold interposes other "+ + "step(s): acknowledge that the generic pass skips them without "+ + "evaluating the threshold (DKT-861). Without it such a resolution "+ + "is refused before anything commits; a step with no interposed "+ + "step(s) never needs it") stepContextCmd.Flags().Bool("meta", false, "Report per-section byte counts alongside the bundle") stepRenderCmd.Flags().String("template", "", "Template file; defaults to the shipped one") diff --git a/internal/cli/step_gates_test.go b/internal/cli/step_gates_test.go index fd375a50..e144b8c5 100644 --- a/internal/cli/step_gates_test.go +++ b/internal/cli/step_gates_test.go @@ -289,6 +289,77 @@ func TestSkippedIsAVisibleRollupColumn(t *testing.T) { } } +// TestGateRollupCountsPreGateRows is DKT-862's rollup half. +// +// `run report`'s Gates tally rendered `ac-commands: pass 0, fail 1` whether the +// failure BLOCKED the step or was advisory. A `pre = true` gate never routes, +// so the two are different facts and the section had no way to tell them apart: +// the rollup counted four verdicts and nothing about WHEN the row was produced. +// +// `Pre` counts ROWS and OVERLAPS the verdict columns, exactly as `Stub` does — +// it describes the phase that produced the row, not what the row decided. +func TestGateRollupCountsPreGateRows(t *testing.T) { + conn := newTestDB(t) + runID := activatedRunForNext(t, conn) + + exitZero, exitTwo := 0, 2 + rows := []db.GateResultRow{ + // RUN-61's shape: a pre-gate that failed and routed nothing. + {RunID: runID, StepID: 1, Gate: "ac-commands", Ordinal: 0, Exit: &exitTwo, + Verdict: db.GateVerdictFail, Pre: true, CreatedAtMS: model.NowMS()}, + // A blocking gate that failed identically. + {RunID: runID, StepID: 1, Gate: "build", Ordinal: 0, Exit: &exitTwo, + Verdict: db.GateVerdictFail, CreatedAtMS: model.NowMS()}, + // One NAME, both declarations — the case the tally cannot mark whole. + {RunID: runID, StepID: 1, Gate: "render-verify", Ordinal: 0, Exit: &exitZero, + Verdict: db.GateVerdictPass, Pre: true, CreatedAtMS: model.NowMS()}, + {RunID: runID, StepID: 1, Gate: "render-verify", Ordinal: 1, Exit: &exitTwo, + Verdict: db.GateVerdictFail, CreatedAtMS: model.NowMS()}, + } + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + for _, r := range rows { + testsupport.Must(t, db.InsertGateResultTx(tx, r), "InsertGateResultTx: %v", err) + } + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + + counts, err := db.GateRollup(conn, runID) + testsupport.Must(t, err, "GateRollup: %v", err) + + byName := map[string]db.VerdictCount{} + for _, c := range counts { + byName[c.Name] = c + } + if got := byName["ac-commands"]; got.Pre != 1 || got.Fail != 1 { + t.Errorf("ac-commands rolled up as pre %d fail %d, want pre 1 fail 1 — "+ + "an advisory failure must be countable as one", got.Pre, got.Fail) + } + // The marker must not spread: a blocking gate that failed the same way is + // the row the report has to keep distinguishable. + if got := byName["build"]; got.Pre != 0 { + t.Errorf("build rolled up as pre %d, want 0", got.Pre) + } + if got := byName["render-verify"]; got.Pre != 1 || got.Pass+got.Fail != 2 { + t.Errorf("render-verify rolled up as pre %d over %d rows, want pre 1 "+ + "of 2 — the tally must be able to say which half was advisory", + got.Pre, got.Pass+got.Fail) + } + + // Actions have no `pre` column, and the shared rollup body must return an + // honest zero for them rather than failing on a column that table lacks — + // the same asymmetry `stub_entry` already has. + actions, err := db.ActionRollup(conn, runID) + if err != nil { + t.Fatalf("ActionRollup failed after the gate rollup gained a pre "+ + "column: %v", err) + } + for _, a := range actions { + if a.Pre != 0 { + t.Errorf("action %q rolled up as pre %d, want 0", a.Name, a.Pre) + } + } +} + // TestVoteUsageCoverageMakesSilenceVisible is DKT-257. // // The vote_usage ledger has existed since v14 and held ZERO rows for an entire @@ -351,6 +422,43 @@ func TestVoteUsageCoverageIsZeroWithNoCasts(t *testing.T) { } } +// TestSilentVoteSeatsAreNamed is DKT-733: the coverage count said 12 of 57 +// seats reported nothing and nothing anywhere said WHICH twelve, so the +// backfill verb built to close the gap could not be aimed. The enumeration +// goes through the SAME membership builder as the count it explains. +func TestSilentVoteSeatsAreNamed(t *testing.T) { + conn := newTestDB(t) + + proposalID, err := db.CreateProposalIdempotent(conn, &model.Proposal{ + Description: "verify the change", Status: model.ProposalStatusOpen, + Criticality: model.CriticalityMedium, RequiredVoters: 2, Threshold: 0.67, + }, "vote-step:1:1:verify-tribunal@0") + testsupport.Must(t, err, "creating the proposal: %v", err) + + castSeat(t, conn, proposalID, "tribunal-architecture") // stays silent + loud := castSeat(t, conn, proposalID, "tribunal-security") + + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + testsupport.Must(t, db.InsertVoteUsageTx(tx, loud, "output_tokens", 39500, "", model.NowMS()), + "InsertVoteUsageTx: %v", err) + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + + got, err := db.SilentVoteSeatsFor(conn, db.ScopeVoteCreate, "vote-step:1:") + testsupport.Must(t, err, "SilentVoteSeatsFor: %v", err) + + if len(got) != 1 { + t.Fatalf("silent seats = %+v, want exactly the one quiet cast", got) + } + if got[0].ProposalID != proposalID || got[0].Voter != "tribunal-architecture" { + t.Errorf("silent seat = %+v, want tribunal-architecture on proposal %d", + got[0], proposalID) + } + if got[0].Role != "judge" { + t.Errorf("silent seat role = %q, want the cast's recorded role", got[0].Role) + } +} + // --- DKT-425: what `step gates --json` returns by default ------------------- // hygieneLog stands in for the row DKT-425 was opened about: a PASSING gate diff --git a/internal/cli/step_resolve_interposed_test.go b/internal/cli/step_resolve_interposed_test.go new file mode 100644 index 00000000..97de4b35 --- /dev/null +++ b/internal/cli/step_resolve_interposed_test.go @@ -0,0 +1,227 @@ +package cli + +import ( + "bytes" + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/spf13/cobra" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-861 at the CLI boundary. internal/engine/dkt861_test.go proves the +// refusal, the acknowledgment, and the no-interposed regression guard against +// resolveStep directly; what this file asserts is the VERB — that +// `step resolve --as override-pass` without --drop-interposed surfaces the +// engine's refusal as a validation error carrying the DKT-470 sentence, that +// `--drop-interposed` reaches the engine and prints that same sentence as a +// stderr warning BEFORE the resolution commits, and that a step with no +// interposed dependents still resolves with no flag and no warning. + +// interposedResolveSrc is dkt470_test.go's shape: a verify step whose +// threshold interposes a tribunal vote, parked `waiting-human` on its own +// single-attempt failure. +const interposedResolveSrc = ` +[pipeline] +name = "cli-interpose" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "verify" +executor = "verify" +emits = "report" +max_attempts = 1 +on_fail = "waiting-human" +threshold = { "tribunal" = "any(status == blocked)" } + +[[step]] +name = "tribunal" +after = ["verify"] +type = "vote" +voters = ["seat-a", "seat-b", "seat-c"] +vote_rule = "majority" +on_fail = "waiting-human" +` + +// plainResolveSrc is the same park with NO threshold at all — the AC3 shape. +const plainResolveSrc = ` +[pipeline] +name = "cli-plain" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "verify" +executor = "verify" +emits = "report" +max_attempts = 1 +on_fail = "waiting-human" +` + +// parkedResolveRun registers src, activates a one-task run over it, and fails +// verify@0 into `waiting-human` — the state `resolve --as override-pass` is +// reached for. +func parkedResolveRun(t *testing.T, conn *sql.DB, src string) (stepID int) { + t.Helper() + + registerForRun(t, conn, src) + issueID, err := db.CreateIssue(conn, &model.Issue{ + Title: "park me", Description: "a body", + Status: model.StatusBacklog, Priority: model.PriorityNone, + Kind: model.IssueKindTask, + }, nil, nil) + testsupport.Must(t, err, "creating issue: %v", err) + run, err := db.InsertRun(conn, 1, "", 0, model.NowMS()) + testsupport.Must(t, err, "starting run: %v", err) + testsupport.Must(t, db.AddRunIssue(conn, run.ID, issueID), + "adding issue to run: %v", err) + _, err = engine.Activate(conn, run.ID, engine.ActivateOptions{NowMS: model.NowMS()}) + testsupport.Must(t, err, "activate: %v", err) + + err = conn.QueryRow( + `SELECT id FROM steps WHERE instance = 'verify@0'`).Scan(&stepID) + testsupport.Must(t, err, "finding verify@0: %v", err) + + claim, err := engine.ClaimStep(conn, stepID, + engine.ClaimOptions{Owner: "w", NowMS: model.NowMS()}) + testsupport.Must(t, err, "claim: %v", err) + testsupport.Must(t, engine.NewEngine().FailStep( + conn, stepID, claim.Token, "gate failed", "", model.NowMS()), "fail: %v", err) + + var status string + err = conn.QueryRow( + `SELECT status FROM steps WHERE id = ?`, stepID).Scan(&status) + testsupport.Must(t, err, "reading status: %v", err) + if status != string(db.StepWaitingHuman) { + t.Fatalf("premise: verify@0 = %q, want %q", status, db.StepWaitingHuman) + } + return stepID +} + +// resolveCmdWithDB builds a `step resolve` command with its real flag names. +func resolveCmdWithDB(conn *sql.DB) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("as", "", "") + cmd.Flags().String("note", "", "") + cmd.Flags().Bool("batch", false, "") + cmd.Flags().Bool("drop-interposed", false, "") + return cmd +} + +func stepStatusByInstance(t *testing.T, conn *sql.DB, instance string) string { + t.Helper() + var status string + err := conn.QueryRow( + `SELECT status FROM steps WHERE instance = ?`, instance).Scan(&status) + testsupport.Must(t, err, "reading status of %s: %v", instance, err) + return status +} + +// TestResolveOverridePassRefusesInterposedWithoutFlag: without the +// acknowledgment the verb refuses as a validation error, the refusal carries +// the DKT-470 sentence and names the flag, and nothing mutates. +func TestResolveOverridePassRefusesInterposedWithoutFlag(t *testing.T) { + conn := newTestDB(t) + stepID := parkedResolveRun(t, conn, interposedResolveSrc) + + cmd := resolveCmdWithDB(conn) + testsupport.Must(t, cmd.Flags().Set("as", "override-pass"), "set --as: %v", nil) + outBuf, errBuf := &bytes.Buffer{}, &bytes.Buffer{} + w := &output.Writer{Stdout: outBuf, Stderr: errBuf} + + err := runStepResolve(cmd, []string{fmt.Sprintf("STEP-%d", stepID)}, w) + if err == nil { + t.Fatal("override-pass with interposed dependents and no " + + "--drop-interposed was accepted") + } + if got := codeOf(t, err); got != output.ErrValidation { + t.Errorf("code = %s, want %s", got, output.ErrValidation) + } + for _, want := range []string{ + "tribunal", "will NOT be routed", "--drop-interposed", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("refusal = %q, want it to contain %q", err, want) + } + } + if got := stepStatusByInstance(t, conn, "verify@0"); got != string(db.StepWaitingHuman) { + t.Errorf("verify@0 = %q after the refusal, want it still %q", + got, db.StepWaitingHuman) + } + if got := stepStatusByInstance(t, conn, "tribunal@0"); got != string(db.StepPending) { + t.Errorf("tribunal@0 = %q after the refusal, want it still %q", + got, db.StepPending) + } +} + +// TestResolveOverridePassDropInterposedWarnsAndProceeds: the acknowledged +// resolution prints the same DKT-470 sentence as a warning — emitted before +// the resolution commits — and then behaves exactly as override-pass always +// has: verify passes, the interposed tribunal is skipped. +func TestResolveOverridePassDropInterposedWarnsAndProceeds(t *testing.T) { + conn := newTestDB(t) + stepID := parkedResolveRun(t, conn, interposedResolveSrc) + + cmd := resolveCmdWithDB(conn) + testsupport.Must(t, cmd.Flags().Set("as", "override-pass"), "set --as: %v", nil) + testsupport.Must(t, cmd.Flags().Set("drop-interposed", "true"), + "set --drop-interposed: %v", nil) + outBuf, errBuf := &bytes.Buffer{}, &bytes.Buffer{} + w := &output.Writer{Stdout: outBuf, Stderr: errBuf} + + err := runStepResolve(cmd, []string{fmt.Sprintf("STEP-%d", stepID)}, w) + testsupport.Must(t, err, "acknowledged override-pass: %v", err) + + for _, want := range []string{"tribunal", "will NOT be routed"} { + if !strings.Contains(errBuf.String(), want) { + t.Errorf("stderr = %q, want the pre-mutation warning to contain %q", + errBuf.String(), want) + } + } + if got := stepStatusByInstance(t, conn, "verify@0"); got != string(db.StepDone) { + t.Errorf("verify@0 = %q, want %q", got, db.StepDone) + } + if got := stepStatusByInstance(t, conn, "tribunal@0"); got != string(db.StepSkipped) { + t.Errorf("tribunal@0 = %q, want %q — the acknowledged pass still "+ + "skips the interposed step, exactly as today", got, db.StepSkipped) + } +} + +// TestResolveOverridePassNoInterposedUnchanged is AC3 at the verb: a step +// with no interposed dependents resolves with no flag, no warning, and no +// refusal — exactly the bytes it always produced. +func TestResolveOverridePassNoInterposedUnchanged(t *testing.T) { + conn := newTestDB(t) + stepID := parkedResolveRun(t, conn, plainResolveSrc) + + cmd := resolveCmdWithDB(conn) + testsupport.Must(t, cmd.Flags().Set("as", "override-pass"), "set --as: %v", nil) + outBuf, errBuf := &bytes.Buffer{}, &bytes.Buffer{} + w := &output.Writer{Stdout: outBuf, Stderr: errBuf} + + err := runStepResolve(cmd, []string{fmt.Sprintf("STEP-%d", stepID)}, w) + testsupport.Must(t, err, "override-pass with no interposed dependents: %v", err) + + if errBuf.Len() != 0 { + t.Errorf("stderr = %q, want no warning for a step with no interposed "+ + "dependents", errBuf.String()) + } + if !strings.Contains(outBuf.String(), "Resolved") { + t.Errorf("stdout = %q, want the ordinary resolution message", outBuf.String()) + } + if got := stepStatusByInstance(t, conn, "verify@0"); got != string(db.StepDone) { + t.Errorf("verify@0 = %q, want %q", got, db.StepDone) + } +} diff --git a/internal/cli/step_test.go b/internal/cli/step_test.go index 18e84428..8b8e06b3 100644 --- a/internal/cli/step_test.go +++ b/internal/cli/step_test.go @@ -2,12 +2,17 @@ package cli import ( "database/sql" + "encoding/json" + "os" + "strings" "testing" "github.com/ALT-F4-LLC/docket/internal/db" "github.com/ALT-F4-LLC/docket/internal/engine" "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/render" + "github.com/ALT-F4-LLC/docket/internal/testsupport" ) // The §6.9 refusal matrix AT THE CLI BOUNDARY. @@ -261,3 +266,248 @@ func TestStepRecordIsCompleteAlias(t *testing.T) { "object as `docket step complete` (%p)", found, stepCompleteCmd) } } + +// DKT-982 — `step record`'s stdout when a completion gate failed. +// +// The saga runs the completion gates synchronously and parks the step when one +// fails, and the line the recording verb printed was `✔ Completed STEP-N +// (waiting-human)`: no gate, no verdict, a success glyph on a park. RUN-63's +// executor read that line, answered "Record succeeded", and reported the wave +// green over a real cargo-fmt failure. These drive emitRecordState — the whole +// of what the RunE does after the saga returns — over recorded gate rows. + +// recordFixture activates a run and returns the connection and the executor +// step every case here records against. +func recordFixture(t *testing.T, conn *sql.DB) int { + t.Helper() + activatedRunForNext(t, conn) + var id int + err := conn.QueryRow(`SELECT id FROM steps WHERE instance = 'first@0'`).Scan(&id) + testsupport.Must(t, err, "finding the executor step: %v", err) + return id +} + +// seedGateRows records gate results against a step and sets the status the +// routing would have left it in — the state the printer reads. +func seedGateRows( + t *testing.T, conn *sql.DB, stepID int, status string, rows ...db.GateResultRow, +) { + t.Helper() + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + + tx, err := conn.Begin() + testsupport.Must(t, err, "begin: %v", err) + defer func() { _ = tx.Rollback() }() + for _, r := range rows { + r.RunID, r.StepID, r.CreatedAtMS = step.RunID, stepID, model.NowMS() + testsupport.Must(t, db.InsertGateResultTx(tx, r), "inserting a gate row: %v", err) + } + testsupport.Must(t, + db.SetStepStatusTx(tx, stepID, status, model.NowMS(), model.NowMS()), + "setting the step status: %v", err) + testsupport.Must(t, tx.Commit(), "commit: %v", err) +} + +// colorfulTerminal makes render.ColorsEnabled() true, which is the condition +// under which a glyph is printed at all — the RUN-63 terminal's condition. The +// variable must be genuinely ABSENT: ColorsEnabled uses LookupEnv, so +// NO_COLOR="" would still disable colors. +func colorfulTerminal(t *testing.T) { + t.Helper() + if prev, ok := os.LookupEnv("NO_COLOR"); ok { + testsupport.Must(t, os.Unsetenv("NO_COLOR"), "unsetting NO_COLOR: %v", nil) + t.Cleanup(func() { _ = os.Setenv("NO_COLOR", prev) }) + } + t.Setenv("TERM", "xterm-256color") + if !render.ColorsEnabled() { + t.Fatal("premise: colors are disabled, so neither glyph would be printed") + } +} + +// TestStepRecordNamesTheFailedGateAndItsExit is acceptance criteria 1 and 2: +// the failed gate's NAME and EXIT CODE are on stdout, and the park is not +// wearing a checkmark. +func TestStepRecordNamesTheFailedGateAndItsExit(t *testing.T) { + colorfulTerminal(t) + + conn := newTestDB(t) + stepID := recordFixture(t, conn) + exit := 2 + seedGateRows(t, conn, stepID, db.StepWaitingHuman, + db.GateResultRow{Gate: "build", Verdict: db.GateVerdictPass, Exit: intPtr(0)}, + db.GateResultRow{Gate: "tests", Verdict: db.GateVerdictPass, Exit: intPtr(0)}, + db.GateResultRow{Gate: "self-hygiene", Verdict: db.GateVerdictFail, Exit: &exit}) + + w, buf := bufWriter(false) + testsupport.Must(t, emitRecordState(w, conn, stepID, nil), "emitRecordState: %v", nil) + + out := buf.String() + for _, want := range []string{ + "self-hygiene failed (exit 2)", + "parked waiting-human", + model.FormatStepID(stepID), + "✘", + } { + if !strings.Contains(out, want) { + t.Errorf("stdout = %q, want it to carry %q", out, want) + } + } + if strings.Contains(out, "✔") { + t.Errorf("stdout = %q — the park is behind a success glyph, which is the "+ + "misread DKT-982 was filed for", out) + } + // The gates that PASSED are not on the line: three of them named at every + // record is how the one that matters gets skimmed past. + if strings.Contains(out, "build") || strings.Contains(out, "tests") { + t.Errorf("stdout = %q, want only the gate that failed", out) + } +} + +// TestStepRecordNamesEveryFailedGate covers the plural case and the gate that +// never ran: `unmatched` has no exit code, and printing `exit 0` for a process +// that did not exist would read as a pass (T11). +func TestStepRecordNamesEveryFailedGate(t *testing.T) { + t.Setenv("NO_COLOR", "1") + + conn := newTestDB(t) + stepID := recordFixture(t, conn) + exit := 1 + seedGateRows(t, conn, stepID, db.StepWaitingHuman, + db.GateResultRow{Gate: "secret-scan", Verdict: db.GateVerdictUnmatched, + Reason: "no trust entry matched"}, + db.GateResultRow{Gate: "build", Verdict: db.GateVerdictFail, Exit: &exit}) + + w, buf := bufWriter(false) + testsupport.Must(t, emitRecordState(w, conn, stepID, nil), "emitRecordState: %v", nil) + + out := buf.String() + want := "gates build failed (exit 1), secret-scan unmatched (no exit); " + + model.FormatStepID(stepID) + " parked waiting-human — `docket step gates " + + model.FormatStepID(stepID) + "` has the captured output\n" + if out != want { + t.Errorf("stdout = %q\nwant %q", out, want) + } +} + +// TestStepRecordJSONCarriesTheFailedGates is the machine channel: the envelope +// stays a SUCCESS envelope — the recording did succeed — and the failed gates +// ride beside the step row, so a --json consumer learns the same fact without a +// second command. +func TestStepRecordJSONCarriesTheFailedGates(t *testing.T) { + conn := newTestDB(t) + stepID := recordFixture(t, conn) + exit := 2 + seedGateRows(t, conn, stepID, db.StepWaitingHuman, + db.GateResultRow{Gate: "self-hygiene", Verdict: db.GateVerdictFail, Exit: &exit}) + + w, buf := bufWriter(true) + testsupport.Must(t, emitRecordState(w, conn, stepID, nil), "emitRecordState: %v", nil) + + var envelope struct { + OK bool `json:"ok"` + Data struct { + Step string `json:"step"` + Status string `json:"status"` + FailedGates []engine.FailedGate `json:"failed_gates"` + } `json:"data"` + Message string `json:"message"` + } + testsupport.Must(t, json.Unmarshal(buf.Bytes(), &envelope), "unmarshal: %v", nil) + + if !envelope.OK { + t.Error("ok = false; the recording succeeded — its subject did not") + } + if envelope.Data.Step != model.FormatStepID(stepID) { + t.Errorf("data.step = %q, want the row the completion has always emitted", + envelope.Data.Step) + } + if envelope.Data.Status != db.StepWaitingHuman { + t.Errorf("data.status = %q, want %q", envelope.Data.Status, db.StepWaitingHuman) + } + if len(envelope.Data.FailedGates) != 1 { + t.Fatalf("data.failed_gates = %v, want the one gate that failed", + envelope.Data.FailedGates) + } + got := envelope.Data.FailedGates[0] + if got.Gate != "self-hygiene" || got.Exit == nil || *got.Exit != 2 { + t.Errorf("data.failed_gates[0] = %+v, want self-hygiene at exit 2", got) + } + if !strings.Contains(envelope.Message, "self-hygiene failed (exit 2)") { + t.Errorf("message = %q, want the failed gate named", envelope.Message) + } +} + +// TestStepRecordKeepsItsSuccessLineWhenGatesPass is acceptance criterion 3: a +// clean record is byte-for-byte the line it has always printed, checkmark +// included. Everything above is a NEW branch, and this is the assertion that +// keeps it one. +func TestStepRecordKeepsItsSuccessLineWhenGatesPass(t *testing.T) { + colorfulTerminal(t) + + conn := newTestDB(t) + stepID := recordFixture(t, conn) + seedGateRows(t, conn, stepID, db.StepDone, + db.GateResultRow{Gate: "build", Verdict: db.GateVerdictPass, Exit: intPtr(0)}, + // A PRE-gate that failed is an input to the step, not a judgment of it + // (PG4) — it routed nothing, so it must not turn a clean record into a + // reported failure. + db.GateResultRow{Gate: "pre-scan", Verdict: db.GateVerdictFail, + Exit: intPtr(3), Pre: true}) + + w, buf := bufWriter(false) + testsupport.Must(t, emitRecordState(w, conn, stepID, nil), "emitRecordState: %v", nil) + + want := "✔ Completed " + model.FormatStepID(stepID) + " (done)\n" + if buf.String() != want { + t.Errorf("stdout = %q, want %q", buf.String(), want) + } +} + +func intPtr(n int) *int { return &n } + +// TestStepClaimCommandWritesTheMetadataFlag drives the REAL `step claim` RunE +// over its REAL flag set and reads the step back out of the store (DKT-592). +// +// It exists because this is the exact join DKT-68 lost: a flag that parses, an +// option field that is documented as merged, and nothing in between reading +// it. Everything either side is pinned in internal/engine, so a RunE that +// looked up a flag name nobody registers — or registered one nobody reads — +// would drop every dispatcher's bag with the whole engine suite green. +// +// The flag set comes from `stepClaimCmd` itself rather than being re-declared +// here, so the registration is part of what is under test. +func TestStepClaimCommandWritesTheMetadataFlag(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + _, err := engine.Activate(conn, runID, engine.ActivateOptions{NowMS: model.NowMS()}) + testsupport.Must(t, err, "activate: %v", err) + + var stepID int + err = conn.QueryRow( + `SELECT id FROM steps WHERE run_id = ? AND step_name = 'first'`, runID, + ).Scan(&stepID) + testsupport.Must(t, err, "reading the claimable step: %v", err) + + cmd := cmdWithDB(conn) + cmd.Flags().AddFlagSet(stepClaimCmd.Flags()) + // AddFlagSet shares the flag VALUES with the package-level command, so the + // two set here are put back before any later test reads them. + t.Cleanup(func() { + _ = stepClaimCmd.Flags().Set("owner", "") + _ = stepClaimCmd.Flags().Set("metadata", "") + }) + testsupport.Must(t, cmd.Flags().Set("owner", "worker"), "setting --owner: %v", err) + testsupport.Must(t, cmd.Flags().Set("metadata", `{"tier_requested":"a"}`), + "setting --metadata: %v", err) + + err = stepClaimCmd.RunE(cmd, []string{model.FormatStepID(stepID)}) + testsupport.Must(t, err, "step claim --metadata: %v", err) + + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + if !strings.Contains(step.Metadata, `"tier_requested":"a"`) { + t.Errorf("metadata = %q, want the claim's bag — the flag was parsed and dropped", + step.Metadata) + } +} diff --git a/internal/cli/trust.go b/internal/cli/trust.go index f8782dc5..7369a82a 100644 --- a/internal/cli/trust.go +++ b/internal/cli/trust.go @@ -101,6 +101,14 @@ disclosure.`, // both directions. It changes nothing about how the command runs. cmd.Flags().Bool("stub", false, "this is a placeholder, not the check its name implies; its passes are marked hollow in reports") + // --stub-reason records the DECISION behind the placeholder (DKT-607): why + // no real check exists yet and which issue tracks replacing it. The reason + // surfaces wherever the stub does — `trust list`, the activation gate + // preflight, the grant event — so the decision is discoverable without a + // tribunal transcript. Requires --stub; a reason on a real check is a + // contradiction and is refused. + cmd.Flags().String("stub-reason", "", + "why this entry is a stub and which issue tracks replacing it, e.g. \"no scanner selected; tracked by DKT-607\" (requires --stub)") cmd.Flags().String("timeout", "", "per-command timeout (e.g. 90s, 10m); defaults to 5m") // --network DECLARES a requirement; it does not grant one. Core cannot @@ -162,6 +170,7 @@ type trustAddResult struct { Tree bool `json:"tree"` Flaky bool `json:"flaky"` Stub bool `json:"stub"` + StubReason string `json:"stub_reason,omitempty"` Timeout string `json:"timeout,omitempty"` Idempotent bool `json:"idempotent"` Warnings []string `json:"warnings,omitempty"` @@ -179,6 +188,7 @@ type trustEntryView struct { Tree bool `json:"tree"` Flaky bool `json:"flaky"` Stub bool `json:"stub"` + StubReason string `json:"stub_reason,omitempty"` Timeout string `json:"timeout,omitempty"` AddedAtMS int64 `json:"added_at_ms"` } @@ -231,6 +241,7 @@ func runTrustAdd(cmd *cobra.Command, args []string) error { tree, _ := cmd.Flags().GetBool("tree") flaky, _ := cmd.Flags().GetBool("flaky") stub, _ := cmd.Flags().GetBool("stub") + stubReason, _ := cmd.Flags().GetString("stub-reason") timeout, _ := cmd.Flags().GetString("timeout") network, _ := cmd.Flags().GetStringSlice("network") @@ -252,7 +263,7 @@ func runTrustAdd(cmd *cobra.Command, args []string) error { res, err := trust.Add(trust.AddRequest{ Name: name, Argv: argv, RepoRoot: repoRoot, Global: global, Prefix: prefix, ReRunnable: reRunnable, Tree: tree, Flaky: flaky, - Stub: stub, + Stub: stub, StubReason: stubReason, Timeout: timeout, Network: network, NowMS: time.Now().UnixMilli(), OnChange: recorder.record, }) @@ -289,7 +300,7 @@ func runTrustAdd(cmd *cobra.Command, args []string) error { Name: res.Entry.Name, Argv: res.Entry.Argv, ArgvSHA256: res.Entry.ArgvSHA256, Repo: res.Entry.Repo, Global: res.Entry.Global, Prefix: res.Entry.Prefix, ReRunnable: res.Entry.ReRunnable, Tree: res.Entry.Tree, Flaky: res.Entry.Flaky, - Stub: res.Entry.Stub, + Stub: res.Entry.Stub, StubReason: res.Entry.StubReason, Network: res.Entry.Network, Timeout: res.Entry.Timeout, Idempotent: res.Idempotent, Warnings: warnings, } @@ -341,7 +352,7 @@ func runTrustList(cmd *cobra.Command, args []string) error { Name: e.Name, Argv: e.Argv, ArgvSHA256: e.ArgvSHA256, Repo: e.Repo, Global: e.Global, Prefix: e.Prefix, ReRunnable: e.ReRunnable, Network: e.Network, - Tree: e.Tree, Flaky: e.Flaky, Stub: e.Stub, + Tree: e.Tree, Flaky: e.Flaky, Stub: e.Stub, StubReason: e.StubReason, Timeout: e.Timeout, AddedAtMS: e.AddedAtMS, }) } @@ -424,16 +435,24 @@ func trustFlagSummary(e trustEntryView) string { {e.ReRunnable, " re-runnable"}, {e.Tree, " tree"}, {e.Flaky, " flaky"}, - // Spelled out rather than abbreviated to "stub". This line is what an - // operator auditing the store reads, and the fact worth reading is not - // that a word was set but that everything this entry ever passes is - // hollow (DKT-265). - {e.Stub, " stub(no-real-check)"}, } { if f.on { out += f.name } } + // Spelled out rather than abbreviated to "stub". This line is what an + // operator auditing the store reads, and the fact worth reading is not + // that a word was set but that everything this entry ever passes is + // hollow (DKT-265) — and, when one was recorded, WHY and what tracks + // fixing it (DKT-607). The reason renders through the escaping renderer + // like the argv: it is free text an operator may have pasted. + if e.Stub { + if e.StubReason != "" { + out += " stub(no-real-check: " + exec.Render(e.StubReason) + ")" + } else { + out += " stub(no-real-check)" + } + } // The declared hosts, named rather than flagged: "network" alone would say // an entry reaches SOMETHING, and which hosts is the whole content of the // declaration an operator is auditing. @@ -443,6 +462,14 @@ func trustFlagSummary(e trustEntryView) string { return out } +// getwd is os.Getwd, swappable ONLY by tests. The failure it lets a test +// inject is real — a working directory deleted or made unsearchable +// mid-session — but not portably reproducible: on darwin, getcwd answers from +// the kernel's name cache even after the directory is unlinked, so a test that +// deleted its own cwd would pass on Linux and assert nothing here. The refusal +// this guards (DKT-595) is worth the seam. +var getwd = os.Getwd + // errTrustUnrecorded marks the failure to write §3.6's event, so the taxonomy // can tell it from the store failures around it: nothing is wrong with the // operator's request, and telling them it was invalid would send them to fix an @@ -463,6 +490,13 @@ var errTrustUnrecorded = errors.New("the trust change was not recorded") // failed to land, which is loud rather than silent and errs toward over-reporting // authority rather than under-reporting it. // +// MANDATORY INCLUDES ATTRIBUTED (DKT-595). An event that lands without its +// actor and cwd is the 2026-08-19 shape — a ledger that shows privilege +// widening and cannot say by whom — so "recorded" means recorded WITH both +// fields: a working directory that cannot be read aborts the change through +// this same hook, and RecordTrustEvent itself refuses an empty actor or cwd +// from any writer. +// // OUTSIDE A REPO THE VERB STILL WORKS. The store is user-level (§3.5): requiring // a database to manage it would mean someone who installed docket could not // approve a command until they created a tracker. What §3.6 buys there is @@ -510,15 +544,26 @@ func (r *trustEventRecorder) record(entry trust.Entry) error { // HERE, at the call site, rather than inside the recorder: the engine has // no business shelling out to `git config`, and a helper that resolved its // own actor would be a helper that could attribute an event to whoever - // happened to be running the process that called it. + // happened to be running the process that called it. The actor resolver + // never yields the empty string — git identity, then the OS username, then + // the literal "unknown", which is its honest report of an anonymous + // environment and counts as an answer. // - // A cwd that cannot be read is not a failure to record the grant. The grant - // is the thing that matters; the empty string says the disambiguator was - // unavailable, which is a smaller and more honest loss than refusing to - // write the event at all. - cwd, err := os.Getwd() + // A cwd that cannot be read REFUSES THE WHOLE CHANGE (DKT-595). This + // reverses a decision this comment used to state — that the empty string + // was "a smaller and more honest loss than refusing to write the event at + // all" — and the 2026-08-19 batch is why it was wrong: 43 unattributed + // events, twelve of them widening network egress, and the ledger could not + // say who or from where. The disambiguator is not a nice-to-have riding on + // the grant; it is the half of the record the audit question is made of, + // and it can never be backfilled. The grant, by contrast, can be retried + // from a readable directory — so between an unrecordable attribution and a + // retryable refusal, the refusal is the smaller loss. Inside a repo the + // record is already MANDATORY, and this failure aborts through the same + // hook, with nothing written to either file. + cwd, err := getwd() if err != nil { - cwd = "" + return fmt.Errorf("%w: the working directory could not be resolved (%w); a trust change must record where it was made from, so run it again from a readable directory", errTrustUnrecorded, err) } if err := engine.RecordTrustEvent(conn, r.kind, engine.TrustGrant{ Name: entry.Name, @@ -532,6 +577,7 @@ func (r *trustEventRecorder) record(entry trust.Entry) error { Network: entry.Network, Timeout: entry.Timeout, Stub: entry.Stub, + StubReason: entry.StubReason, Actor: config.DefaultAuthor(), Cwd: cwd, }, time.Now().UnixMilli()); err != nil { diff --git a/internal/cli/trust_event_test.go b/internal/cli/trust_event_test.go index 804a262a..0fa6840a 100644 --- a/internal/cli/trust_event_test.go +++ b/internal/cli/trust_event_test.go @@ -173,6 +173,11 @@ func TestTrustAddEventCarriesEveryFlag(t *testing.T) { // reader unable to tell "not a stub" from "this docket does not record // stubs". "stub": false, + // EMPTY, and asserted rather than skipped, same reasoning: an add that + // recorded no stub rationale (DKT-607) must show an empty reason, not a + // missing key a reader cannot tell from "this docket does not record + // reasons". + "stub_reason": "", } want["kind"] = engine.EventTrustAdded @@ -588,6 +593,99 @@ func TestTrustAddStubEventSaysTheAssuranceIsHollow(t *testing.T) { } } +// TestTrustAddRefusesWhenTheCwdCannotBeResolved is DKT-595's CLI half: inside a +// repo, a grant whose working directory cannot be read is REFUSED, with nothing +// written to either file. +// +// This reverses the recorder's original call — degrade cwd to "" and write the +// event anyway — because the 2026-08-19 batch showed what the degraded shape is +// worth: privilege widened and a ledger that could not say by whom or from +// where. The refusal rides the same OnChange-hook abort as a failed event +// write, so the store mutation does not land either. +// +// The failure arrives through the `getwd` seam rather than by deleting the +// process's cwd, because the real thing is not portably reproducible: on +// darwin, getcwd answers from the kernel's name cache even after the directory +// is unlinked, so the deleted-cwd version of this test asserts nothing there. +func TestTrustAddRefusesWhenTheCwdCannotBeResolved(t *testing.T) { + _, cfg := trustRepo(t) + stubGetwdFailure(t) + + runErr := runTrustVerb(t, cfg, newTrustAddCmd(), "checks", "--", "make", "test") + if runErr == nil { + t.Fatal("the add succeeded with an unresolvable cwd; an unattributable grant must be refused") + } + // The failure reads as the unrecorded-change one (CmdError does not + // unwrap, so the taxonomy code and the message are the assertable surface, + // the same surface a caller sees). + if !strings.Contains(runErr.Error(), "not recorded") { + t.Errorf("the refusal does not say the change went unrecorded: %v", runErr) + } + if code := errorCodeOf(t, runErr); code != output.ErrGeneral { + t.Errorf("the refusal is %s, want GENERAL_ERROR — the argv was fine, and a VALIDATION code would send the operator to fix it: %v", code, runErr) + } + + // NOTHING LANDED: not the entry, not the event. The refusal runs inside + // trust.Add's hook, ahead of the store write, exactly like a failed record. + if entries := trustEntries(t); len(entries) != 0 { + t.Errorf("the grant landed anyway: %+v", entries) + } + if events := trustEvents(t, cfg); len(events) != 0 { + t.Errorf("a refused grant left %d event(s): %+v", len(events), events) + } +} + +// TestTrustRmRefusesWhenTheCwdCannotBeResolved: the same refusal on the other +// verb. A revocation is an attributable act for the same reason a grant is, and +// a refusal that guarded only the add would leave half the trail degradable. +func TestTrustRmRefusesWhenTheCwdCannotBeResolved(t *testing.T) { + _, cfg := trustRepo(t) + + err := runTrustVerb(t, cfg, newTrustAddCmd(), "checks", "--", "make", "test") + testsupport.Must(t, err, "trust add: %v", err) + + stubGetwdFailure(t) + runErr := runTrustVerb(t, cfg, newTrustRmCmd(), "checks") + if runErr == nil { + t.Fatal("the removal succeeded with an unresolvable cwd; an unattributable revocation must be refused") + } + if entries := trustEntries(t); len(entries) != 1 { + t.Errorf("the refused removal changed the store: %+v", entries) + } + if events := trustEvents(t, cfg); len(events) != 1 { + t.Errorf("got %d events, want only the original add — a refused removal records nothing", len(events)) + } +} + +// TestTrustAddOutsideARepoWorksWithAnUnreadableCwd pins the refusal's SCOPE: +// outside a repo there is no event log, the recorder warns-and-continues +// (§3.5), and an unreadable cwd changes none of that — the recorder never +// reaches cwd resolution on that path. DKT-595 hardens the trail where one +// exists; it does not make managing the user-level store require one. +func TestTrustAddOutsideARepoWorksWithAnUnreadableCwd(t *testing.T) { + _, cfg := trustNoRepo(t) + stubGetwdFailure(t) + + runErr := runTrustVerb(t, cfg, newTrustAddCmd(), "checks", "--", "make", "test") + testsupport.Must(t, runErr, "trust add outside a repo must still work with an unreadable cwd: %v", runErr) + if entries := trustEntries(t); len(entries) != 1 { + t.Fatalf("the entry was not written; got %+v", entries) + } +} + +// stubGetwdFailure makes the recorder's cwd resolution fail for one test, the +// way a deleted or unsearchable working directory would, restoring os.Getwd +// afterwards. See the `getwd` seam's comment for why the real failure cannot be +// staged portably. +func stubGetwdFailure(t *testing.T) { + t.Helper() + saved := getwd + getwd = func() (string, error) { + return "", errors.New("getwd: no such file or directory") + } + t.Cleanup(func() { getwd = saved }) +} + // TestReAddFlippingStubConflicts: `stub` joins the conflict set. // // A silent flip in either direction is the failure. Turning it OFF converts @@ -614,3 +712,78 @@ func TestReAddFlippingStubConflicts(t *testing.T) { t.Errorf("the conflict does not name `stub`: %v", err) } } + +// TestStubReasonTravelsWithTheGrant is DKT-607: a stub's WHY — and the issue +// tracking its removal — is recorded once at add time and then reaches every +// surface the stub itself reaches: the store, the grant event, and (proven in +// the engine package) the activation gate preflight. Before this, the decision +// that a gate stays a stub lived only in tribunal transcripts. +func TestStubReasonTravelsWithTheGrant(t *testing.T) { + _, cfg := trustRepo(t) + + reason := "no scanner selected yet; removal tracked by DKT-607" + err := runTrustVerb(t, cfg, newTrustAddCmd(), + "secret-scan", "--stub", "--stub-reason", reason, "--", "/usr/bin/true") + testsupport.Must(t, err, "trust add --stub --stub-reason: %v", err) + + events := trustEvents(t, cfg) + if len(events) != 1 { + t.Fatalf("got %d trust events, want the one add", len(events)) + } + if events[0]["stub_reason"] != reason { + t.Errorf("the event's stub_reason = %#v, want %q — the trail must carry "+ + "the documented decision, not only the hollowness", events[0]["stub_reason"], reason) + } + + entries := trustEntries(t) + if len(entries) != 1 || entries[0].StubReason != reason { + t.Errorf("the stored entry does not carry the reason; got %+v", entries) + } +} + +// TestStubReasonRequiresStub: a reason describes a stub, so supplying one on a +// real check is a contradiction and is refused as VALIDATION — one of the two +// flags is a mistake, and guessing which would either hide a hollow check or +// hollow a real one. +func TestStubReasonRequiresStub(t *testing.T) { + _, cfg := trustRepo(t) + + err := runTrustVerb(t, cfg, newTrustAddCmd(), + "secret-scan", "--stub-reason", "tracked by DKT-607", "--", "/usr/bin/true") + if err == nil { + t.Fatal("--stub-reason without --stub succeeded; a reason on a non-stub entry is a contradiction") + } + if code := errorCodeOf(t, err); code != output.ErrValidation { + t.Errorf("the refusal is %s, want VALIDATION: %v", code, err) + } + if entries := trustEntries(t); len(entries) != 0 { + t.Errorf("the refused add landed anyway: %+v", entries) + } + if events := trustEvents(t, cfg); len(events) != 0 { + t.Errorf("a refused add left %d event(s)", len(events)) + } +} + +// TestReAddChangingStubReasonConflicts: the reason joins the conflict set. It +// IS the documented decision, and a re-add that silently rewrote or erased it +// would swap one decision for another with the store showing only that +// something of that name was re-approved. +func TestReAddChangingStubReasonConflicts(t *testing.T) { + _, cfg := trustRepo(t) + + err := runTrustVerb(t, cfg, newTrustAddCmd(), + "scan", "--stub", "--stub-reason", "tracked by DKT-607", "--", "make", "test") + testsupport.Must(t, err, "trust add: %v", err) + + err = runTrustVerb(t, cfg, newTrustAddCmd(), + "scan", "--stub", "--", "make", "test") + if err == nil { + t.Fatal("re-adding without the reason succeeded; erasing the documented decision must conflict") + } + if code := errorCodeOf(t, err); code != output.ErrConflict { + t.Errorf("the refusal is %v, want %v", code, output.ErrConflict) + } + if !strings.Contains(err.Error(), "stub_reason") { + t.Errorf("the conflict does not name `stub_reason`: %v", err) + } +} diff --git a/internal/cli/workflow_deprecate.go b/internal/cli/workflow_deprecate.go index febbadb9..85c3859a 100644 --- a/internal/cli/workflow_deprecate.go +++ b/internal/cli/workflow_deprecate.go @@ -1,6 +1,7 @@ package cli import ( + "database/sql" "errors" "fmt" @@ -42,6 +43,15 @@ nothing else can stop it from binding. Use --restore to put a retired version back into binding. +A registry is PER PROJECT, and so is retirement. By default this retires the +version in the project the working directory resolves to; --project retires it +in one other project, and --all-projects retires it in every project in the +store. Both report each project's own outcome: a project where the version was +never registered is reported as not-registered and a project where it was +already retired as already-deprecated, while every project that did retire it +still did. This is the answer to "this orphaned name has to go everywhere", +which previously meant one invocation per checkout. + THE OLDER ESCAPE HATCH, for context. Before this verb, the only way to stop a name from binding was to re-register that SAME name at a HIGHER version whose [match] admits nothing — for example labels_all and unless_labels naming the @@ -76,6 +86,16 @@ func runWorkflowDeprecate(cmd *cobra.Command, args []string, w *output.Writer) e } restore, _ := cmd.Flags().GetBool("restore") + + targets, fannedOut, err := resolveRegistryTargets(cmd, conn) + if err != nil { + return err + } + if fannedOut { + return deprecateWorkflowAcrossProjects( + cmd, w, conn, args[0], name, version, restore, targets) + } + if restore { wf, err := db.RestoreWorkflow(conn, getProjectID(cmd), name, version) if err != nil { @@ -100,8 +120,78 @@ func runWorkflowDeprecate(cmd *cobra.Command, args []string, w *output.Writer) e return nil } +// describeDeprecateFailure gives a fanned-out refusal the SAME sentence its +// single-project counterpart writes, so the per-project detail line reads as +// what the operator would have seen standing in that checkout. The sentinels +// stay wrapped, since the classifier and the report both match on them. +func describeDeprecateFailure(err error, ref string) error { + if errors.Is(err, db.ErrWorkflowAlreadyDeprecated) { + return fmt.Errorf("%s is already deprecated: %w", ref, err) + } + return describeMissingWorkflow(err, ref) +} + +// deprecateWorkflowAcrossProjects is the --project / --all-projects path for +// both directions of the verb. +// +// RESTORE IS IDEMPOTENT AND DEPRECATE IS NOT, and the report keeps that +// asymmetry visible rather than smoothing it: db.RestoreWorkflow returns the +// row unchanged when it was never retired (reported as already-binding, a +// success), while db.DeprecateWorkflow refuses a second retirement (reported as +// already-deprecated, a failure carrying CONFLICT). Both readings are the +// single-project behavior, per project, which is the criterion — a sweep must +// not turn one project's refusal into another's silence. +func deprecateWorkflowAcrossProjects( + cmd *cobra.Command, w *output.Writer, conn *sql.DB, + ref, name string, version int, restore bool, targets []*model.Project, +) error { + operation := "workflow deprecate" + if restore { + operation = "workflow deprecate --restore" + } + report := ®istryFanoutReport{ + Operation: operation, Subject: ref, Scope: fanoutScope(cmd), + } + + for _, target := range targets { + var ( + wf *model.Workflow + err error + outcome string + ) + if restore { + // The pre-state decides which success this is; RestoreWorkflow + // itself returns the same row either way. + before, lookupErr := db.GetWorkflow(conn, target.ID, name, version) + if lookupErr != nil { + report.Results = append(report.Results, registryFailureResult( + target, describeDeprecateFailure(lookupErr, ref), workflowErr)) + continue + } + outcome = outcomeAlreadyBinding + if before.Deprecated() { + outcome = outcomeRestored + } + wf, err = db.RestoreWorkflow(conn, target.ID, name, version) + } else { + outcome = outcomeDeprecated + wf, err = db.DeprecateWorkflow(conn, target.ID, name, version, model.NowMS()) + } + if err != nil { + report.Results = append(report.Results, registryFailureResult( + target, describeDeprecateFailure(err, ref), workflowErr)) + continue + } + report.Results = append(report.Results, + registrySuccessResult(target, outcome, wf.Ref())) + } + + return finishRegistryFanout(w, report) +} + func init() { workflowDeprecateCmd.Flags().Bool( "restore", false, "Return a retired version to binding") + addRegistryTargetFlags(workflowDeprecateCmd, "Retire the version") workflowCmd.AddCommand(workflowDeprecateCmd) } diff --git a/internal/cli/workflow_deprecated_visibility_test.go b/internal/cli/workflow_deprecated_visibility_test.go index 16330d00..61c0c9f5 100644 --- a/internal/cli/workflow_deprecated_visibility_test.go +++ b/internal/cli/workflow_deprecated_visibility_test.go @@ -33,7 +33,7 @@ func TestWorkflowListExposesDeprecation(t *testing.T) { } // Human render: the retired version is marked. - if got := renderWorkflowList(workflows); !strings.Contains(got, "[deprecated]") { + if got := renderWorkflowList(workflows, false); !strings.Contains(got, "[deprecated]") { t.Errorf("human render %q does not mark the retired version", got) } @@ -63,3 +63,45 @@ func TestWorkflowListExposesDeprecation(t *testing.T) { } } } + +// TestWorkflowListHidesDeprecatedByDefault: a retired version costs a reader +// nothing to filter mentally when the verb has already done it — the default +// listing shows only what can still bind, and --deprecated is what opts a +// cleanup pass back into seeing the retired lineage too. +func TestWorkflowListHidesDeprecatedByDefault(t *testing.T) { + conn := newTestDB(t) + testsupport.Must(t, registerSource(t, conn, minimalWorkflow), "register") + _, err := db.DeprecateWorkflow(conn, 1, "unit", 1, model.NowMS()) + testsupport.Must(t, err, "deprecate: %v", err) + + cmd := workflowListCmdWithDB(conn, 50) + w, buf := bufWriter(true) + testsupport.Must(t, runWorkflowList(cmd, nil, w), "list") + + var envelope struct { + Data struct { + Workflows []json.RawMessage `json:"workflows"` + Total int `json:"total"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if len(envelope.Data.Workflows) != 0 { + t.Errorf("default list carries %d rows, want the sole deprecated "+ + "version hidden: %s", len(envelope.Data.Workflows), buf.String()) + } + if envelope.Data.Total != 0 { + t.Errorf("total = %d, want the filtered population's count (0)", + envelope.Data.Total) + } + + // --deprecated opts the retired version back in, marked. + cmd = workflowListCmdWithDB(conn, 50) + testsupport.Must(t, cmd.Flags().Set("deprecated", "true"), "setting --deprecated") + w, buf = bufWriter(false) + testsupport.Must(t, runWorkflowList(cmd, nil, w), "list --deprecated") + if !strings.Contains(buf.String(), "[deprecated]") { + t.Errorf("--deprecated does not surface the retired version: %q", buf.String()) + } +} diff --git a/internal/cli/workflow_lint.go b/internal/cli/workflow_lint.go index 18992f8b..cfe9c82b 100644 --- a/internal/cli/workflow_lint.go +++ b/internal/cli/workflow_lint.go @@ -74,6 +74,21 @@ func runWorkflowLint(cmd *cobra.Command, args []string, w *output.Writer) error // edited file at a frozen name@version is the retro-loop's observed trap — // the definition validates, then the next activation refuses the whole run // — so lint surfaces it while the author is still looking at the file. + // + // THIS IS ALREADY DKT-590's CHECK, FROM THE FILE'S SIDE, and it is why lint + // gains nothing from the source-drift verdict `workflow show` and + // `run activate` now report. Those two start from a REGISTERED ROW and read + // the path the row recorded; lint starts from bytes the caller handed it + // and looks up the row those bytes declare. When the file passed here IS + // the file a row recorded, the comparison below is the same comparison on + // the same two hashes, refusing where the others refuse. When it is not — + // the DKT-590 population, where the file has moved on to version 8 while + // version 4 stays registered — lint resolves @8, finds a free slot, and + // correctly reports `new`: nothing about @8 is wrong, and @4's stale + // provenance is a fact about a row this invocation was never asked about. + // Reporting it here would fire on every ordinary version bump, since a + // bumped file always leaves the superseded version's recorded path holding + // different bytes. sum := workflow.SHA256(src) registration := "new" existing, err := db.GetWorkflow(conn, getProjectID(cmd), def.Pipeline.Name, def.Pipeline.Version) diff --git a/internal/cli/workflow_list.go b/internal/cli/workflow_list.go index 0a155956..56e53634 100644 --- a/internal/cli/workflow_list.go +++ b/internal/cli/workflow_list.go @@ -6,6 +6,7 @@ import ( "strings" "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/output" "github.com/spf13/cobra" @@ -54,6 +55,25 @@ var workflowListCmd = &cobra.Command{ Use: "list", Short: "List registered workflows", Aliases: []string{"ls"}, + Long: `List registered workflows. + +--orphans narrows the list to ORPHANED REGISTRATIONS: registered names that no +file in any instance-config root declares any more. A registration is a row and +not a file, so renaming a workflow leaves every version of the old name live and +still binding — which is how one issue's label came to match two workflows, one +of which had not existed on disk for weeks (DKT-609). + +The verdict is per NAME, not per version: a superseded version whose name is +still declared somewhere is ordinary lineage, not an orphan. Deprecated rows are +listed like any other, as they are without the flag — they are precisely what a +cleanup pass wants to see, and hiding an already-retired orphan would make the +verb unable to show its own work. --deprecated has no effect under --orphans +for the same reason: orphan status is a filesystem verdict on the NAME, not a +binding verdict on the version, so narrowing by one must not silently narrow +by the other too. + +--deprecated includes retired versions (deprecated_at_ms set) in the plain +listing; without it, only versions still eligible to bind are shown.`, RunE: func(cmd *cobra.Command, args []string) error { return runWorkflowList(cmd, args, getWriter(cmd)) }, @@ -64,31 +84,110 @@ func runWorkflowList(cmd *cobra.Command, args []string, w *output.Writer) error name, _ := cmd.Flags().GetString("name") limit, _ := cmd.Flags().GetInt("limit") + orphans, _ := cmd.Flags().GetBool("orphans") + deprecated, _ := cmd.Flags().GetBool("deprecated") if err := validateLimit(cmd, limit); err != nil { return err } + // The orphan verdict needs the FILESYSTEM, which internal/db does not + // touch anywhere else, so it is applied here over the query's rows — the + // same shape `workflow show` uses for the source check. That is also why + // the query runs UNLIMITED under --orphans: a SQL LIMIT would cut the rows + // before the filter saw them, so `total` would count workflows rather than + // orphans and truncation would be computed against the wrong population. + queryLimit := limit + if orphans { + queryLimit = 0 + } + + // --orphans keeps its own rule (deprecated rows are listed like any + // other) regardless of --deprecated: orphan status is a filesystem + // verdict on the NAME, binding eligibility a verdict on the VERSION, and + // narrowing by one must not silently narrow by the other. + excludeDeprecated := !deprecated && !orphans + workflows, total, err := db.ListWorkflows(conn, db.WorkflowListOptions{ - ProjectID: getProjectID(cmd), - Name: name, - Limit: limit, + ProjectID: getProjectID(cmd), + Name: name, + Limit: queryLimit, + ExcludeDeprecated: excludeDeprecated, }) if err != nil { return cmdErr(fmt.Errorf("listing workflows: %w", err), output.ErrGeneral) } + if orphans { + if workflows, total, err = keepOrphans(workflows, limit); err != nil { + return err + } + } + result := workflowListResult{Workflows: workflows, Total: total, limit: limit} var message string if !w.JSONMode { - message = renderWorkflowList(workflows) + message = renderWorkflowList(workflows, orphans) } w.Success(result, message) return nil } -func renderWorkflowList(workflows []*model.Workflow) string { +// keepOrphans reduces the listed rows to the orphaned ones, stamping each with +// the verdict that put it there, and returns the true pre-limit total of THAT +// population. +// +// IT REFUSES RATHER THAN REPORTING AN UNCHECKED ANSWER. With no instance-config +// root to scan, every registration in the store trivially has "no file in any +// root", and rendering that as a list of orphans would be the verb inventing +// its entire result out of having looked nowhere. The two ways that happens — +// no root exists, and a root that would not scan — are separate messages +// because they send an operator to different places. +func keepOrphans(workflows []*model.Workflow, limit int) ([]*model.Workflow, int, error) { + index, err := engine.ScanWorkflowOrigins() + if err != nil { + return nil, 0, cmdErr(err, output.ErrValidation) + } + if !index.Scanned() { + return nil, 0, cmdErr(fmt.Errorf( + "no instance-config root exists on this machine, so no registration "+ + "can be classified: with nothing to scan, every registered name "+ + "would look orphaned"), output.ErrValidation) + } + if err := index.Err(); err != nil { + return nil, 0, cmdErr(fmt.Errorf( + "scanning the instance config for workflow definitions: %w", err), + output.ErrGeneral) + } + + out := make([]*model.Workflow, 0, len(workflows)) + for _, wf := range workflows { + status := index.Status(wf.Name) + if !status.Orphaned() { + continue + } + // The verdict rides on the row, so the JSON payload carries the FACT + // and not merely the fact that a flag was passed. + wf.Origin = status + out = append(out, wf) + } + + total := len(out) + if limit > 0 && len(out) > limit { + out = out[:limit] + } + return out, total, nil +} + +func renderWorkflowList(workflows []*model.Workflow, orphans bool) string { if len(workflows) == 0 { + if orphans { + // NOT "No workflows registered." — under --orphans an empty result + // is a clean bill of health for a store that may hold dozens of + // workflows, and the two readings are opposite. + return "No orphaned registrations: every registered workflow name " + + "is still declared by a file in the instance config." + } return "No workflows registered." } @@ -99,6 +198,14 @@ func renderWorkflowList(workflows []*model.Workflow) string { var b strings.Builder for _, wf := range workflows { fmt.Fprintf(&b, "%-28s %s", wf.Ref(), shortSHA(wf.SourceSHA256)) + // The orphan marker prints only where a reader WENT AND LOOKED, which + // today is --orphans alone. An unmarked row in the default listing + // therefore means "not asked", never "checked and fine" — the same + // rule `source_status` follows, and the reason the default output is + // byte-identical to what it was. + if wf.Origin.Orphaned() { + b.WriteString(" [orphaned]") + } if wf.Deprecated() { b.WriteString(" [deprecated]") } @@ -122,5 +229,9 @@ func shortSHA(sha string) string { func init() { workflowListCmd.Flags().String("name", "", "Filter by workflow name") workflowListCmd.Flags().Int("limit", 50, "Maximum number of results") + workflowListCmd.Flags().Bool("orphans", false, + "List only registrations whose name no file in any instance-config root declares") + workflowListCmd.Flags().Bool("deprecated", false, + "Include deprecated (retired) workflow versions in the listing") workflowCmd.AddCommand(workflowListCmd) } diff --git a/internal/cli/workflow_orphans_test.go b/internal/cli/workflow_orphans_test.go new file mode 100644 index 00000000..e25b00a2 --- /dev/null +++ b/internal/cli/workflow_orphans_test.go @@ -0,0 +1,218 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// DKT-609: `workflow list --orphans` — registered names that no file in any +// instance-config root declares any more. +// +// The state it exists for: a rename left four versions of the old name live in +// the store, and NOTHING IN ANY READ VERB SAID SO. `workflow list` showed them +// exactly as it showed the workflows that still had definitions, and the only +// way to tell the difference was to go and read the corpus's git history. + +// renamedWorkflow is the definition that stays on disk. Its name differs from +// minimalWorkflow's ("unit"), which is what makes "unit" the stranded +// registration in these tests. +const renamedWorkflow = ` +[pipeline] +name = "renamed" +version = 1 +[[step]] +name = "first" +after = [] +executor = "someone" +emits = "result" +` + +// configRootWith points DOCKET_PATH at a fresh store whose instance config +// holds exactly the workflow files named, and returns the store directory. +// +// THE TEMP DIR COMES FIRST, BEFORE t.Setenv — t.TempDir() reads TMPDIR, and a +// test that rewrote the environment before taking its directory would get a +// path that does not exist. The engine's own configRepo helper carries the +// same warning for the same reason. +func configRootWith(t *testing.T, files map[string]string) string { + t.Helper() + docketDir := filepath.Join(t.TempDir(), ".docket") + workflows := filepath.Join(docketDir, "config", "workflows") + testsupport.Must(t, os.MkdirAll(workflows, 0o755), + "creating the instance-config workflows directory") + for name, body := range files { + testsupport.Must(t, os.WriteFile(filepath.Join(workflows, name), []byte(body), 0o644), + "writing %s", name) + } + t.Setenv("DOCKET_PATH", docketDir) + return docketDir +} + +// listOrphans runs `workflow list --orphans` and returns the human render and +// the parsed JSON rows. +func listOrphans(t *testing.T, conn *sql.DB) (string, []orphanRow) { + t.Helper() + human := runOrphanList(t, conn, false) + raw := runOrphanList(t, conn, true) + + var envelope struct { + Data struct { + Workflows []orphanRow `json:"workflows"` + Total int `json:"total"` + } `json:"data"` + } + if err := json.Unmarshal([]byte(raw), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, raw) + } + if envelope.Data.Total != len(envelope.Data.Workflows) { + // The Collection contract's `total` is the TRUE pre-limit count, and + // under --orphans the population it counts is the orphans, not the + // registry: a total that still counted workflows would report + // truncation that never happened. + t.Errorf("total = %d over %d listed orphans", envelope.Data.Total, + len(envelope.Data.Workflows)) + } + return human, envelope.Data.Workflows +} + +type orphanRow struct { + Name string `json:"name"` + Version int `json:"version"` + Origin *model.WorkflowOriginStatus `json:"origin"` +} + +func runOrphanList(t *testing.T, conn *sql.DB, jsonMode bool) string { + t.Helper() + cmd := orphanListCmd(t, conn) + w, buf := bufWriter(jsonMode) + testsupport.Must(t, runWorkflowList(cmd, nil, w), "workflow list --orphans") + return buf.String() +} + +func orphanListCmd(t *testing.T, conn *sql.DB) *cobra.Command { + t.Helper() + cmd := workflowListCmdWithDB(conn, 50) + testsupport.Must(t, cmd.Flags().Set("orphans", "true"), "setting --orphans") + return cmd +} + +// TestWorkflowListOrphansNamesTheStrandedRegistration is DKT-609's first +// acceptance criterion: the orphan is identifiable through a READ VERB, with no +// git archaeology, and the workflow that still has a file is not listed +// alongside it. +func TestWorkflowListOrphansNamesTheStrandedRegistration(t *testing.T) { + conn := newTestDB(t) + configRootWith(t, map[string]string{"renamed.toml": renamedWorkflow}) + testsupport.Must(t, registerSource(t, conn, minimalWorkflow), "registering unit@1") + testsupport.Must(t, registerSource(t, conn, renamedWorkflow), "registering renamed@1") + + human, rows := listOrphans(t, conn) + + if len(rows) != 1 { + t.Fatalf("listed %d orphans, want exactly 1 (unit@1): %+v", len(rows), rows) + } + if rows[0].Name != "unit" || rows[0].Version != 1 { + t.Errorf("listed %s@%d, want unit@1 — the name whose file is gone", + rows[0].Name, rows[0].Version) + } + // The row carries the VERDICT, not merely the fact that a flag was passed: + // a JSON consumer can tell what was checked and where. + if !rows[0].Origin.Orphaned() { + t.Errorf("the listed row carries no orphan verdict: %+v", rows[0].Origin) + } + if len(rows[0].Origin.Roots) == 0 { + t.Error("the verdict names no roots, so a reader cannot tell where the " + + "name was looked for") + } + if !strings.Contains(human, "unit@1") || !strings.Contains(human, "[orphaned]") { + t.Errorf("the human render does not mark the orphan: %q", human) + } + if strings.Contains(human, "renamed@1") { + t.Errorf("the human render lists a workflow that still has a file: %q", human) + } +} + +// TestWorkflowListLeavesTheDefaultListingAlone: the orphan verdict costs a +// filesystem scan, so it is computed ONLY where it was asked for. An unmarked +// row in the default listing therefore means "not asked", never "checked and +// fine" — the rule `source_status` already follows, and the reason the default +// output is byte-identical to what it was. +func TestWorkflowListLeavesTheDefaultListingAlone(t *testing.T) { + conn := newTestDB(t) + configRootWith(t, map[string]string{"renamed.toml": renamedWorkflow}) + testsupport.Must(t, registerSource(t, conn, minimalWorkflow), "registering unit@1") + + cmd := workflowListCmdWithDB(conn, 50) + w, buf := bufWriter(true) + testsupport.Must(t, runWorkflowList(cmd, nil, w), "workflow list") + if strings.Contains(buf.String(), "origin") { + t.Errorf("the default listing carries an origin verdict nobody asked "+ + "for: %s", buf.String()) + } + + w, buf = bufWriter(false) + testsupport.Must(t, runWorkflowList(cmd, nil, w), "workflow list") + if strings.Contains(buf.String(), "[orphaned]") { + t.Errorf("the default listing marks an orphan it never checked: %q", buf.String()) + } +} + +// TestWorkflowListOrphansIncludesDeprecatedRows: a retired orphan is still +// listed, marked as retired. +// +// `workflow list` shows every registered version and marks the deprecated ones +// rather than hiding them, and --orphans follows that convention rather than +// diverging: the rows an operator is mid-way through cleaning up are exactly +// the ones a cleanup verb must keep showing, and hiding them would make the +// verb unable to show its own work. +func TestWorkflowListOrphansIncludesDeprecatedRows(t *testing.T) { + conn := newTestDB(t) + configRootWith(t, map[string]string{"renamed.toml": renamedWorkflow}) + testsupport.Must(t, registerSource(t, conn, minimalWorkflow), "registering unit@1") + _, err := db.DeprecateWorkflow(conn, 1, "unit", 1, model.NowMS()) + testsupport.Must(t, err, "deprecating unit@1: %v", err) + + human, rows := listOrphans(t, conn) + if len(rows) != 1 { + t.Fatalf("listed %d orphans, want 1: a retired orphan is still an "+ + "orphan, and is what a cleanup pass wants to see", len(rows)) + } + if !strings.Contains(human, "[orphaned]") || !strings.Contains(human, "[deprecated]") { + t.Errorf("the render does not carry both facts: %q", human) + } +} + +// TestWorkflowListOrphansRefusesWithNothingToScan: with no instance-config +// root, EVERY registered name trivially has no file in any root. Rendering +// that as a list of orphans would be the verb inventing its whole result out +// of having looked nowhere, so it refuses instead — VALIDATION, the same code +// every other "this invocation cannot answer that" refusal carries. +func TestWorkflowListOrphansRefusesWithNothingToScan(t *testing.T) { + conn := newTestDB(t) + // A store directory with NO config/ subdirectory: the dormancy case. + docketDir := filepath.Join(t.TempDir(), ".docket") + testsupport.Must(t, os.MkdirAll(docketDir, 0o755), "creating the store directory") + t.Setenv("DOCKET_PATH", docketDir) + testsupport.Must(t, registerSource(t, conn, minimalWorkflow), "registering unit@1") + + cmd := orphanListCmd(t, conn) + w, _ := bufWriter(true) + err := runWorkflowList(cmd, nil, w) + if err == nil { + t.Fatal("--orphans reported orphans with no instance-config root to " + + "scan; every registration in the store would qualify") + } + if got := codeOf(t, err); got != output.ErrValidation { + t.Errorf("error code = %q, want %q", got, output.ErrValidation) + } +} diff --git a/internal/cli/workflow_register.go b/internal/cli/workflow_register.go index 6161ff58..b6348407 100644 --- a/internal/cli/workflow_register.go +++ b/internal/cli/workflow_register.go @@ -26,7 +26,22 @@ needs no temporary file. Registering the same bytes at an existing name@version again is a success and changes nothing. Registering DIFFERENT bytes at an existing name@version is a CONFLICT: a registered name@version is frozen, so that a run which pinned it -cannot have the definition swapped underneath it.`, +cannot have the definition swapped underneath it. + +A registry is PER PROJECT. By default the definition lands in the project the +working directory resolves to; --project registers it into one other project, +and --all-projects registers it into every project in the store. Both report +each project's own outcome, and each project's idempotency and conflict rules +are decided there — a conflict in one project neither cancels nor hides another +project's registration. + +VALIDATION IS PER TARGET, not once for the invocation. A definition's +'vote_rule' and 'payload' references resolve against the registry of the project +being written to, so the SAME bytes can be valid in one project and reference a +schema that does not exist in the next. The project that cannot resolve them is +refused there and reported as invalid; the rest still register. Register the +schemas first — 'docket schema register --all-projects' — when +sweeping a corpus into a store that has not seen it.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runWorkflowRegister(cmd, args, getWriter(cmd)) @@ -47,22 +62,21 @@ func runWorkflowRegister(cmd *cobra.Command, args []string, w *output.Writer) er return workflowErr(err) } - // V26 (gates-trust §8.2): every `vote_rule` names a REGISTERED threshold - // configuration. It runs here rather than inside workflow.Validate because - // it is the one rule that asks a question about the environment rather than - // about the bytes, and Validate stays pure. - if err := workflow.ValidateVoteRules(def, voteRuleResolver{conn, getProjectID(cmd)}); err != nil { - return workflowErr(err) + // The GRAMMAR is decided before the targets are: a definition that does not + // parse is wrong everywhere, and telling an operator which of thirteen + // projects rejected a typo would be noise. Everything past this point is + // environment, and environment is per project. + targets, fannedOut, err := resolveRegistryTargets(cmd, conn) + if err != nil { + return err + } + if fannedOut { + return registerWorkflowAcrossProjects(cmd, w, conn, def, src, path, targets) } - // V21a-V21d and V25a (§4.9): threshold fields and literals cross-validated - // against the REGISTERED schema, per §11.2's "validated against the - // registered schema at `workflow register` time". - // - // The order — Validate, then vote rules, then schemas — is deliberate: an - // author sees GRAMMAR errors before ENVIRONMENT errors, so a definition with - // a typo in a step name is not first told that a schema is missing. - if err := workflow.ValidateSchemas(def, schemaResolver{conn, getProjectID(cmd)}); err != nil { + // V26 (gates-trust §8.2) and V21a-V21d/V25a (§4.9): vote rules and schemas, + // resolved against THIS project's registry. See validateWorkflowEnvironment. + if err := validateWorkflowEnvironment(conn, def, getProjectID(cmd)); err != nil { return workflowErr(err) } @@ -71,15 +85,33 @@ func runWorkflowRegister(cmd *cobra.Command, args []string, w *output.Writer) er return cmdErr(err, output.ErrGeneral) } + stored, created, err := db.InsertWorkflow( + conn, workflowRow(def, src, path, parsed, getProjectID(cmd)), model.NowMS()) + if err != nil { + return workflowErr(err) + } + + message := fmt.Sprintf("Registered %s", stored.Ref()) + if !created { + message = fmt.Sprintf("%s is already registered with these bytes", stored.Ref()) + } + w.Success(stored, message) + return nil +} + +// workflowRow builds the row to insert. Shared by the single-project path and +// the fan-out so the two can never store different columns for the same bytes. +func workflowRow( + def *workflow.Definition, src []byte, path string, parsed []byte, projectID int, +) *model.Workflow { // source_path is provenance only — it records where the bytes came from // and is never re-read. stdin has no path to record. sourcePath := path if path == "-" { sourcePath = "" } - - wf := &model.Workflow{ - ProjectID: getProjectID(cmd), + return &model.Workflow{ + ProjectID: projectID, Name: def.Pipeline.Name, Version: def.Pipeline.Version, Description: def.Pipeline.Description, @@ -88,18 +120,82 @@ func runWorkflowRegister(cmd *cobra.Command, args []string, w *output.Writer) er Body: string(src), Parsed: string(parsed), } +} - stored, created, err := db.InsertWorkflow(conn, wf, model.NowMS()) +// validateWorkflowEnvironment runs the two checks that ask a question about a +// PROJECT rather than about the bytes. +// +// V26 (gates-trust §8.2): every `vote_rule` names a REGISTERED threshold +// configuration. V21a-V21d and V25a (§4.9): threshold fields and literals +// cross-validated against the REGISTERED schema, per §11.2's "validated against +// the registered schema at `workflow register` time". Both run here rather than +// inside workflow.Validate, which stays pure and holds no database handle. +// +// The order — Validate, then vote rules, then schemas — is deliberate: an +// author sees GRAMMAR errors before ENVIRONMENT errors, so a definition with a +// typo in a step name is not first told that a schema is missing. +// +// Both resolve against a project's own registry, which is why the fan-out runs +// them once per target instead of once per invocation: a `payload` reference +// that resolves in the invoking project may name nothing in the next one, and +// registering there anyway would store a definition guaranteed to fail at +// activation — the exact failure this validation exists to move forward in time. +func validateWorkflowEnvironment( + conn *sql.DB, def *workflow.Definition, projectID int, +) error { + if err := workflow.ValidateVoteRules(def, voteRuleResolver{conn, projectID}); err != nil { + return err + } + return workflow.ValidateSchemas(def, schemaResolver{conn, projectID}) +} + +// registerWorkflowAcrossProjects is the --project / --all-projects path. +// +// The bytes are canonicalized ONCE — that is a pure function of the definition +// and cannot differ per project — and then each target gets its own validation +// and its own insert. A failure is recorded and the loop continues: the whole +// point of the flag is that one command covers a store, and stopping at the +// first conflict would leave the operator to work out by hand which projects +// the sweep had reached. +func registerWorkflowAcrossProjects( + cmd *cobra.Command, w *output.Writer, conn *sql.DB, + def *workflow.Definition, src []byte, path string, targets []*model.Project, +) error { + parsed, err := workflow.Canonical(def) if err != nil { - return workflowErr(err) + return cmdErr(err, output.ErrGeneral) } - message := fmt.Sprintf("Registered %s", stored.Ref()) - if !created { - message = fmt.Sprintf("%s is already registered with these bytes", stored.Ref()) + report := ®istryFanoutReport{ + Operation: "workflow register", + Subject: fmt.Sprintf("%s@%d", def.Pipeline.Name, def.Pipeline.Version), + Scope: fanoutScope(cmd), } - w.Success(stored, message) - return nil + + for _, target := range targets { + if err := validateWorkflowEnvironment(conn, def, target.ID); err != nil { + report.Results = append(report.Results, + registryFailureResult(target, err, workflowErr)) + continue + } + + stored, created, err := db.InsertWorkflow( + conn, workflowRow(def, src, path, parsed, target.ID), model.NowMS()) + if err != nil { + report.Results = append(report.Results, + registryFailureResult(target, err, workflowErr)) + continue + } + + outcome := outcomeUnchanged + if created { + outcome = outcomeRegistered + } + report.Results = append(report.Results, + registrySuccessResult(target, outcome, stored.Ref())) + } + + return finishRegistryFanout(w, report) } // readWorkflowSource reads the definition bytes from a file or, for "-", from @@ -129,6 +225,7 @@ func readWorkflowSource(cmd *cobra.Command, path string) ([]byte, error) { } func init() { + addRegistryTargetFlags(workflowRegisterCmd, "Register") workflowCmd.AddCommand(workflowRegisterCmd) } diff --git a/internal/cli/workflow_show.go b/internal/cli/workflow_show.go index a8fe95e7..a662634d 100644 --- a/internal/cli/workflow_show.go +++ b/internal/cli/workflow_show.go @@ -5,6 +5,7 @@ import ( "strings" "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/output" "github.com/ALT-F4-LLC/docket/internal/workflow" @@ -16,8 +17,18 @@ var workflowShowCmd = &cobra.Command{ Short: "Show a registered workflow definition", Long: `Show a registered workflow. -Omitting @version selects the highest registered version. --source emits the -stored TOML verbatim, which is the exact bytes that were registered and hashed.`, +Omitting @version selects the highest registered version that is NOT +deprecated — the same version a new run would bind. Deprecated versions are +skipped; name them explicitly (name@version) to show one. A name whose every +version is deprecated is reported as not found. --source emits the stored TOML +verbatim, which is the exact bytes that were registered and hashed. + +The summary reports a SOURCE STATUS: whether the file at the recorded +source_path still hashes to source_sha256 (matches), holds different bytes +(drifted), cannot be read at all (unreadable), or names nothing this invocation +can resolve — no path recorded, or a relative one (unchecked). Drift is reported, +never repaired: a registered name@version is frozen, and registering an edited +file is the install path's job.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runWorkflowShow(cmd, args, getWriter(cmd)) @@ -51,6 +62,14 @@ func runWorkflowShow(cmd *cobra.Command, args []string, w *output.Writer) error return nil } + // DKT-590: the two provenance columns, checked against each other instead + // of merely printed. `source_path` and `source_sha256` were reported side + // by side while the file at that path had been a different definition for + // four versions, and nothing in this output said so. The verdict rides on + // the row itself (populated nowhere else), so it reaches the JSON envelope + // and the human summary from one place. + wf.SourceStatus = engine.CheckWorkflowSource(wf.SourcePath, wf.SourceSHA256) + var message string if !w.JSONMode { message = renderWorkflowShow(wf) @@ -78,6 +97,13 @@ func renderWorkflowShow(wf *model.Workflow) string { if wf.SourcePath != "" { fmt.Fprintf(&b, "source: %s\n", wf.SourcePath) } + // The verdict prints on its own line, ALWAYS when it was computed — a + // clean match included. "Nothing was said" and "the bytes still match" are + // different facts, and the whole defect this closes is a reader who could + // not tell them apart (DKT-590). + if wf.SourceStatus != nil { + fmt.Fprintf(&b, "source status: %s\n", engine.DescribeWorkflowSource(wf.SourceStatus)) + } // A retired version renders as retired. Without this the summary of a // version that can no longer bind is indistinguishable from one that can. if wf.Deprecated() { diff --git a/internal/cli/workflow_source_test.go b/internal/cli/workflow_source_test.go new file mode 100644 index 00000000..684de75d --- /dev/null +++ b/internal/cli/workflow_source_test.go @@ -0,0 +1,171 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-590: `workflow show` printed `source_path` and `source_sha256` side by +// side and never asked whether they still agreed. On the machine that filed the +// issue they had not agreed for four versions — the row said investigation@4 at +// 4cb066e3, the file at that very path was version 8 at 6ed74d17 — and the +// output looked exactly as it does when they do agree. +// +// These tests fix the three verdicts apart, because they send an operator to +// three different places: nothing to do, re-register through the install path, +// restore a missing file. + +// registerSourceAt registers src from an absolute path the test controls, and +// returns that path. `workflow show`'s check declines relative paths on +// purpose, so the path a test registers from has to be a real absolute one. +func registerSourceAt(t *testing.T, conn *sql.DB, src string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "wf.toml") + err := os.WriteFile(path, []byte(src), 0o644) + testsupport.Must(t, err, "writing the definition: %v", err) + + cmd := workflowRegisterCmdWithDB(conn) + w, _ := bufWriter(true) + err = runWorkflowRegister(cmd, []string{path}, w) + testsupport.Must(t, err, "register: %v", err) + return path +} + +// showSourceStatus runs `workflow show --json` and returns the verdict. +func showSourceStatus(t *testing.T, conn *sql.DB, ref string) model.WorkflowSourceStatus { + t.Helper() + cmd := workflowShowCmdWithDB(conn, false) + w, buf := bufWriter(true) + err := runWorkflowShow(cmd, []string{ref}, w) + testsupport.Must(t, err, "show: %v", err) + + var envelope struct { + Data struct { + SourcePath string `json:"source_path"` + SourceSHA256 string `json:"source_sha256"` + SourceStatus model.WorkflowSourceStatus `json:"source_status"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if envelope.Data.SourceStatus.RegisteredSHA256 != envelope.Data.SourceSHA256 { + t.Errorf("source_status.registered_sha256 = %q, want the row's own %q", + envelope.Data.SourceStatus.RegisteredSHA256, envelope.Data.SourceSHA256) + } + return envelope.Data.SourceStatus +} + +// TestWorkflowShowReportsAMatchingSource: the clean case says so explicitly. +// Silence would be the defect all over again — "nothing was checked" and "the +// bytes still match" have to be distinguishable. +func TestWorkflowShowReportsAMatchingSource(t *testing.T) { + conn := newTestDB(t) + registerSourceAt(t, conn, minimalWorkflow) + + got := showSourceStatus(t, conn, "unit@1") + if got.State != model.WorkflowSourceMatches { + t.Errorf("state = %q, want %q (%+v)", got.State, model.WorkflowSourceMatches, got) + } + if got.CurrentSHA256 != got.RegisteredSHA256 { + t.Errorf("a matching source reports two different hashes: %+v", got) + } + + cmd := workflowShowCmdWithDB(conn, false) + w, buf := bufWriter(false) + err := runWorkflowShow(cmd, []string{"unit@1"}, w) + testsupport.Must(t, err, "show: %v", err) + if !strings.Contains(buf.String(), "source status: matches") { + t.Errorf("the human summary does not report the source status:\n%s", buf.String()) + } +} + +// TestWorkflowShowReportsADriftedSource is the filed defect: the file at the +// recorded path is now a DIFFERENT definition, and the row goes on reporting +// the hash it was registered with. +func TestWorkflowShowReportsADriftedSource(t *testing.T) { + conn := newTestDB(t) + path := registerSourceAt(t, conn, minimalWorkflow) + + // The v8-over-v4 shape: the file moves on, the registration does not. + bumped := strings.Replace(minimalWorkflow, "version = 1", "version = 8", 1) + testsupport.Must(t, os.WriteFile(path, []byte(bumped), 0o644), "rewriting %s", path) + + got := showSourceStatus(t, conn, "unit@1") + if got.State != model.WorkflowSourceDrifted { + t.Fatalf("state = %q, want %q (%+v)", got.State, model.WorkflowSourceDrifted, got) + } + // BOTH hashes, or the report cannot be acted on. + if got.CurrentSHA256 == "" || got.CurrentSHA256 == got.RegisteredSHA256 { + t.Errorf("drift reports one hash, not both: %+v", got) + } + if got.Path != path { + t.Errorf("path = %q, want the recorded %q", got.Path, path) + } + + cmd := workflowShowCmdWithDB(conn, false) + w, buf := bufWriter(false) + err := runWorkflowShow(cmd, []string{"unit@1"}, w) + testsupport.Must(t, err, "show: %v", err) + out := buf.String() + for _, want := range []string{"DRIFTED", got.CurrentSHA256, got.RegisteredSHA256} { + if !strings.Contains(out, want) { + t.Errorf("the human summary is missing %q:\n%s", want, out) + } + } +} + +// TestWorkflowShowReportsAMissingSourceDistinctly: a file that is GONE is not a +// file that CHANGED. Conflating them would tell an operator to bump a version +// when what they need is to restore a path — and would claim an on-disk hash +// that was never read. +func TestWorkflowShowReportsAMissingSourceDistinctly(t *testing.T) { + conn := newTestDB(t) + path := registerSourceAt(t, conn, minimalWorkflow) + testsupport.Must(t, os.Remove(path), "removing %s", path) + + got := showSourceStatus(t, conn, "unit@1") + if got.State != model.WorkflowSourceUnreadable { + t.Fatalf("state = %q, want %q (%+v)", got.State, model.WorkflowSourceUnreadable, got) + } + if got.CurrentSHA256 != "" { + t.Errorf("current_sha256 = %q on a file that was never read", got.CurrentSHA256) + } + if got.Reason == "" { + t.Error("an unreadable source carries no reason; the operator learns nothing") + } + + cmd := workflowShowCmdWithDB(conn, false) + w, buf := bufWriter(false) + err := runWorkflowShow(cmd, []string{"unit@1"}, w) + testsupport.Must(t, err, "show: %v", err) + if !strings.Contains(buf.String(), "UNREADABLE") { + t.Errorf("the human summary does not report the missing file:\n%s", buf.String()) + } +} + +// TestWorkflowShowReportsAnUncheckableSource: a definition registered from +// stdin has no file to compare against, and says so rather than reporting a +// verdict it did not reach. +func TestWorkflowShowReportsAnUncheckableSource(t *testing.T) { + conn := newTestDB(t) + cmd := workflowRegisterCmdWithDB(conn) + cmd.SetIn(strings.NewReader(minimalWorkflow)) + w, _ := bufWriter(true) + testsupport.Must(t, runWorkflowRegister(cmd, []string{"-"}, w), "register from stdin") + + got := showSourceStatus(t, conn, "unit@1") + if got.State != model.WorkflowSourceUnchecked { + t.Errorf("state = %q, want %q (%+v)", got.State, model.WorkflowSourceUnchecked, got) + } + if got.Reason == "" { + t.Error("an unchecked source carries no reason") + } +} diff --git a/internal/cli/workflow_test.go b/internal/cli/workflow_test.go index ee9c8c5c..911cde95 100644 --- a/internal/cli/workflow_test.go +++ b/internal/cli/workflow_test.go @@ -42,6 +42,8 @@ func workflowListCmdWithDB(conn *sql.DB, limit int) *cobra.Command { cmd := cmdWithDB(conn) cmd.Flags().String("name", "", "") cmd.Flags().Int("limit", limit, "") + cmd.Flags().Bool("orphans", false, "") + cmd.Flags().Bool("deprecated", false, "") return cmd } diff --git a/internal/db/engineconfig.go b/internal/db/engineconfig.go index d90b0faa..5e14c701 100644 --- a/internal/db/engineconfig.go +++ b/internal/db/engineconfig.go @@ -140,6 +140,17 @@ const ( // whichever pipeline held, so who answers it is a project-level policy. KeyVoteHoldRule = "vote.hold.rule" KeyVoteHoldVoters = "vote.hold.voters" + // KeyVoteHoldCost is the declared `expected_cost` a MATERIALIZED HELD vote + // step is minted with (DKT-584). An engine-minted held ballot has no + // `[[step]]` table to declare a cost in, so it carried 0 by construction — + // and since a vote step's declared cost accrues to the budget floor at + // materialization, that made every held panel invisible to the floor. + // + // Default "0", which is EXACTLY the prior behavior: a held ballot accrues + // nothing until an instance states what its panels cost. It applies only + // when a hold is minted as `vote` (both vote.hold.* keys set); a hold + // minted `human` is one operator's decision, not a panel's spend. + KeyVoteHoldCost = "vote.hold.cost" // KeyAutoRegister toggles §9's auto-registration: whether `run activate` // registers a workflow/schema it finds in an instance-config root @@ -408,6 +419,14 @@ var engineConfigSpecs = []ConfigSpec{ Doc: "Comma-separated voters on a materialized held step. Empty (the " + "default) mints held steps as `human` for one operator to decide", }, + { + Key: KeyVoteHoldCost, + Kind: KindNonNegativeNumber, + Default: "0", + Doc: "Declared expected_cost a materialized held VOTE step is minted " + + "with, accrued to the run's budget floor at materialization. 0 " + + "(the default) accrues nothing, the prior behavior", + }, { Key: KeyAutoRegister, Kind: KindBool, diff --git a/internal/db/gate_override_grants.go b/internal/db/gate_override_grants.go new file mode 100644 index 00000000..5a07ae1e --- /dev/null +++ b/internal/db/gate_override_grants.go @@ -0,0 +1,124 @@ +package db + +import ( + "database/sql" + "fmt" +) + +// GateOverrideGrant is one operator ruling that a gate's failure signature is +// environmental for the remainder of ONE run (DKT-546): later steps of the +// same run whose SAME gate fails with the SAME exit and reason auto-pass at +// routing instead of re-asking the operator. +// +// Exit is a POINTER for gate_results' own reason: an `unmatched` gate never +// ran, so it has no exit code, and NULL must match only NULL — "no process +// existed" is not exit 0. Reason is compared verbatim; it is the engine's +// classification field (unmatched cause, timeout, network annotation), not +// free prose. +// +// The grant is RUN-SCOPED BY CONSTRUCTION — the run_id foreign key is the +// whole scope rule. A new run re-asks, exactly as DKT-546 requires; nothing +// here consults project config, and there is deliberately no way to widen a +// grant past its run. +// +// RUN-SCOPED INCLUDES STEPS THAT DO NOT EXIST YET (DKT-734). A fix round mints +// new steps INSIDE the same run, so a grant taken on `fix@7`'s park covers +// `fix@8` and `fix@9` when a later round's same gate fails the same way. That +// is the design and not a leak: RUN-51 read three authorizations there where +// one standing authorization was spent three times. Nothing narrower is +// available — the row has an `origin_step_id`, but it is provenance, never +// consulted by grantMatches — and the safety is elsewhere: the gates still run +// every round, a round failing DIFFERENTLY still parks, and each application +// records its own `step-batch-overridden` event naming this row's id, so the +// ledger distinguishes one ruling spent N times from N rulings. +type GateOverrideGrant struct { + ID int + RunID int + OriginStepID int + Gate string + Exit *int + Reason string + Note string + CoveredSteps int + CreatedAtMS int64 +} + +// InsertGateOverrideGrantTx records one grant and returns its id. +// +// It takes a transaction because the grant commits with the resolution that +// raised it (the DKT-237 loop-grant discipline): a grant recorded without the +// override-pass would cover failures the operator never ruled on, and an +// override-pass without the grant would silently drop the ruling's reach. +func InsertGateOverrideGrantTx(tx *sql.Tx, g GateOverrideGrant) (int, error) { + var exit any + if g.Exit != nil { + exit = *g.Exit + } + res, err := tx.Exec( + `INSERT INTO gate_override_grants + (run_id, origin_step_id, gate, exit, reason, note, covered_steps, + created_at_ms) + VALUES (?, ?, ?, ?, ?, ?, 0, ?)`, + g.RunID, g.OriginStepID, g.Gate, exit, g.Reason, g.Note, g.CreatedAtMS) + if err != nil { + return 0, fmt.Errorf("recording the gate override grant: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, fmt.Errorf("reading the gate override grant id: %w", err) + } + return int(id), nil +} + +// GateOverrideGrantsForRun returns every grant recorded for one run, in +// insertion order. +func GateOverrideGrantsForRun(conn *sql.DB, runID int) ([]GateOverrideGrant, error) { + rows, err := conn.Query( + `SELECT id, run_id, origin_step_id, gate, exit, reason, note, + covered_steps, created_at_ms + FROM gate_override_grants WHERE run_id = ? ORDER BY id`, runID) + if err != nil { + return nil, fmt.Errorf("reading gate override grants: %w", err) + } + return scanRows(rows, "gate override grants", + func(r *sql.Rows) (GateOverrideGrant, error) { + var ( + g GateOverrideGrant + exit sql.NullInt64 + ) + if err := r.Scan( + &g.ID, &g.RunID, &g.OriginStepID, &g.Gate, &exit, &g.Reason, + &g.Note, &g.CoveredSteps, &g.CreatedAtMS, + ); err != nil { + return GateOverrideGrant{}, fmt.Errorf( + "reading a gate override grant: %w", err) + } + if exit.Valid { + code := int(exit.Int64) + g.Exit = &code + } + return g, nil + }) +} + +// CoverGateOverrideGrantsTx bumps `covered_steps` on each grant that just +// auto-passed one step, in the same transaction as the routing it decided — +// the counter and the pass must not be separable by a crash. +// +// A miss is a caller error rather than a no-op: the ids came from a read of +// this same table moments ago, and a grant that vanished between the read and +// the cover means the routing was decided on evidence that no longer stands. +func CoverGateOverrideGrantsTx(tx *sql.Tx, ids []int) error { + for _, id := range ids { + res, err := tx.Exec( + `UPDATE gate_override_grants SET covered_steps = covered_steps + 1 + WHERE id = ?`, id) + if err != nil { + return fmt.Errorf("covering gate override grant %d: %w", id, err) + } + if n, err := res.RowsAffected(); err == nil && n == 0 { + return fmt.Errorf("gate override grant %d no longer exists", id) + } + } + return nil +} diff --git a/internal/db/idempotency.go b/internal/db/idempotency.go index 1e6538cd..d3bba849 100644 --- a/internal/db/idempotency.go +++ b/internal/db/idempotency.go @@ -101,6 +101,38 @@ func LookupIdempotencyKeysTx(tx *sql.Tx, scope, prefix string) (map[string]int, return out, nil } +// LookupIdempotencyKeys is LookupIdempotencyKeysTx through the POOL, for a +// reader that holds no transaction — the run report's rollup phase, which +// deliberately reads everything it can before its snapshot opens (DKT-584's +// reap-ack attribution). Same query, same escaping; two spellings of one scan +// would drift at the first key family that mattered. +func LookupIdempotencyKeys(db *sql.DB, scope, prefix string) (map[string]int, error) { + rows, err := db.Query( + `SELECT key, entity_id FROM idempotency_keys + WHERE scope = ? AND key LIKE ? ESCAPE '\'`, + scope, escapeLike(prefix)+"%") + if err != nil { + return nil, fmt.Errorf("looking up idempotency keys: %w", err) + } + defer rows.Close() + + out := map[string]int{} + for rows.Next() { + var ( + key string + entityID int + ) + if err := rows.Scan(&key, &entityID); err != nil { + return nil, fmt.Errorf("scanning an idempotency key: %w", err) + } + out[key] = entityID + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("looking up idempotency keys: %w", err) + } + return out, nil +} + // escapeLike neutralizes LIKE's wildcards in a literal prefix. func escapeLike(s string) string { return strings.NewReplacer(`\`, `\\`, `%`, `\%`, `_`, `\_`).Replace(s) diff --git a/internal/db/migrate_v10_test.go b/internal/db/migrate_v10_test.go index 50f056ed..da009634 100644 --- a/internal/db/migrate_v10_test.go +++ b/internal/db/migrate_v10_test.go @@ -92,14 +92,15 @@ func TestMigrateToV10(t *testing.T) { // GAPS, because a missing entry turns Migrate's loop into a runtime error on a // user's database rather than a build-time failure on ours. func TestSchemaSpanIsComplete(t *testing.T) { - // The span now ends at v23. v11 through v23 are AMENDMENTS, not stages — + // The span now ends at v25. v11 through v25 are AMENDMENTS, not stages — // workflow retirement (DKT-21), the projects dimension (operator request, // 2026-08-09), vote provenance (DKT-71), per-seat vote spend (DKT-95), // artifact revisions (DKT-70), the retry-budget base (DKT-86/DKT-90), // vote-usage provenance (DKT-115), issue resolution (DKT-245), the // measured-usage cap (DKT-238), operator loop grants (DKT-237), the - // hollow-assurance marker (DKT-265), the pause origin (DKT-305), and the - // attempt-outcome breakdown (DKT-490) — each recorded in + // hollow-assurance marker (DKT-265), the pause origin (DKT-305), the + // attempt-outcome breakdown (DKT-490), the batch gate-override grant + // (DKT-546), and the stale-target waiver (DKT-742) — each recorded in // docs/tdd/reliability-delta.md §2 under its own heading, with the reason // it needed a version and the argument that it leaves the ratified v5-v10 // arithmetic untouched. @@ -109,11 +110,11 @@ func TestSchemaSpanIsComplete(t *testing.T) { // this test firing is exactly how v11 through v23 came to be documented // rather than discovered afterwards. Raising the number without editing // that section is the move it exists to stop. - if currentSchemaVersion != 23 { - t.Errorf("currentSchemaVersion = %d, want 23 — the span of "+ - "docs/tdd/reliability-delta.md §2 ends at v23 (the attempt-outcome "+ - "breakdown). Moving past 23 needs an amendment against "+ - "that section, per docs/design/amendments.md", currentSchemaVersion) + if currentSchemaVersion != 25 { + t.Errorf("currentSchemaVersion = %d, want 25 — the span of "+ + "docs/tdd/reliability-delta.md §2 ends at v25 (the stale-target "+ + "waiver). Moving past 25 needs an amendment against that "+ + "section, per docs/design/amendments.md", currentSchemaVersion) } for v := 2; v <= currentSchemaVersion; v++ { diff --git a/internal/db/migrate_v24_test.go b/internal/db/migrate_v24_test.go new file mode 100644 index 00000000..7125455f --- /dev/null +++ b/internal/db/migrate_v24_test.go @@ -0,0 +1,111 @@ +package db + +import ( + "regexp" + "sort" + "testing" +) + +// v24 — the batch gate-override grant table (DKT-546). Two obligations: the +// table arrives on a migrated store, and the rewind guard converges a store +// stamped 24 without it. There is deliberately no back-fill to test — no +// operator granted a batch override before the verb for granting one existed. +// The behavior the table exists for — a grant minted at resolve, spent at +// routing, dead with its run — is the engine's to prove. + +// v24Tables is stated independently of v24Sentinels so the sentinel test +// cannot pass by both lists drifting together — the v7/v8 discipline. +var v24Tables = []string{ + "gate_override_grants", +} + +func TestMigrateToV24(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + v, err := SchemaVersion(db) + if err != nil { + t.Fatalf("SchemaVersion: %v", err) + } + if v != currentSchemaVersion { + t.Errorf("schema_version = %d, want %d", v, currentSchemaVersion) + } + for _, table := range v24Tables { + if !hasTable(t, db, table) { + t.Errorf("%s missing after migration", table) + } + } + if !hasIndex(t, db, "idx_gate_override_grants_run") { + t.Error("idx_gate_override_grants_run missing after migration") + } +} + +// TestRewindGuardProbesEveryV24Sentinel derives the table list from the DDL +// itself: a later edit that adds a CREATE TABLE to v24DDL and forgets the +// sentinel fails HERE rather than shipping a database the guard silently +// declines to repair. +func TestRewindGuardProbesEveryV24Sentinel(t *testing.T) { + re := regexp.MustCompile(`(?i)CREATE TABLE IF NOT EXISTS\s+(\w+)`) + var created []string + for _, m := range re.FindAllStringSubmatch(v24DDL, -1) { + created = append(created, m[1]) + } + if len(created) == 0 { + t.Fatal("no CREATE TABLE statements found in v24DDL") + } + sort.Strings(created) + + for _, pair := range []struct { + name string + list []string + }{ + {"v24Sentinels", append([]string(nil), v24Sentinels...)}, + {"v24Tables", append([]string(nil), v24Tables...)}, + } { + got := pair.list + sort.Strings(got) + if len(got) != len(created) { + t.Fatalf("%s = %v, but v24DDL creates %v", pair.name, got, created) + } + for i := range created { + if got[i] != created[i] { + t.Errorf("%s = %v, but v24DDL creates %v", pair.name, got, created) + break + } + } + } +} + +// TestV24RewindGuardConvergesAStampedStore drops the table while leaving the +// stamp at 24 — the mid-change-binary database the guard comments describe — +// and asserts Migrate converges it. The TABLE form matters: v24 adds no +// column, so every v23 column sentinel is present on such a store and a +// column probe would never fire. +func TestV24RewindGuardConvergesAStampedStore(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + mustExec(t, db, `DROP TABLE gate_override_grants`) + if hasTable(t, db, "gate_override_grants") { + t.Fatal("the fixture did not remove the table it is testing the recovery of") + } + + if err := Migrate(db); err != nil { + t.Fatalf("re-running Migrate on the stamped store: %v", err) + } + for _, table := range v24Sentinels { + if !hasTable(t, db, table) { + t.Fatalf("the rewind guard did not converge %s back", table) + } + } +} diff --git a/internal/db/migrate_v25_test.go b/internal/db/migrate_v25_test.go new file mode 100644 index 00000000..c1277b1c --- /dev/null +++ b/internal/db/migrate_v25_test.go @@ -0,0 +1,112 @@ +package db + +import ( + "regexp" + "sort" + "testing" +) + +// v25 — the stale-target waiver table (DKT-742). The same two obligations as +// v24: the table arrives on a migrated store, and the rewind guard converges a +// store stamped 25 without it. There is deliberately no back-fill to test — no +// operator waived a stale-target warning before the verb for waiving one +// existed. The behavior the table exists for — a waiver minted at +// `dispatch waive-target`, consulted read-only by the advisory, dead with its +// run — is the engine's to prove. + +// v25Tables is stated independently of v25Sentinels so the sentinel test +// cannot pass by both lists drifting together — the v7/v8 discipline. +var v25Tables = []string{ + "stale_target_waivers", +} + +func TestMigrateToV25(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + v, err := SchemaVersion(db) + if err != nil { + t.Fatalf("SchemaVersion: %v", err) + } + if v != currentSchemaVersion { + t.Errorf("schema_version = %d, want %d", v, currentSchemaVersion) + } + for _, table := range v25Tables { + if !hasTable(t, db, table) { + t.Errorf("%s missing after migration", table) + } + } + if !hasIndex(t, db, "idx_stale_target_waivers_run") { + t.Error("idx_stale_target_waivers_run missing after migration") + } +} + +// TestRewindGuardProbesEveryV25Sentinel derives the table list from the DDL +// itself: a later edit that adds a CREATE TABLE to v25DDL and forgets the +// sentinel fails HERE rather than shipping a database the guard silently +// declines to repair. +func TestRewindGuardProbesEveryV25Sentinel(t *testing.T) { + re := regexp.MustCompile(`(?i)CREATE TABLE IF NOT EXISTS\s+(\w+)`) + var created []string + for _, m := range re.FindAllStringSubmatch(v25DDL, -1) { + created = append(created, m[1]) + } + if len(created) == 0 { + t.Fatal("no CREATE TABLE statements found in v25DDL") + } + sort.Strings(created) + + for _, pair := range []struct { + name string + list []string + }{ + {"v25Sentinels", append([]string(nil), v25Sentinels...)}, + {"v25Tables", append([]string(nil), v25Tables...)}, + } { + got := pair.list + sort.Strings(got) + if len(got) != len(created) { + t.Fatalf("%s = %v, but v25DDL creates %v", pair.name, got, created) + } + for i := range created { + if got[i] != created[i] { + t.Errorf("%s = %v, but v25DDL creates %v", pair.name, got, created) + break + } + } + } +} + +// TestV25RewindGuardConvergesAStampedStore drops the table while leaving the +// stamp at 25 — the mid-change-binary database the guard comments describe — +// and asserts Migrate converges it. The TABLE form matters: v25 adds no +// column, so every v24 sentinel is present on such a store and a column probe +// would never fire. +func TestV25RewindGuardConvergesAStampedStore(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + mustExec(t, db, `DROP TABLE stale_target_waivers`) + if hasTable(t, db, "stale_target_waivers") { + t.Fatal("the fixture did not remove the table it is testing the recovery of") + } + + if err := Migrate(db); err != nil { + t.Fatalf("re-running Migrate on the stamped store: %v", err) + } + for _, table := range v25Sentinels { + if !hasTable(t, db, table) { + t.Fatalf("the rewind guard did not converge %s back", table) + } + } +} diff --git a/internal/db/migrate_v7_test.go b/internal/db/migrate_v7_test.go index 1afa44be..0e3da369 100644 --- a/internal/db/migrate_v7_test.go +++ b/internal/db/migrate_v7_test.go @@ -82,6 +82,7 @@ func dropV7Tables(t *testing.T, db *sql.DB) { for _, table := range []string{ "usage_ledger", "dispatch_rows", "dispatches", "reap_acks", "action_results", + "gate_override_grants", "gate_results", "trust_cache", "events", "step_inputs", "artifacts", @@ -162,6 +163,7 @@ func TestMigrateV7UpgradesAPhase1Database(t *testing.T) { for _, table := range []string{ "usage_ledger", "dispatch_rows", "dispatches", "reap_acks", "action_results", + "gate_override_grants", "gate_results", "trust_cache", "events", "step_inputs", "artifacts", diff --git a/internal/db/proposals.go b/internal/db/proposals.go index a1cf09e4..d7c2530c 100644 --- a/internal/db/proposals.go +++ b/internal/db/proposals.go @@ -106,9 +106,29 @@ func CreateProposalIdempotent(db *sql.DB, p *model.Proposal, idempotencyKey stri return id, nil } +// proposalQuerier is the read surface GetProposal/GetProposalVotes need, +// satisfied by *sql.DB and *sql.Tx alike — one query text for both callers, so +// the pooled and the in-transaction reads cannot drift. +type proposalQuerier interface { + QueryRow(query string, args ...any) *sql.Row + Query(query string, args ...any) (*sql.Rows, error) +} + // GetProposal returns a proposal by ID, or ErrNotFound if it does not exist. func GetProposal(db *sql.DB, id int) (*model.Proposal, error) { - row := db.QueryRow( + return getProposal(db, id) +} + +// GetProposalTx is GetProposal inside a CALLER'S transaction — the engine's +// context assembly resolves `.vote-record` inputs (DKT-545) in the claim +// transaction, and reading through the pool there would see a different +// snapshot than the bundle it is assembling. +func GetProposalTx(tx *sql.Tx, id int) (*model.Proposal, error) { + return getProposal(tx, id) +} + +func getProposal(q proposalQuerier, id int) (*model.Proposal, error) { + row := q.QueryRow( `SELECT id, description, rationale, domain_tags, files_changed, criticality, status, final_outcome, escalation_reason, required_voters, threshold, weighted_score, created_by, created_at, updated_at FROM proposals WHERE id = ?`, id, ) @@ -389,7 +409,17 @@ func CastVote(db *sql.DB, v *model.Vote) (*CastVoteResult, error) { // GetProposalVotes returns all votes for a proposal, ordered by creation time. func GetProposalVotes(db *sql.DB, proposalID int) ([]*model.Vote, error) { - rows, err := db.Query( + return getProposalVotes(db, proposalID) +} + +// GetProposalVotesTx is GetProposalVotes inside a CALLER'S transaction — see +// GetProposalTx for why the vote-record input resolution needs one. +func GetProposalVotesTx(tx *sql.Tx, proposalID int) ([]*model.Vote, error) { + return getProposalVotes(tx, proposalID) +} + +func getProposalVotes(q proposalQuerier, proposalID int) ([]*model.Vote, error) { + rows, err := q.Query( `SELECT id, proposal_id, voter_name, voter_role, verdict, confidence, domain_relevance, findings, findings_json, summary, metadata, created_at FROM votes WHERE proposal_id = ? ORDER BY created_at ASC`, proposalID, ) diff --git a/internal/db/rollups.go b/internal/db/rollups.go index 97c7178b..497e1b8b 100644 --- a/internal/db/rollups.go +++ b/internal/db/rollups.go @@ -5,6 +5,7 @@ import ( "encoding/json" "fmt" "sort" + "strings" "github.com/ALT-F4-LLC/docket/internal/model" ) @@ -40,11 +41,27 @@ type VerdictCount struct { // is an honest zero and not a missing feature: an action's `builtin` column // already says whether core computed it. Stub int `json:"stub"` + // Pre counts rows recorded by the PRE-GATE phase (DKT-862). Like Stub it + // counts ROWS and overlaps the four verdict columns rather than + // partitioning with them, because it describes WHEN the row was produced, + // not what it decided. + // + // It matters because a pre-gate NEVER ROUTES: §11.1 runs it at claim, with + // its results carried into the step's context bundle, and PG4 excludes it + // from the saga's verdict. Without this count `ac-commands: pass 0, fail 1` + // read identically whether the failure blocked the step or was an advisory + // input to it — RUN-61 had three such rows and a conductor nearly reported + // a fix round as burned on one of them. + // + // It reads the SAME `pre` column `step show`'s `[pre]` marker reads, so the + // two surfaces cannot disagree about which gates were advisory. It is zero + // for actions, which have no pre phase. + Pre int `json:"pre"` } // GateRollup counts a run's gate results per gate name (R4). func GateRollup(db *sql.DB, runID int) ([]VerdictCount, error) { - return verdictRollup(db, `gate_results`, `gate`, `stub_entry`, runID) + return verdictRollup(db, `gate_results`, `gate`, `stub_entry`, `pre`, runID) } // ActionRollup is the same rollup over `action_results` (R5). @@ -60,10 +77,11 @@ func GateRollup(db *sql.DB, runID int) ([]VerdictCount, error) { // shape — the alternative, omitting the column for one of the two seams, is // how the two definitions of "what outcomes exist" start to drift. func ActionRollup(db *sql.DB, runID int) ([]VerdictCount, error) { - // The empty stub column is the shape's one asymmetry, and it is a fact - // about the tables rather than an omission: only a GATE runs a - // trust-authorized command that an operator could have declared a stub. - return verdictRollup(db, `action_results`, `action`, ``, runID) + // The empty stub and pre columns are the shape's one asymmetry, and they + // are a fact about the tables rather than an omission: only a GATE runs a + // trust-authorized command that an operator could have declared a stub, and + // only a gate has a pre phase. + return verdictRollup(db, `action_results`, `action`, ``, ``, runID) } // verdictRollup is the shared body. The table and column names are INTERNAL @@ -71,24 +89,30 @@ func ActionRollup(db *sql.DB, runID int) ([]VerdictCount, error) { // carries no injection surface — and the query is parameterized on everything // that does come from outside. func verdictRollup( - db *sql.DB, table, subject, stubColumn string, runID int, + db *sql.DB, table, subject, stubColumn, preColumn string, runID int, ) ([]VerdictCount, error) { // A table with no stub column selects the literal 0 rather than being // given a second query. One query means one definition of "pass" for both - // seams, which is the property the shared body exists to hold. - stubSum := `0` + // seams, which is the property the shared body exists to hold. `pre` is + // absent from `action_results` for the same reason and takes the same + // treatment. + stubSum, preSum := `0`, `0` if stubColumn != "" { stubSum = fmt.Sprintf("SUM(CASE WHEN %s = 1 THEN 1 ELSE 0 END)", stubColumn) } + if preColumn != "" { + preSum = fmt.Sprintf("SUM(CASE WHEN %s = 1 THEN 1 ELSE 0 END)", preColumn) + } rows, err := db.Query(fmt.Sprintf( `SELECT %[2]s, SUM(CASE WHEN verdict = 'pass' THEN 1 ELSE 0 END), SUM(CASE WHEN verdict = 'fail' THEN 1 ELSE 0 END), SUM(CASE WHEN verdict = 'unmatched' THEN 1 ELSE 0 END), SUM(CASE WHEN verdict = 'skipped' THEN 1 ELSE 0 END), - %[3]s + %[3]s, + %[4]s FROM %[1]s WHERE run_id = ? GROUP BY %[2]s ORDER BY %[2]s`, - table, subject, stubSum), runID) + table, subject, stubSum, preSum), runID) if err != nil { return nil, fmt.Errorf("rolling up %s for %s: %w", table, model.FormatRunID(runID), err) @@ -99,6 +123,7 @@ func verdictRollup( var v VerdictCount if err := r.Scan( &v.Name, &v.Pass, &v.Fail, &v.Unmatched, &v.Skipped, &v.Stub, + &v.Pre, ); err != nil { return VerdictCount{}, fmt.Errorf("reading a %s rollup row: %w", table, err) } @@ -262,14 +287,18 @@ func VoteMetadataRollup(db *sql.DB, scope, prefix string) ([]MetadataKeyRollup, // The proposals are selected by the same caller-supplied idempotency scope // and prefix VoteMetadataRollup reads by, and units stay opaque: summed and // counted, never interpreted. -func VoteUsageRollup(db *sql.DB, scope, prefix string) ([]UnitTotal, error) { +// +// extraIDs widens the selection to proposals the caller attributed to the run +// some OTHER way than the vote-step key family (DKT-584): reap-ack ballots and +// conversational-gate proposals that name the run in their text. Membership is +// a set test per vote row, so an id also matched by the prefix counts once. +func VoteUsageRollup(db *sql.DB, scope, prefix string, extraIDs ...int) ([]UnitTotal, error) { + clause, args := proposalMembership(scope, prefix, extraIDs) rows, err := db.Query( `SELECT vu.unit, SUM(vu.quantity), COUNT(*) FROM vote_usage vu JOIN votes v ON v.id = vu.vote_id - WHERE v.proposal_id IN ( - SELECT entity_id FROM idempotency_keys - WHERE scope = ? AND key LIKE ? ESCAPE '\') - GROUP BY vu.unit ORDER BY vu.unit`, scope, escapeLike(prefix)+"%") + WHERE `+clause+` + GROUP BY vu.unit ORDER BY vu.unit`, args...) if err != nil { return nil, fmt.Errorf("rolling up vote usage: %w", err) } @@ -283,6 +312,96 @@ func VoteUsageRollup(db *sql.DB, scope, prefix string) ([]UnitTotal, error) { }) } +// proposalMembership builds the WHERE fragment that selects a run's vote +// casts: the idempotency-key family, plus any explicitly attributed proposal +// ids. One builder for both vote_usage readers, so the rollup and its +// coverage line cannot disagree about which casts belong to the run. +// +// The placeholder list is built from the COUNT of ids, never their values, +// and every id is bound — the discipline ProposalStatusesTx states. +func proposalMembership(scope, prefix string, extraIDs []int) (string, []any) { + clause := `v.proposal_id IN ( + SELECT entity_id FROM idempotency_keys + WHERE scope = ? AND key LIKE ? ESCAPE '\')` + args := []any{scope, escapeLike(prefix) + "%"} + if len(extraIDs) > 0 { + placeholders := strings.TrimSuffix(strings.Repeat("?,", len(extraIDs)), ",") + clause = "(" + clause + ` OR v.proposal_id IN (` + placeholders + `))` + for _, id := range extraIDs { + args = append(args, id) + } + } + return clause, args +} + +// ProposalIDsNaming returns the ids of proposals whose description or +// rationale names `token` as a whole word, ordered by id (DKT-584). +// +// It exists for the conversational-gate ballots a conductor opens with `vote +// create` — an activation panel, say — which carry no idempotency key and no +// step, and whose ONLY link to the run they gate is that their text names it +// ("activation panel for RUN-40"). The LIKE narrows candidates in SQL; the +// word-boundary check runs in Go because LIKE '%RUN-4%' would also match +// RUN-40, and a run must never inherit another run's panels. +func ProposalIDsNaming(db *sql.DB, token string) ([]int, error) { + if token == "" { + return nil, nil + } + pattern := "%" + escapeLike(token) + "%" + rows, err := db.Query( + `SELECT id, description, rationale FROM proposals + WHERE description LIKE ? ESCAPE '\' OR rationale LIKE ? ESCAPE '\' + ORDER BY id`, pattern, pattern) + if err != nil { + return nil, fmt.Errorf("finding proposals naming %q: %w", token, err) + } + defer rows.Close() + + var out []int + for rows.Next() { + var ( + id int + description, rationale string + ) + if err := rows.Scan(&id, &description, &rationale); err != nil { + return nil, fmt.Errorf("reading a proposal naming %q: %w", token, err) + } + if namesToken(description, token) || namesToken(rationale, token) { + out = append(out, id) + } + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("finding proposals naming %q: %w", token, err) + } + return out, nil +} + +// namesToken reports whether text contains token as a whole word: not +// preceded by an ASCII letter or digit, and not followed by a digit — so +// "RUN-4" never matches inside "RUN-40" or "XRUN-4", while "(RUN-4)" and +// "RUN-4," match. +func namesToken(text, token string) bool { + for from := 0; ; { + at := strings.Index(text[from:], token) + if at < 0 { + return false + } + start := from + at + end := start + len(token) + beforeOK := start == 0 || !isWordByte(text[start-1]) + afterOK := end == len(text) || !(text[end] >= '0' && text[end] <= '9') + if beforeOK && afterOK { + return true + } + from = start + 1 + } +} + +// isWordByte is namesToken's boundary alphabet: ASCII letters and digits. +func isWordByte(b byte) bool { + return (b >= 'a' && b <= 'z') || (b >= 'A' && b <= 'Z') || (b >= '0' && b <= '9') +} + // rollupMetadataRows is the one counting loop both rollups share: raw bags in, // sorted key -> value -> count out. func rollupMetadataRows(rows *sql.Rows) ([]MetadataKeyRollup, error) { @@ -306,7 +425,7 @@ func rollupMetadataRows(rows *sql.Rows) ([]MetadataKeyRollup, error) { if counts[key] == nil { counts[key] = make(map[string]int) } - counts[key][renderMetadataValue(value)]++ + counts[key][RenderMetadataValue(value)]++ } } if err := rows.Err(); err != nil { @@ -339,13 +458,19 @@ type MetadataValueCount struct { Count int `json:"count"` } -// renderMetadataValue turns one opaque metadata value into the string the +// RenderMetadataValue turns one opaque metadata value into the string the // rollup groups by. // // A non-string value is rendered as its JSON, which keeps `1` and `"1"` // DISTINCT — they are different values in the bag, and merging them would be // the rollup deciding they mean the same thing. -func renderMetadataValue(v any) string { +// +// EXPORTED for DKT-868's per-step section, which renders the same column in the +// same document: the rollup spells a value one way and a report that spelled the +// per-step copy another would show a reader `1` in one section and `"1"` in the +// next and invite them to conclude the two came from different bags. One +// definition, so the two cannot disagree. +func RenderMetadataValue(v any) string { if s, ok := v.(string); ok { return s } @@ -398,7 +523,11 @@ func (c VoteUsageCoverage) Silent() int { return c.Casts - c.Reported } // number. What it can do is stop the silence from looking like a zero, which is // this: a run whose seats all reported reads `12/12`, and one whose seats // reported nothing reads `0/12` instead of an absent section. -func VoteUsageCoverageFor(db *sql.DB, scope, prefix string) (VoteUsageCoverage, error) { +// extraIDs widens the count to explicitly attributed proposals exactly as +// VoteUsageRollup's does (DKT-584), through the same membership builder — so +// a cast counted in the rollup is always counted in its coverage line. +func VoteUsageCoverageFor(db *sql.DB, scope, prefix string, extraIDs ...int) (VoteUsageCoverage, error) { + clause, args := proposalMembership(scope, prefix, extraIDs) var out VoteUsageCoverage err := db.QueryRow( // COALESCE, because SUM over ZERO ROWS is NULL in SQLite and a run with @@ -411,12 +540,53 @@ func VoteUsageCoverageFor(db *sql.DB, scope, prefix string) (VoteUsageCoverage, SELECT 1 FROM vote_usage vu WHERE vu.vote_id = v.id ) THEN 1 ELSE 0 END), 0) FROM votes v - WHERE v.proposal_id IN ( - SELECT entity_id FROM idempotency_keys - WHERE scope = ? AND key LIKE ? ESCAPE '\')`, - scope, escapeLike(prefix)+"%").Scan(&out.Casts, &out.Reported) + WHERE `+clause, + args...).Scan(&out.Casts, &out.Reported) if err != nil { return VoteUsageCoverage{}, fmt.Errorf("counting vote usage coverage: %w", err) } return out, nil } + +// SilentVoteSeatRow is one cast VoteUsageCoverageFor counts as silent: which +// seat, on which proposal, reported no spend at all (DKT-733). +// +// The coverage COUNT told an operator that 12 of 57 seats went silent and +// nothing anywhere said WHICH twelve — so the backfill verb that exists +// precisely to close the gap (`vote backfill-usage`, DKT-115) could not be +// aimed. This row is the identity behind the count. +// +// The seating path is deliberately NOT here: this package selects by an +// opaque scope and prefix and cannot know which key family or run-naming rule +// minted a proposal. The engine, which owns those spellings, labels each row +// (engine.SilentVoteSeat). +type SilentVoteSeatRow struct { + ProposalID int + Voter string + Role string +} + +// SilentVoteSeatsFor enumerates the casts VoteUsageCoverageFor counts as +// silent, through the SAME membership builder — so the list and the count it +// explains cannot disagree about which casts belong to the run. Ordered by +// (proposal, voter): a total key, per R9. +func SilentVoteSeatsFor(db *sql.DB, scope, prefix string, extraIDs ...int) ([]SilentVoteSeatRow, error) { + clause, args := proposalMembership(scope, prefix, extraIDs) + rows, err := db.Query( + `SELECT v.proposal_id, v.voter_name, v.voter_role + FROM votes v + WHERE `+clause+` + AND NOT EXISTS(SELECT 1 FROM vote_usage vu WHERE vu.vote_id = v.id) + ORDER BY v.proposal_id, v.voter_name`, args...) + if err != nil { + return nil, fmt.Errorf("listing silent vote seats: %w", err) + } + return scanRows(rows, "silent vote seats", + func(r *sql.Rows) (SilentVoteSeatRow, error) { + var row SilentVoteSeatRow + if err := r.Scan(&row.ProposalID, &row.Voter, &row.Role); err != nil { + return row, err + } + return row, nil + }) +} diff --git a/internal/db/runs.go b/internal/db/runs.go index ee226041..b7e266b3 100644 --- a/internal/db/runs.go +++ b/internal/db/runs.go @@ -444,6 +444,34 @@ func BindRunIssueTx(tx *sql.Tx, ri *RunIssue) error { return nil } +// SetRunIssueSnapshotTx rewrites ONE run-issue's `issue_snapshot` blob and +// nothing else (DKT-869). +// +// It is deliberately narrower than BindRunIssueTx, which writes the binding and +// all three snapshot columns together because activation produces them as one +// fact. The scope refresh is not that fact: it re-reads exactly one field of an +// already-frozen snapshot and must leave `workflow_id`, `body_snapshot`, and +// `body_sha256` untouched — a refresh that re-bound the workflow, or rewrote +// the description snapshot from the live issue, would smuggle §9 item 5's +// mid-run edit immunity out through a verb that says it is about scope. +// +// A miss is a caller error rather than a no-op: the row was read moments ago +// inside this same transaction, so zero rows affected means the membership the +// decision was made on no longer stands. +func SetRunIssueSnapshotTx(tx *sql.Tx, runID, issueID int, snapshot string) error { + res, err := tx.Exec( + `UPDATE run_issues SET issue_snapshot = ? WHERE run_id = ? AND issue_id = ?`, + snapshot, runID, issueID, + ) + if err != nil { + return fmt.Errorf("rewriting issue %d's snapshot: %w", issueID, err) + } + if n, err := res.RowsAffected(); err == nil && n == 0 { + return fmt.Errorf("issue %d is no longer part of run %d", issueID, runID) + } + return nil +} + // MarkExpandedTx stamps an issue as expanded — stage 6's record that this // issue's phase is done, so re-activation (RA1) skips it. func MarkExpandedTx(tx *sql.Tx, runID, issueID int, nowMS int64) error { diff --git a/internal/db/schema.go b/internal/db/schema.go index ed7345b5..034c999d 100644 --- a/internal/db/schema.go +++ b/internal/db/schema.go @@ -10,7 +10,7 @@ import ( "github.com/ALT-F4-LLC/docket/internal/schema" ) -const currentSchemaVersion = 23 +const currentSchemaVersion = 25 // schemaDDL contains the CREATE TABLE statements for the initial schema. // @@ -192,6 +192,8 @@ var migrations = map[int]func(tx *sql.Tx) error{ 21: migrateV20ToV21, 22: migrateV21ToV22, 23: migrateV22ToV23, + 24: migrateV23ToV24, + 25: migrateV24ToV25, } // migrationsNeedingFKOff names the migrations that REBUILD tables and so must @@ -2195,6 +2197,107 @@ func migrateV22ToV23(tx *sql.Tx) error { return nil } +// v24Sentinels is the table the v24 DDL creates, probed by the rewind guard in +// the TABLE form v7 and v8 use. TestRewindGuardProbesEveryV24Sentinel derives +// the list from the DDL, so an added table cannot ship without its sentinel. +var v24Sentinels = []string{ + "gate_override_grants", +} + +// v24DDL is the run-scoped batch gate-override grant (DKT-546): one operator +// ruling that a gate's failure signature is environmental, covering later +// steps of the SAME run that fail the same gate with the same exit and reason. +// +// It is a TABLE rather than a `run_issues` column (the v20 loop-grant shape) +// because a grant is keyed by (run, gate, signature), not by (run, issue) — +// one ruling covers every issue's steps in the run, which is the toil DKT-546 +// measured. `exit` is nullable for gate_results' own reason: an `unmatched` +// gate never ran, and NULL is the honest encoding of "no process existed" — +// a NULL signature matches only NULL, never exit 0. +// +// `covered_steps` counts the steps the grant auto-passed, bumped in the same +// transaction as each auto-pass, so the ledger shows one tracked grant +// covering N steps rather than N unattributed passes. +const v24DDL = ` +CREATE TABLE IF NOT EXISTS gate_override_grants ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER NOT NULL REFERENCES runs(id) ON DELETE CASCADE, + origin_step_id INTEGER NOT NULL REFERENCES steps(id) ON DELETE CASCADE, + gate TEXT NOT NULL, + exit INTEGER, + reason TEXT NOT NULL DEFAULT '', + note TEXT NOT NULL DEFAULT '', + covered_steps INTEGER NOT NULL DEFAULT 0, + created_at_ms INTEGER NOT NULL +); + +CREATE INDEX IF NOT EXISTS idx_gate_override_grants_run + ON gate_override_grants(run_id); +` + +// migrateV23ToV24 creates the batch gate-override grant table (DKT-546). +// +// Additive and dormant: it creates one new table and touches no existing one, +// so a run that never records a grant reads byte-identically to v23. There is +// nothing to back-fill — no operator has granted a batch override before the +// verb for granting one existed. +func migrateV23ToV24(tx *sql.Tx) error { + if _, err := tx.Exec(v24DDL); err != nil { + return fmt.Errorf("migrating v23 to v24: %w", err) + } + return nil +} + +// v25Sentinels is the table the v25 DDL creates, probed by the rewind guard in +// the TABLE form v24 uses. TestRewindGuardProbesEveryV25Sentinel derives the +// list from the DDL, so an added table cannot ship without its sentinel. +var v25Sentinels = []string{ + "stale_target_waivers", +} + +// v25DDL is the run-scoped stale-target waiver (DKT-742): one operator ruling +// that a specific (step instance, target sha) stale-target warning has been +// adjudicated, so dispatch open/verify stop re-firing it unchanged. RUN-52 saw +// the identical warning fire four times across DISPATCH-295/297/301, each +// firing costing an investigation, and the operator's standing waiver lived +// only in session memory where the engine could not see it. +// +// It is a TABLE beside gate_override_grants rather than a column because the +// two are the same shape — a standing adjudication matched by signature, +// run-scoped by the run_id foreign key, dead with its run. The signature here +// is (step_instance, target_sha): a different sha on the same row, or the same +// sha on a different row, is a different question and still warns. +// +// `target_sha` may be an unambiguous PREFIX of the recorded sha (>= 7 hex +// chars), because the warning an operator copies it from renders the sha at 12 +// characters; matching is case-insensitive prefix. +const v25DDL = ` +CREATE TABLE IF NOT EXISTS stale_target_waivers ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER NOT NULL REFERENCES runs(id) ON DELETE CASCADE, + step_instance TEXT NOT NULL, + target_sha TEXT NOT NULL, + note TEXT NOT NULL DEFAULT '', + created_at_ms INTEGER NOT NULL +); + +CREATE INDEX IF NOT EXISTS idx_stale_target_waivers_run + ON stale_target_waivers(run_id); +` + +// migrateV24ToV25 creates the stale-target waiver table (DKT-742). +// +// Additive and dormant, exactly as v24 was: it creates one new table and +// touches no existing one, so a run that never records a waiver reads +// byte-identically to v24. There is nothing to back-fill — no operator has +// waived a stale-target warning before the verb for waiving one existed. +func migrateV24ToV25(tx *sql.Tx) error { + if _, err := tx.Exec(v25DDL); err != nil { + return fmt.Errorf("migrating v24 to v25: %w", err) + } + return nil +} + // migrateV19ToV20 adds the operator loop-grant column. // // It BACK-FILLS NOTHING, and zero is the correct value for every existing row: @@ -2712,6 +2815,42 @@ func Migrate(db *sql.DB) error { } } + // The v24 guard, back in the TABLE form v7 and v8 use and for their reason: + // v24 adds a table and no columns, so a database stamped 24 by a binary + // built mid-change carries every v23 sentinel and the grants table never + // arrives. The v24 migration is CREATE TABLE IF NOT EXISTS throughout, so + // re-running it against such a store is safe. + if version >= 24 { + for _, table := range v24Sentinels { + exists, err := tableExists(db, table) + if err != nil { + break + } + if !exists { + version = 23 + break + } + } + } + + // The v25 guard, same TABLE form as v24 and for its reason: v25 adds a + // table and no columns, so a database stamped 25 by a binary built + // mid-change carries every v24 sentinel and the waiver table never + // arrives. The v25 migration is CREATE TABLE IF NOT EXISTS throughout, so + // re-running it against such a store is safe. + if version >= 25 { + for _, table := range v25Sentinels { + exists, err := tableExists(db, table) + if err != nil { + break + } + if !exists { + version = 24 + break + } + } + } + if version == currentSchemaVersion { return nil } diff --git a/internal/db/stale_target_waivers.go b/internal/db/stale_target_waivers.go new file mode 100644 index 00000000..837a3cf6 --- /dev/null +++ b/internal/db/stale_target_waivers.go @@ -0,0 +1,86 @@ +package db + +import ( + "database/sql" + "fmt" +) + +// StaleTargetWaiver is one operator ruling that a specific stale-target +// warning — one (step instance, target sha) pair — has been adjudicated for +// the remainder of ONE run (DKT-742): dispatch open/verify stop re-firing the +// identical warning, and the standing precedent is engine-visible instead of +// session-memory-only. +// +// The waiver is RUN-SCOPED BY CONSTRUCTION — the run_id foreign key is the +// whole scope rule, exactly gate_override_grants' shape (DKT-546). A new run +// re-warns, and there is deliberately no way to widen a waiver past its run. +// +// THE SIGNATURE IS THE PAIR AND NOTHING ELSE. The warning's rendered reason +// names the shared HEAD, which moves at every integration — RUN-52's four +// firings of one adjudicated warning each carried a different HEAD — so the +// reason is not part of the match. A DIFFERENT target sha on the same row, or +// the same sha on a row this waiver does not name, is a different question +// and still warns. +// +// TargetSHA may be a PREFIX of the recorded sha (>= 7 hex characters, +// enforced at the verb): the warning an operator copies it from renders the +// sha at 12 characters, and demanding the full 40 would make the verb +// unusable from its own advisory. Matching is case-insensitive prefix +// (engine: waiverCovers). +type StaleTargetWaiver struct { + ID int + RunID int + StepInstance string + TargetSHA string + Note string + CreatedAtMS int64 +} + +// InsertStaleTargetWaiverTx records one waiver and returns its id. +// +// It takes a transaction because the verb records one waiver per named step +// plus one event per waiver, and a batch that half-applied would leave the +// operator's ruling covering rows they never listed — or missing rows they +// did. +func InsertStaleTargetWaiverTx(tx *sql.Tx, w StaleTargetWaiver) (int, error) { + res, err := tx.Exec( + `INSERT INTO stale_target_waivers + (run_id, step_instance, target_sha, note, created_at_ms) + VALUES (?, ?, ?, ?, ?)`, + w.RunID, w.StepInstance, w.TargetSHA, w.Note, w.CreatedAtMS) + if err != nil { + return 0, fmt.Errorf("recording the stale-target waiver: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, fmt.Errorf("reading the stale-target waiver id: %w", err) + } + return int(id), nil +} + +// StaleTargetWaiversForRun returns every waiver recorded for one run, in +// insertion order. +// +// It reads on the pooled connection because its one consumer — staleTargets — +// runs OUTSIDE every transaction by design (§6: no subprocess inside one), +// exactly as batchOverrideCover reads grants. +func StaleTargetWaiversForRun(conn *sql.DB, runID int) ([]StaleTargetWaiver, error) { + rows, err := conn.Query( + `SELECT id, run_id, step_instance, target_sha, note, created_at_ms + FROM stale_target_waivers WHERE run_id = ? ORDER BY id`, runID) + if err != nil { + return nil, fmt.Errorf("reading stale-target waivers: %w", err) + } + return scanRows(rows, "stale-target waivers", + func(r *sql.Rows) (StaleTargetWaiver, error) { + var w StaleTargetWaiver + if err := r.Scan( + &w.ID, &w.RunID, &w.StepInstance, &w.TargetSHA, &w.Note, + &w.CreatedAtMS, + ); err != nil { + return StaleTargetWaiver{}, fmt.Errorf( + "reading a stale-target waiver: %w", err) + } + return w, nil + }) +} diff --git a/internal/db/steps.go b/internal/db/steps.go index e605a363..e6d68650 100644 --- a/internal/db/steps.go +++ b/internal/db/steps.go @@ -762,6 +762,47 @@ func ResetStepRetryBudgetTx(tx *sql.Tx, id int, nowMS int64) error { return nil } +// ExemptStepAttemptFromBudgetTx exempts ONE spent attempt from a step's retry +// budget — the forced reap's classification (DKT-585) — by moving +// `attempt_base` forward by exactly one. Exhaustion compares +// `attempt - attempt_base` against `max_attempts`, so from here on the budget +// reads one fewer spent: the attempt still happened, it just does not count +// against the declared allowance. +// +// It is a NUDGE, not ResetStepRetryBudgetTx's reset. A forced reap asserts +// that ONE claim's holder is gone — a relay declaring a dead spawn is not an +// executor failure (RUN-30 STEP-755: a wave stopped by an accidental +// interrupt consumed the step's last attempt, leaving it one interrupt from +// `waiting-human` on a healthy charter) — and the exemption must be as narrow +// as the assertion: this one attempt, nothing else. `attempt_base = attempt` +// here would also forgive every EARLIER genuinely-failed attempt, silently +// granting a full fresh budget on a verb that never claimed to be a retry. +// +// `attempt` ITSELF IS UNTOUCHED, for exactly ResetStepRetryBudgetTx's reason +// (DKT-86, DKT-90): it is the usage ledger's key half (`UNIQUE(step_id, +// attempt, unit)`), and the dead attempt's usage stays back-fillable against +// its own attempt number after this runs — RUN-30's dead attempt had real +// measured usage back-filled, and decrementing the counter would have made +// that row collide with the successor's, the RUN-13 STEP-132 failure mode +// ReleaseStepLeaseTx documents. +// +// The `attempt_base < attempt` guard keeps the base at or below the counter. +// Without it, reaching this twice for one claim — or on a row whose base has +// already caught up — would push the base PAST the counter, and +// `attempt - attempt_base` would go negative: budget minted out of nothing. +func ExemptStepAttemptFromBudgetTx(tx *sql.Tx, id int, nowMS int64) error { + _, err := tx.Exec( + `UPDATE steps SET attempt_base = attempt_base + 1, updated_at_ms = ?, + row_version = row_version + 1 + WHERE id = ? AND attempt_base < attempt`, + nowMS, id, + ) + if err != nil { + return fmt.Errorf("exempting the reaped attempt from the retry budget: %w", err) + } + return nil +} + // Artifact is one `artifacts` row: what a step produced. type Artifact struct { ID int @@ -860,6 +901,21 @@ func ListStepArtifacts(db *sql.DB, stepID int) ([]*Artifact, error) { return scanArtifacts(db.Query(artifactSelect+` WHERE step_id = ? ORDER BY id`, stepID)) } +// GetArtifactTx reads ONE artifact by id, inside a transaction — the reader +// for an id pinned at activation (DKT-547's `issue.linked` form), where the +// row may belong to another run entirely and the run-scoped listing above +// cannot reach it. ErrNotFound when no such row exists. +func GetArtifactTx(tx *sql.Tx, id int) (*Artifact, error) { + out, err := scanArtifacts(tx.Query(artifactSelect+` WHERE id = ?`, id)) + if err != nil { + return nil, err + } + if len(out) == 0 { + return nil, ErrNotFound + } + return out[0], nil +} + func scanArtifacts(rows *sql.Rows, err error) ([]*Artifact, error) { if err != nil { return nil, fmt.Errorf("listing artifacts: %w", err) diff --git a/internal/db/workflows.go b/internal/db/workflows.go index fb6d1ae4..76d1f514 100644 --- a/internal/db/workflows.go +++ b/internal/db/workflows.go @@ -116,8 +116,14 @@ func InsertWorkflowTx( // GetWorkflow returns one registered workflow WITHIN ONE PROJECT (v12 — a // name@version is a per-project registration). A version of 0 selects the -// HIGHEST registered version, which is what `workflow show NAME` without -// `@version` means. +// HIGHEST NON-DEPRECATED registered version, which is what `workflow show +// NAME` without `@version` means — and, by design, the same version +// `run activate` binds (engine.bindableDefinitions retires deprecated rows +// first, then takes the highest of what remains; DKT-616). A name whose every +// version is deprecated therefore resolves to ErrWorkflowNotFound, mirroring +// binding removing the name from routing altogether. An explicit version +// still resolves a deprecated row: retired versions stay registered and +// reachable for the runs that pinned them. func GetWorkflow(db *sql.DB, projectID int, name string, version int) (*model.Workflow, error) { projectID = projectOrDefault(projectID) if version > 0 { @@ -126,7 +132,8 @@ func GetWorkflow(db *sql.DB, projectID int, name string, version int) (*model.Wo projectID, name, version)) } return scanWorkflow(db.QueryRow( - workflowSelect+` WHERE project_id = ? AND name = ? ORDER BY version DESC LIMIT 1`, + workflowSelect+` WHERE project_id = ? AND name = ? AND deprecated_at_ms IS NULL + ORDER BY version DESC LIMIT 1`, projectID, name)) } @@ -136,6 +143,12 @@ type WorkflowListOptions struct { ProjectID int Name string Limit int + // ExcludeDeprecated drops retired versions (`deprecated_at_ms` set) from + // both the rows and the pre-limit total. Zero value is false so every + // existing caller — binding readers that need the full lineage to judge + // staleness — keeps seeing every version; only `workflow list`'s default + // opts this in. + ExcludeDeprecated bool } // ListWorkflows returns registered workflows, newest registration first, and @@ -153,6 +166,9 @@ func ListWorkflows(db *sql.DB, opts WorkflowListOptions) ([]*model.Workflow, int clauses = append(clauses, `name = ?`) args = append(args, opts.Name) } + if opts.ExcludeDeprecated { + clauses = append(clauses, `deprecated_at_ms IS NULL`) + } where := `` if len(clauses) > 0 { where = ` WHERE ` + strings.Join(clauses, ` AND `) @@ -183,6 +199,64 @@ func ListWorkflows(db *sql.DB, opts WorkflowListOptions) ([]*model.Workflow, int return workflows, total, nil } +// WorkflowVersion is one registered version's IDENTITY, without its bytes +// (DKT-594). +// +// It exists because the staleness question — "how many versions has this name +// advanced since the run pinned it" — is answered by the version column alone, +// and the rows that answer it carry a `body` and a `parsed` each. A corpus with +// 41 commits in four days has names registered a dozen deep; loading every one +// of their bodies to subtract two integers would make a read verb's cost scale +// with the size of the definitions it is not reading. +type WorkflowVersion struct { + Name string + Version int + // Binds is false for a version RETIRED from binding (`deprecated_at_ms` set). + // + // The distinction is the same one bindableDefinitions makes: retirement is a + // binding-time filter, so the version a fresh run would pin is the highest + // one that still binds, and a staleness count computed over retired rows + // would tell an operator to chase a version nothing can bind. + Binds bool +} + +// WorkflowVersionsFor returns every registered version of the NAMED workflows +// within one project, ordered by (name, version) so a caller's reduction is +// deterministic. +// +// An empty `names` returns no rows and makes no query: the caller has nothing +// to ask about, and an unfiltered scan of the whole registry is never what +// "these names" means. +func WorkflowVersionsFor(conn *sql.DB, projectID int, names []string) ([]WorkflowVersion, error) { + if len(names) == 0 { + return nil, nil + } + placeholders := make([]string, len(names)) + args := []any{projectOrDefault(projectID)} + for i, n := range names { + placeholders[i] = "?" + args = append(args, n) + } + rows, err := conn.Query( + `SELECT name, version, deprecated_at_ms FROM workflows + WHERE project_id = ? AND name IN (`+strings.Join(placeholders, ", ")+`) + ORDER BY name ASC, version ASC`, args...) + if err != nil { + return nil, fmt.Errorf("listing registered workflow versions: %w", err) + } + return scanRows(rows, "workflow versions", func(r *sql.Rows) (WorkflowVersion, error) { + var ( + v WorkflowVersion + deprecatedAt sql.NullInt64 + ) + if err := r.Scan(&v.Name, &v.Version, &deprecatedAt); err != nil { + return WorkflowVersion{}, fmt.Errorf("reading a workflow version: %w", err) + } + v.Binds = !deprecatedAt.Valid || deprecatedAt.Int64 == 0 + return v, nil + }) +} + // workflowSelect names the columns in a fixed order, so the two scan helpers // cannot drift apart. const workflowSelect = ` diff --git a/internal/db/workflows_test.go b/internal/db/workflows_test.go index da849597..163bdaf1 100644 --- a/internal/db/workflows_test.go +++ b/internal/db/workflows_test.go @@ -191,6 +191,39 @@ func TestGetWorkflowSelectsHighestVersion(t *testing.T) { } } +// TestGetWorkflowSkipsDeprecatedVersions (DKT-616): the unversioned lookup is +// what `workflow show NAME` displays, and it must name the version a new run +// would bind — so a deprecated top version is skipped, an explicit @version +// still reaches it, and a name with nothing left to bind is not found. +func TestGetWorkflowSkipsDeprecatedVersions(t *testing.T) { + db := mustMigrated(t) + + for _, v := range []int{1, 2} { + _, _, err := InsertWorkflow(db, testWorkflow("w", v, "body"+string(rune('a'+v))), int64(v)) + testsupport.Must(t, err, "registering v%d: %v", v, err) + } + _, err := DeprecateWorkflow(db, 1, "w", 2, 1000) + testsupport.Must(t, err, "deprecating w@2: %v", err) + + got, err := GetWorkflow(db, 1, "w", 0) + testsupport.Must(t, err, "GetWorkflow: %v", err) + if got.Version != 1 { + t.Errorf("unversioned lookup returned v%d with v2 deprecated, want v1", got.Version) + } + + exact, err := GetWorkflow(db, 1, "w", 2) + testsupport.Must(t, err, "GetWorkflow(v2): %v", err) + if !exact.Deprecated() { + t.Errorf("explicit lookup of the deprecated v2 returned a row not marked deprecated: %+v", exact) + } + + _, err = DeprecateWorkflow(db, 1, "w", 1, 1000) + testsupport.Must(t, err, "deprecating w@1: %v", err) + if _, err := GetWorkflow(db, 1, "w", 0); !errors.Is(err, ErrWorkflowNotFound) { + t.Errorf("unversioned lookup with every version deprecated returned %v, want ErrWorkflowNotFound", err) + } +} + func TestGetWorkflowNotFound(t *testing.T) { db := mustMigrated(t) @@ -267,6 +300,36 @@ func TestListWorkflowsOrdersDeterministically(t *testing.T) { } } +// TestListWorkflowsExcludeDeprecatedDropsRetiredVersionsFromRowsAndTotal: +// ExcludeDeprecated is a WHERE clause, not a post-filter, so the total a +// caller sees is already the visible population's — the same guarantee +// --orphans gives its own filtered total. +func TestListWorkflowsExcludeDeprecatedDropsRetiredVersionsFromRowsAndTotal(t *testing.T) { + db := mustMigrated(t) + + _, _, err := InsertWorkflow(db, testWorkflow("w", 1, "a"), 1) + testsupport.Must(t, err, "w@1: %v", err) + _, _, err = InsertWorkflow(db, testWorkflow("w", 2, "b"), 2) + testsupport.Must(t, err, "w@2: %v", err) + _, err = DeprecateWorkflow(db, 1, "w", 1, 1000) + testsupport.Must(t, err, "deprecating w@1: %v", err) + + items, total, err := ListWorkflows(db, WorkflowListOptions{ExcludeDeprecated: true}) + testsupport.Must(t, err, "ListWorkflows: %v", err) + if total != 1 || len(items) != 1 || items[0].Ref() != "w@2" { + t.Errorf("ExcludeDeprecated returned %d/%d items %+v, want only w@2", + len(items), total, items) + } + + // The zero value keeps every existing caller's behavior: both versions. + items, total, err = ListWorkflows(db, WorkflowListOptions{}) + testsupport.Must(t, err, "ListWorkflows: %v", err) + if total != 2 || len(items) != 2 { + t.Errorf("the zero value filtered rows it was never asked to: %d/%d %+v", + len(items), total, items) + } +} + // TestWorkflowsTableIsDormantUntilRegistered is the phase-1 dormancy claim at // the storage layer. func TestWorkflowsTableIsDormantUntilRegistered(t *testing.T) { diff --git a/internal/engine/action.go b/internal/engine/action.go index 747c7bf0..36e91ae7 100644 --- a/internal/engine/action.go +++ b/internal/engine/action.go @@ -24,7 +24,7 @@ type ActionSpec struct { Name string // Params is the opaque KV bag, verbatim. CORE NEVER READS A KEY INSIDE IT // for a non-builtin action (§6.2) — a trusted command's params are its - // author's business. The builtin reads exactly the four keys §2 names for + // author's business. The builtin reads exactly the five keys §2 names for // it, and V28 refuses any other. Params map[string]any // Output is `params.output`: the artifact kind this step produces (§4.3.1). diff --git a/internal/engine/action_exec.go b/internal/engine/action_exec.go index a8a50d22..7a68fb94 100644 --- a/internal/engine/action_exec.go +++ b/internal/engine/action_exec.go @@ -127,13 +127,32 @@ func runAggregate(a ActionSpec, sc StepContext) ActionResult { return fail("step %s: %v", sc.Instance, err) } + // `route_at`'s below-floor clusters go to the RECORD (DKT-593): the + // builtin's own `action_results` row, where every attempt's trace already + // lives. They are fully reduced — value, members, demotion trail — and + // they are deliberately NOT in the artifact payload, so the threshold and + // every downstream `inputs` reader see only the emitted set. History is + // attributed; the loop is not fed. + recorded := "" + if len(outcome.Recorded) > 0 { + encoded, err := json.Marshal(outcome.Recorded) + if err != nil { + return fail("step %s: encoding the below-floor clusters: %v", + sc.Instance, err) + } + recorded = string(encoded) + } + return ActionResult{ - Kind: params.Output, - Body: aggregateBody(sc.Instance, len(outcome.Payload), len(outcome.Held)), + Kind: params.Output, + Body: aggregateBody(sc.Instance, + len(outcome.Payload)+len(outcome.Recorded), + len(outcome.Held), len(outcome.Recorded)), Payload: string(payload), Held: outcome.Held, Results: []ActionResultRow{{ Action: a.Name, Verdict: db.ActionVerdictPass, Builtin: true, + Output: recorded, }}, } } diff --git a/internal/engine/action_test.go b/internal/engine/action_test.go index 87862f0e..814f8c66 100644 --- a/internal/engine/action_test.go +++ b/internal/engine/action_test.go @@ -184,6 +184,96 @@ func TestBuiltinSpawnsNoProcess(t *testing.T) { } } +// TestBuiltinRouteAtSplitsTheOutputFromTheRecord is DKT-593 at the runner +// seam: the ARTIFACT payload — the wire the threshold evaluates and every +// downstream `inputs` reader consumes — carries only the clusters at or above +// the floor, while the below-floor clusters land, fully reduced, in the +// builtin's own `action_results` row. Routed, not erased: the record is +// attributed, and the loop is not fed. +func TestBuiltinRouteAtSplitsTheOutputFromTheRecord(t *testing.T) { + runner := &ExecActionRunner{ + RepoRoot: t.TempDir(), + LoadStore: func() (*trust.Store, error) { + t.Error("the builtin consulted the trust store (B1)") + return &trust.Store{}, nil + }, + } + + result, err := runner.Run(context.Background(), ActionSpec{ + Name: workflow.ActionAggregate, Output: "findings", + Params: map[string]any{ + "field": "severity", "method": "median", "output": "findings", + "route_at": "high", + }, + Inputs: []map[string]any{ + {"severity": "low", "id": "A"}, + {"severity": "blocker", "id": "B"}, + }, + Order: severityOrder, + }, StepContext{Instance: "reconcile@0", RunID: 1, IssueID: 1}) + testsupport.Must(t, err, "Run: %v", err) + if result.Failed { + t.Fatalf("the builtin failed: %s", result.Reason) + } + + var emitted []map[string]any + testsupport.Must(t, json.Unmarshal([]byte(result.Payload), &emitted), + "decoding the payload: %v", err) + if len(emitted) != 1 || emitted[0]["id"] != "B" { + t.Errorf("the artifact payload is %s, want only cluster B — the "+ + "below-floor cluster must not reach the loop output", result.Payload) + } + + if len(result.Results) != 1 { + t.Fatalf("results = %+v, want the builtin's one row", result.Results) + } + var recorded []map[string]any + testsupport.Must(t, json.Unmarshal([]byte(result.Results[0].Output), &recorded), + "decoding the recorded clusters: %v", err) + if len(recorded) != 1 || recorded[0]["id"] != "A" || recorded[0]["severity"] != "low" { + t.Errorf("the action row records %s, want the reduced below-floor "+ + "cluster A", result.Results[0].Output) + } + + if !strings.Contains(result.Body, "2 cluster(s) reduced") || + !strings.Contains(result.Body, "1 below the `route_at` floor") { + t.Errorf("the body does not account for both halves: %q", result.Body) + } +} + +// TestBuiltinWithoutRouteAtRecordsNothingInItsRow is the absent case at the +// same seam: no floor, an empty audit Output, and a body with no floor +// sentence — byte-for-byte the pre-route_at builtin. +func TestBuiltinWithoutRouteAtRecordsNothingInItsRow(t *testing.T) { + runner := &ExecActionRunner{ + RepoRoot: t.TempDir(), + LoadStore: func() (*trust.Store, error) { return &trust.Store{}, nil }, + } + result, err := runner.Run(context.Background(), ActionSpec{ + Name: workflow.ActionAggregate, Output: "findings", + Params: map[string]any{ + "field": "severity", "method": "median", "output": "findings", + }, + Inputs: []map[string]any{{"severity": "low", "id": "A"}}, + Order: severityOrder, + }, StepContext{Instance: "reconcile@0", RunID: 1, IssueID: 1}) + testsupport.Must(t, err, "Run: %v", err) + if result.Failed { + t.Fatalf("the builtin failed: %s", result.Reason) + } + if result.Results[0].Output != "" { + t.Errorf("the action row carries %q; with no route_at there is nothing "+ + "to record", result.Results[0].Output) + } + if strings.Contains(result.Body, "route_at") { + t.Errorf("the body mentions a floor nobody declared: %q", result.Body) + } + if result.Payload != `[{"held":false,"id":"A","members":["low"],`+ + `"operator_resolved":false,"severity":"low"}]` { + t.Errorf("the absent case's payload changed shape: %s", result.Payload) + } +} + // childProcessCount counts this process's children, so a spawn is observable. func childProcessCount(t *testing.T) int { t.Helper() diff --git a/internal/engine/activate.go b/internal/engine/activate.go index b6531077..22ccab59 100644 --- a/internal/engine/activate.go +++ b/internal/engine/activate.go @@ -97,6 +97,20 @@ type ActivateResult struct { // the row in JSON mode. Returning them rather than printing them here // keeps the engine free of an output dependency it has no other use for. ContextWarnings []ContextWarning + // SourceWarnings names every BOUND workflow whose registered source file + // could not be verified against `source_sha256` — and that activation did + // not refuse over (DKT-590). Drift on a binding this activation MADE is a + // refusal, so it appears here only in the three cases refusing would be + // wrong: an unreadable source (the registered bytes still reproduce; only + // their provenance is gone), drift under a binding inherited from an + // earlier activation (RA2 keeps a mid-run edit a non-event), and drift + // under `registration.auto = false`, where a registry lagging the corpus is + // the operator's own standing decision. + // + // It travels here for ContextWarnings' reason and takes the same stance: + // the engine holds no output dependency, and the verb picks the channel — + // stderr in human mode, an array in JSON. + SourceWarnings []SourceWarning // ScopeWarnings names every issue that declared no scope at all while // binding a workflow whose steps occupy the tree (lintUnscopedHolders). // @@ -113,9 +127,11 @@ type ActivateResult struct { // VISIBLE IN THE OUTPUT rather than merely true underneath it. Registered []Registration // PinsFromConfig counts the files under `.docket/config/` that were pinned - // rather than registered (F4). They are counted, not listed: a fragment tree - // can hold hundreds of files and none of them is a decision an operator - // needs to read before approving. + // rather than registered (F4) — since DKT-581, only the PACKET CLOSURE the + // bound workflows reach (packet entries, their `packet_includes`, and + // policy.toml), not every file the scan walked. They are counted, not + // listed: a closure can hold dozens of files and none of them is a + // decision an operator needs to read before approving. PinsFromConfig int // Fences is the §7.7 trust report: every harvested fenced command and @@ -499,19 +515,24 @@ func activateTx( // hit F9's collision refusal on a run that was working fine, and the fix // would be to revert their edit. Inheriting makes the edit a non-event — // exactly as a re-registered workflow is already a non-event (RA2). - if !reactivation && scan != nil { - // registration.auto (default true) gates ONLY this half. An operator - // who turns it off is declining silent version adoption, not the - // corpus itself — the pinned half below has no version to adopt, so it - // keeps running regardless: a project with registration off still - // needs the corpus's contracts/fragments/policy.toml to render a - // step's `packet`, and the only alternative would be hand-supplying - // every one of them via `--pin`. - autoRegister, err := db.AutoRegisterEnabledTx(tx, run.ProjectID) - if err != nil { - return nil, err - } + // registration.auto (default true) gates ONLY the registering half below. An + // operator who turns it off is declining silent version adoption, not the + // corpus itself — the pinned half has no version to adopt, so it keeps + // running regardless: a project with registration off still needs the + // corpus's contracts/fragments/policy.toml to render a step's `packet`, and + // the only alternative would be hand-supplying every one of them via + // `--pin`. + // + // It is read HERE, ahead of the scan's own branch, because stage 1's source + // check needs the same answer: with adoption declined, a registry that lags + // the corpus is the operator's own standing decision rather than an + // unexplained divergence, and the two dispositions differ (DKT-590). + autoRegister, err := db.AutoRegisterEnabledTx(tx, run.ProjectID) + if err != nil { + return nil, err + } + if !reactivation && scan != nil { if autoRegister { registered, err := registerScanTx(tx, run.ProjectID, scan, opts.NowMS) if err != nil { @@ -533,11 +554,10 @@ func activateTx( } } - // F4: the pinned half. These join the pin set below rather than being - // written here, so auto-pinned and `--pin` files travel one code path - // and RA2's inheritance covers both. - filePins = append(filePins, scan.pins...) - result.PinsFromConfig = len(scan.pins) + // F4's pinned half moved to stage 3 (DKT-581): the scan's pins join + // the pin set FILTERED to the packet closure the bound workflows + // reach, and the bindings that define that closure do not exist until + // stage 1 has run. } runIssues, err := db.ListRunIssuesTx(tx, runID) @@ -560,7 +580,30 @@ func activateTx( // issue already bound keeps its binding — re-binding would let a workflow // registered mid-run capture an issue the operator already approved into a // different pipeline. + // DKT-609's half of binding: the scan this activation already performed, + // read back as "which registered NAMES still have a definition on disk". + // + // It is built from the SAME scan registration uses, so the two cannot + // disagree about what a root holds, and it is LAZY — nothing reads a file + // unless a refusal below actually names candidates. A re-activation does + // not re-scan config (F15), but the scan itself was computed regardless, + // so an inherited-binding run annotates from exactly the same facts. + // + // Laziness is why it sits here rather than beside the scan outside the + // transaction: eager would re-read and re-parse every workflow in the + // corpus on EVERY activation to answer a question almost none of them ask. + // The reads it does perform happen on a refusal path, in a transaction + // that is about to roll back and write nothing — the same latitude §6 + // gives activation's other reads, spent only where an operator is already + // receiving an error. + origins := newWorkflowOriginIndex(scan) + bindings := make(map[int]*boundDefinition, len(runIssues)) + // inherited marks the bindings a PRIOR activation made, which this one is + // only re-reading. DKT-590's source check disposes of the two differently — + // RA2 says an edit must not reach a run already under way — so the + // distinction the loop already knows is recorded rather than re-derived. + inherited := make(map[int]bool, len(runIssues)) for _, ri := range runIssues { issue := issues[ri.IssueID] if ri.WorkflowID != nil { @@ -571,15 +614,31 @@ func activateTx( model.FormatID(ri.IssueID), *ri.WorkflowID) } bindings[ri.IssueID] = bound + inherited[ri.IssueID] = true continue } - bound, err := bindIssue(issue, definitions) + bound, err := bindIssue(issue, definitions, origins) if err != nil { return nil, err } bindings[ri.IssueID] = bound } + // Binding's INTEGRITY half (DKT-590): does each bound workflow's own + // recorded `source_path` still hold the bytes it was registered with? + // + // It runs HERE — after binding, before anything is pinned or expanded — + // because binding is what selects the definitions whose provenance matters, + // and because a refusal is worth more before the transaction has computed a + // topology than after. A dry run reaches it on this same path and refuses + // identically: `--dry-run` is the real activation rolled back, so what it + // reports is what a real one would do. + sourceWarnings, err := checkBoundSources(runIssues, bindings, inherited, autoRegister) + if err != nil { + return nil, err + } + result.SourceWarnings = sourceWarnings + // Binding's lint half: an issue that holds the tree while declaring nothing // about what it touches. // @@ -641,6 +700,33 @@ func activateTx( pinned[p.Kind+"\x00"+p.Ref] = struct{}{} } + // DKT-581: the config scan's pins join the set filtered to the PACKET + // CLOSURE the bound workflows reach — each step's `packet` entries with + // `{executor}` substituted the way expansion substitutes it, the files + // those entries' `packet_includes` declare (transitively), and + // `policy.toml` — rather than every file the scan walked. A corpus edit + // to a contract no bound step references is then a non-event for this + // run's `verify-pins`, which is the whole remedy: 7 of 18 terminal runs + // in the measured week were abandoned over drift in files they never + // read. `--pin` files are the operator's explicit additions and are + // never filtered. + // + // On a RE-ACTIVATION the closure is recomputed only to cover an issue + // added since activation that bound a workflow whose files the original + // set never pinned (RA3) — refs an earlier activation recorded are + // dropped first, so RA2's inheritance is untouched: an edited pinned + // file stays at its original hash, and an inherited ref keeps resolving + // as present-with-unknown-size in the declared-packet index below. + if scan != nil { + closure := packetClosurePins(scan, runIssues, bindings) + if reactivation { + closure = withoutAlreadyPinned(closure, existingPins, scan.roots) + } else { + result.PinsFromConfig = len(closure) + } + filePins = append(filePins, closure...) + } + pins := make([]db.Pin, 0, len(bindings)+len(filePins)) for _, ri := range runIssues { bound := bindings[ri.IssueID] @@ -761,7 +847,18 @@ func activateTx( // re-activation would defeat the immunity for every issue already // under way. if ri.BodySnapshot == "" && ri.IssueSnapshot == "" { - snapshot, err := issueSnapshot(tx, issue) + // The cross-issue bindings (DKT-547): every `issue.linked. + // .` input the bound workflow declares is resolved + // NOW — linked issue(s) by relation, then each one's latest + // recorded artifact of the kind — and pinned into the snapshot by + // artifact id. A missing relation or artifact refuses the whole + // activation, loudly, inside the fat transaction: the binding is + // enforced here or it is an issue-body citation nothing checks. + linked, err := resolveLinkedInputs(tx, issue, bound.definition) + if err != nil { + return nil, err + } + snapshot, err := issueSnapshot(tx, issue, linked) if err != nil { return nil, err } @@ -986,9 +1083,12 @@ func definitionByID(definitions []*boundDefinition, id int) *boundDefinition { // bumped name matched, and exactly-one-match refused (the M2a toy run). // // The resolution deliberately MIRRORS `workflow show NAME` without `@version` -// (db.GetWorkflow's `ORDER BY version DESC LIMIT 1`). Binding and show -// disagreeing about what "the" workflow of a name is was the defect; one -// helper's worth of agreement is asserted directly by +// (db.GetWorkflow's `deprecated_at_ms IS NULL ... ORDER BY version DESC LIMIT +// 1`). Binding and show disagreeing about what "the" workflow of a name is +// was the defect — twice: first two versions of one name both binding while +// show picked one, then (DKT-616) show resolving a deprecated top version that +// binding had already retired. One helper's worth of agreement, including the +// deprecated-top-version case, is asserted directly by // TestBindingAgreesWithWorkflowShowResolution. // // Exactly-one-match therefore applies across NAMES, which is what makes its @@ -1046,7 +1146,16 @@ func bindableDefinitions(definitions []*boundDefinition) []*boundDefinition { // The candidates NAMED are the bindable ones, not every registered row: an // error that listed superseded versions would send an operator to edit a // definition that could not have bound the issue anyway. -func bindIssue(issue *model.Issue, definitions []*boundDefinition) (*boundDefinition, error) { +// +// `origins` decorates that candidate set and NEVER CHANGES IT (DKT-609). An +// orphaned registration still binds — a registration is a row, not a file — +// so it is still a candidate and still causes the ambiguity; what the +// annotation adds is which of the named candidates has nothing behind it on +// disk. It may be nil, and every verdict then reads `unchecked`, which is the +// pre-DKT-609 message exactly. +func bindIssue( + issue *model.Issue, definitions []*boundDefinition, origins *WorkflowOriginIndex, +) (*boundDefinition, error) { subject := workflow.Subject{Kind: string(issue.Kind), Labels: issue.Labels} candidates := bindableDefinitions(definitions) @@ -1081,27 +1190,53 @@ func bindIssue(issue *model.Issue, definitions []*boundDefinition) (*boundDefini } return nil, validationErr( "issue %s (kind %s, labels [%s]) matches no registered workflow; "+ - "candidates considered: %s", + "candidates considered: %s%s", model.FormatID(issue.ID), issue.Kind, strings.Join(issue.Labels, " "), - refList(candidates)) + refList(candidates, origins), orphanHint(candidates, origins)) default: return nil, validationErr( "issue %s matches %d workflows, and exactly one must match; "+ - "candidates: %s", - model.FormatID(issue.ID), len(matched), refList(matched)) + "candidates: %s%s", + model.FormatID(issue.ID), len(matched), refList(matched, origins), + orphanHint(matched, origins)) } } // refList renders candidate workflows as `name@version`, in the stable order -// loadDefinitions established. -func refList(definitions []*boundDefinition) string { +// loadDefinitions established, ANNOTATING each one whose name no longer has a +// definition in any instance-config root (DKT-609). +// +// The annotation rides HERE rather than in a parallel renderer so both binding +// refusals carry it from one place: a rename strands a name, and the refusal +// that names the stranded registration alongside its replacement is the exact +// moment the distinction is worth money. +func refList(definitions []*boundDefinition, origins *WorkflowOriginIndex) string { refs := make([]string, 0, len(definitions)) for _, d := range definitions { - refs = append(refs, d.workflow.Ref()) + ref := d.workflow.Ref() + if origins.Orphaned(d.workflow.Name) { + ref += orphanAnnotation + } + refs = append(refs, ref) } return strings.Join(refs, ", ") } +// orphanHint returns the remedy sentence when at least one named candidate is +// orphaned, and the empty string otherwise. +// +// It is conditional because a refusal between two live definitions is an +// authoring problem — narrow a `[match]` — and appending a paragraph about +// deprecation to that one would be advice for a state the repo is not in. +func orphanHint(definitions []*boundDefinition, origins *WorkflowOriginIndex) string { + for _, d := range definitions { + if origins.Orphaned(d.workflow.Name) { + return orphanRefusalHint + } + } + return "" +} + // lintUnscopedHolders reports every issue that declares NO SCOPE while binding a // workflow that occupies the tree. // @@ -1362,7 +1497,32 @@ func issuePredecessorsSatisfied( // requires that answer frozen. Reading live for the bundle would break mid-run // edit immunity; freezing the scheduler would ignore a correction that exists // precisely to prevent a collision. -func issueSnapshot(tx *sql.Tx, issue *model.Issue) (string, error) { +// `linked` is the activation-resolved cross-issue pin set (DKT-547): declared +// `issue.linked..` suffix -> pinned artifact ids. It rides the +// snapshot because it IS a snapshot — the answer to a question about live +// state, frozen at activation — and because the snapshot's lifecycle is +// exactly the pin's: written once at binding, never rewritten at +// re-activation, read only by context assembly. `omitempty` keeps every +// snapshot without the form byte-identical to what it always was, which is +// what the golden bundles require. +// issueSnapshotFields is the snapshot blob's shape, declared ONCE (DKT-869). +// +// Field ORDER is the canonical JSON's key order — encoding/json emits struct +// fields by declaration — so this declaration is the format, not a convenience +// view of it. It is a named type rather than an anonymous struct literal +// because the scope refresh (scope_refresh.go) re-encodes an existing snapshot +// through the SAME type: two independent literals would drift the moment +// §11.4's issue shape grew a field, and a refresh that dropped a key +// activation had written would silently truncate a live run's snapshot. +type issueSnapshotFields struct { + Title string `json:"title"` + Kind string `json:"kind"` + Labels []string `json:"labels"` + Scope []string `json:"scope"` + Linked map[string][]int `json:"linked,omitempty"` +} + +func issueSnapshot(tx *sql.Tx, issue *model.Issue, linked map[string][]int) (string, error) { scopeJSON, err := db.IssueScopeGlobsTx(tx, issue.ID) if err != nil { return "", fmt.Errorf("reading scope for %s: %w", model.FormatID(issue.ID), err) @@ -1382,16 +1542,12 @@ func issueSnapshot(tx *sql.Tx, issue *model.Issue) (string, error) { labels = []string{} } - snapshot := struct { - Title string `json:"title"` - Kind string `json:"kind"` - Labels []string `json:"labels"` - Scope []string `json:"scope"` - }{ + snapshot := issueSnapshotFields{ Title: issue.Title, Kind: string(issue.Kind), Labels: labels, Scope: scope, + Linked: linked, } out, err := json.Marshal(snapshot) diff --git a/internal/engine/activate_test.go b/internal/engine/activate_test.go index 023e200e..ee48b5ce 100644 --- a/internal/engine/activate_test.go +++ b/internal/engine/activate_test.go @@ -2247,6 +2247,55 @@ func TestBindingAgreesWithWorkflowShowResolution(t *testing.T) { t.Errorf("superseded docs-review@%d is no longer resolvable: %v", v, err) } } + + // DKT-616: deprecating the TOP version is the one case where the two + // resolutions used to diverge — binding retired @3 and fell back to @2, + // while show still displayed @3. Both must now land on @2, and the + // deprecated @3 must remain reachable by explicit @version. + _, err = db.DeprecateWorkflow(conn, 1, "docs-review", 3, nowMS) + testsupport.Must(t, err, "deprecating docs-review@3: %v", err) + + shown, err = db.GetWorkflow(conn, 1, "docs-review", 0) + testsupport.Must(t, err, "resolving docs-review after deprecating @3: %v", err) + if shown.Version != 2 { + t.Errorf("`workflow show docs-review` resolves to @%d after deprecating @3, want @2", shown.Version) + } + + definitions, err = loadDefinitions(conn, 1) + testsupport.Must(t, err, "reloading definitions: %v", err) + bound = nil + for _, d := range bindableDefinitions(definitions) { + if d.workflow.Name == "docs-review" { + bound = d + } + } + if bound == nil { + t.Fatal("binding kept no candidate for docs-review after deprecating @3") + } + if bound.workflow.Ref() != shown.Ref() { + t.Errorf("with @3 deprecated, binding resolves docs-review to %s but `workflow show` resolves it to %s; "+ + "§11.1 requires they agree", bound.workflow.Ref(), shown.Ref()) + } + if _, err := db.GetWorkflow(conn, 1, "docs-review", 3); err != nil { + t.Errorf("deprecated docs-review@3 is no longer resolvable by explicit version: %v", err) + } + + // Retiring EVERY version removes the name from binding; show agrees by + // reporting not-found rather than surfacing a retired row. + for _, v := range []int{1, 2} { + _, err := db.DeprecateWorkflow(conn, 1, "docs-review", v, nowMS) + testsupport.Must(t, err, "deprecating docs-review@%d: %v", v, err) + } + if _, err := db.GetWorkflow(conn, 1, "docs-review", 0); !errors.Is(err, db.ErrWorkflowNotFound) { + t.Errorf("with every version deprecated, `workflow show docs-review` returned %v, want ErrWorkflowNotFound", err) + } + definitions, err = loadDefinitions(conn, 1) + testsupport.Must(t, err, "reloading definitions: %v", err) + for _, d := range bindableDefinitions(definitions) { + if d.workflow.Name == "docs-review" { + t.Errorf("with every version deprecated, binding still kept %s", d.workflow.Ref()) + } + } } // --------------------------------------------------------------------------- @@ -2395,7 +2444,7 @@ func TestRenamePlusBumpRefusesActivation(t *testing.T) { t.Fatal("activation succeeded on an issue two workflow NAMES match " + "after a rename-plus-bump; today's binding refuses this") } - assertRenamePlusBumpWedge(t, err, issue) + assertRenamePlusBumpWedge(t, err, issue, wedgeCandidatesOrphaned) // Scoped to this run: the earlier activation that registered gone@1 wrote // its own steps, pins and binding on this same connection. @@ -2412,14 +2461,22 @@ func TestRenamePlusBumpRefusesActivation(t *testing.T) { } // assertRenamePlusBumpWedge asserts that err is the exact rename-plus-bump -// wedge refusal: CodeValidation, naming the issue and the candidates -// "gone@1, gone-renamed@2" (name-ascending, per loadDefinitions' sort at +// wedge refusal: CodeValidation, naming the issue and rendering its candidates +// as `wantCandidates` (name-ascending, per loadDefinitions' sort at // activate.go:610-615). Shared by TestRenamePlusBumpRefusesActivation and // TestTombstoneRetiresAWedgedName's setup, so the tombstone test's // precondition is pinned as precisely as the refusal it retires — a zero-match // refusal or an unrelated error would satisfy a bare "err != nil" check just // as readily, and would leave the tombstone test's premise silently wrong. -func assertRenamePlusBumpWedge(t *testing.T, err error, issue int) { +// +// The candidate rendering is a PARAMETER since DKT-609, because the two call +// sites arrange the same refusal over different repository states and the +// refusal now says so: one deletes gone.toml from a real config root (so +// "gone" is an orphaned registration and is annotated), the other registers +// both names directly with no config root to scan (so nothing was checked and +// nothing is annotated). Hard-coding one string would have made the other call +// site pass on an assertion it did not mean. +func assertRenamePlusBumpWedge(t *testing.T, err error, issue int, wantCandidates string) { t.Helper() if code, _ := CodeOf(err); code != CodeValidation { t.Errorf("error code = %q, want %q", code, CodeValidation) @@ -2429,15 +2486,22 @@ func assertRenamePlusBumpWedge(t *testing.T, err error, issue int) { if !strings.Contains(msg, model.FormatID(issue)) { t.Errorf("error does not name the issue: %s", msg) } - // Name-ascending, matching loadDefinitions' sort: "gone" before - // "gone-renamed". - const wantCandidates = "gone@1, gone-renamed@2" if !strings.Contains(msg, wantCandidates) { t.Errorf("error does not name the candidates as %q: %s", wantCandidates, msg) } assertMultiMatchBranch(t, err) } +// wedgeCandidatesUnchecked is the candidate rendering when NO instance-config +// root exists to scan: bare refs, name-ascending, exactly as before DKT-609. +// "Nothing was checked" must never render as "nothing is orphaned". +const wedgeCandidatesUnchecked = "gone@1, gone-renamed@2" + +// wedgeCandidatesOrphaned is the same refusal after a real rename in a real +// config root: the stranded registration is annotated and its live replacement +// is not, which is the whole of DKT-609's second acceptance criterion. +const wedgeCandidatesOrphaned = "gone@1" + orphanAnnotation + ", gone-renamed@2" + // multiMatchDiscriminator is the ONLY text distinguishing bindIssue's two // refusal branches (activate.go:762-767 versus :756-761). // @@ -2509,7 +2573,7 @@ func TestTombstoneRetiresAWedgedName(t *testing.T) { t.Fatal("setup: expected the rename-plus-bump wedge to refuse before " + "the tombstone is registered") } - assertRenamePlusBumpWedge(t, err, wedged) + assertRenamePlusBumpWedge(t, err, wedged, wedgeCandidatesUnchecked) // The tombstone: re-register "gone" at a HIGHER version with a [match] // that admits NOTHING, by construction rather than by picking an issue diff --git a/internal/engine/aggregate.go b/internal/engine/aggregate.go index 1b045a5f..b0008fce 100644 --- a/internal/engine/aggregate.go +++ b/internal/engine/aggregate.go @@ -14,9 +14,9 @@ import ( // engine-spec §2, verbatim, is the specification: // // One is builtin and generic: `action = "aggregate"` with -// `params = { field, method = median|max|min, hold_spread, output }` computes -// over any ordered-enum payload field — median, spread-hold, and a recorded -// demotion trail work for severities, priorities, or tiers alike. +// `params = { field, method = median|max|min, hold_spread, output, route_at }` +// computes over any ordered-enum payload field — median, spread-hold, and a +// recorded demotion trail work for severities, priorities, or tiers alike. // // EVERY VALUE HERE IS COMPARED BY ITS POSITION IN THE USER'S DECLARED ORDER AND // BY NOTHING ELSE. `field` is a key to look up; the values under it are opaque @@ -29,7 +29,7 @@ import ( // what makes §9 item 5's determinism a property of the function rather than of // the environment it ran in. -// Aggregate params, as §7.1's table names them. Core reads exactly these four +// Aggregate params, as §7.1's table names them. Core reads exactly these five // keys of the opaque bag and V28 refuses any other, so "the engine ignored my // param" is a register-time sentence rather than a run-time mystery. const ( @@ -37,6 +37,13 @@ const ( ParamMethod = "method" ParamHoldSpread = "hold_spread" ParamOutput = "output" + // ParamRouteAt is the routing floor (DKT-593): a value of the field's + // declared order. Clusters whose reduced value sits at or above its + // position are emitted to the step's output payload — the wire the + // threshold and the loop read — and the rest go to the record. Like every + // other value here it is an OPAQUE TOKEN compared only by position: core + // does not know it is a severity, only where the author put it. + ParamRouteAt = "route_at" ) // The three reductions of §7.3. There are exactly three because §2 names @@ -89,19 +96,36 @@ type AggregateParams struct { Method string HoldSpread int Output string + // RouteAt is the routing floor, or "" when the step declares none — the + // absent case, in which Aggregate emits every cluster exactly as it always + // has. + RouteAt string } // AggregateOutcome is one aggregation's result. type AggregateOutcome struct { - // Payload is the output payload, one element per input element (§7.6). + // Payload is the output payload, one element per emitted cluster (§7.6). + // Without `route_at` that is one element per input element; with it, only + // the clusters whose reduced value's position reached the floor. Payload []map[string]any - // Held indexes the elements whose spread tripped `hold_spread`. It is a - // list rather than a count because §7.7's materialization has to be able to - // say WHICH clusters are open, and a count cannot. + // Held indexes the PAYLOAD elements whose spread tripped `hold_spread`. It + // is a list rather than a count because §7.7's materialization has to be + // able to say WHICH clusters are open, and a count cannot. The indices are + // positions in Payload — the payload the artifact records and held + // resolution re-reads (H2a) — which coincide with input positions whenever + // `route_at` is absent. Held []int + // Recorded holds the clusters `route_at` routed below the floor (DKT-593): + // fully reduced — value, members, demotion trail — but destined for the + // record rather than the loop output. The caller writes them into the + // action's own audit row; they never enter the artifact payload, so the + // threshold and every downstream `inputs` reader see only the emitted set. + // Nil whenever `route_at` is absent, which is what keeps the absent case + // byte-identical to the pre-`route_at` builtin. + Recorded []map[string]any } -// ParseAggregateParams reads §7.1's four keys out of the opaque bag. +// ParseAggregateParams reads §7.1's five keys out of the opaque bag. // // It is the ONE place core reads inside `params`, and it reads exactly the keys // §2 names. The same rules run at register time (V28) so a typo'd @@ -154,19 +178,35 @@ func ParseAggregateParams(params map[string]any) (AggregateParams, error) { ParamHoldSpread, hold) } + // `route_at` is optional; present, it must be a non-empty string. WHETHER + // the value has a position is the declared order's question and is asked in + // Aggregate, where the resolver is — the same split V28/V28a make at + // register time. + routeAt, err := stringParam(params, ParamRouteAt) + if err != nil { + return out, err + } + if _, present := params[ParamRouteAt]; present && routeAt == "" { + return out, fmt.Errorf("`params.%s` must be a non-empty string naming "+ + "a value of `params.%s`'s declared order", ParamRouteAt, ParamField) + } + // V28's "no other keys". An unread key is a declaration the author believes // is doing something; saying so at register time is the whole discipline. for key := range params { switch key { - case ParamField, ParamMethod, ParamHoldSpread, ParamOutput: + case ParamField, ParamMethod, ParamHoldSpread, ParamOutput, ParamRouteAt: continue } return out, fmt.Errorf("`params.%s` is not a parameter of `aggregate`; "+ "it takes exactly %v", key, - []string{ParamField, ParamMethod, ParamHoldSpread, ParamOutput}) + []string{ParamField, ParamMethod, ParamHoldSpread, ParamOutput, ParamRouteAt}) } - out = AggregateParams{Field: field, Method: method, HoldSpread: hold, Output: output} + out = AggregateParams{ + Field: field, Method: method, HoldSpread: hold, Output: output, + RouteAt: routeAt, + } return out, nil } @@ -196,6 +236,24 @@ func Aggregate( params.Field) } + // `route_at` resolves to a POSITION, once, before any cluster is read. A + // value the declared order does not name is a step failure naming it — + // V28a makes this unreachable through `workflow register`, and it is + // reachable from a restored database, where it follows G4's discipline: no + // position, no guess. -1 means no floor; every cluster is emitted. + routeFloor := -1 + if params.RouteAt != "" { + pos, ok := order.Position(params.Field, params.RouteAt) + if !ok { + return nil, fmt.Errorf( + "`params.%s`: the value %q is not in the order this step's "+ + "pinned schema declares for %q, so it has no position; core "+ + "does not guess a routing floor for an unknown value", + ParamRouteAt, params.RouteAt, params.Field) + } + routeFloor = pos + } + out := &AggregateOutcome{Payload: make([]map[string]any, 0, len(payloads))} for i, element := range payloads { @@ -249,9 +307,32 @@ func Aggregate( result[KeyDemotedFrom] = members[top.at] } + // `route_at` (DKT-593): a cluster whose reduced value's position is + // below the floor goes to the RECORD, not the loop output. The routing + // is CORE's, over its own computed positions — the same comparison the + // threshold would make — which is what keeps a downstream fixer's "the + // reconciled set is the work" contract intact: the set it receives has + // already been routed, so it has nothing to filter. + // + // A HELD cluster is never routed below the floor. Its computed value is + // exactly the value the hold refuses to trust — the spread says the + // members disagree, and an operator resolving the hold may accept a + // different value entirely — so routing it out by that value would + // spend the decision the hold exists to ask. It is emitted, it gates + // the step as every held cluster does, and the floor's opinion waits + // for the operator's. + if routeFloor >= 0 && !held && taken.rank < routeFloor { + out.Recorded = append(out.Recorded, result) + continue + } + out.Payload = append(out.Payload, result) if held { - out.Held = append(out.Held, i) + // The index is the element's position in the EMITTED payload — the + // payload the artifact records and H2a's `-held@k#i` suffix + // addresses — which is `i` exactly whenever no cluster was routed + // below the floor. + out.Held = append(out.Held, len(out.Payload)-1) } } @@ -419,13 +500,18 @@ func stringParam(params map[string]any, key string) (string, error) { // AggregateBody renders the human half of the artifact an aggregate produces. // -// It names counts and nothing else: how many clusters were reduced and how many -// are open. A body that summarized the VALUES would be core narrating an -// instance's meaning back to its operator. -func aggregateBody(step string, clusters, held int) string { +// It names counts and nothing else: how many clusters were reduced, how many +// are open, and how many the routing floor sent to the record. A body that +// summarized the VALUES would be core narrating an instance's meaning back to +// its operator. +func aggregateBody(step string, clusters, held, recorded int) string { body := fmt.Sprintf("aggregate on %s: %d cluster(s) reduced", step, clusters) if held > 0 { body += fmt.Sprintf(", %d held for an operator decision", held) } + if recorded > 0 { + body += fmt.Sprintf( + ", %d below the `route_at` floor, recorded and not routed", recorded) + } return body } diff --git a/internal/engine/aggregate_test.go b/internal/engine/aggregate_test.go index c423c2ed..b134d9ad 100644 --- a/internal/engine/aggregate_test.go +++ b/internal/engine/aggregate_test.go @@ -435,7 +435,8 @@ func TestParseAggregateParams(t *testing.T) { t.Errorf("params = %+v", params) } - // `hold_spread` absent defaults to 0, which never holds. + // `hold_spread` absent defaults to 0, which never holds; `route_at` absent + // defaults to "", which never routes a cluster to the record. params, err = ParseAggregateParams(map[string]any{ "field": "severity", "method": "max", "output": "findings", }) @@ -443,6 +444,20 @@ func TestParseAggregateParams(t *testing.T) { if params.HoldSpread != 0 { t.Errorf("hold_spread defaulted to %d, want 0", params.HoldSpread) } + if params.RouteAt != "" { + t.Errorf("route_at defaulted to %q, want the empty no-floor case", params.RouteAt) + } + + // `route_at` present is carried through verbatim; whether the value has a + // position is Aggregate's question, where the resolver is. + params, err = ParseAggregateParams(map[string]any{ + "field": "severity", "method": "median", "output": "findings", + "route_at": "high", + }) + testsupport.Must(t, err, "ParseAggregateParams with route_at: %v", err) + if params.RouteAt != "high" { + t.Errorf("route_at = %q, want high", params.RouteAt) + } bad := []struct { name string @@ -459,6 +474,12 @@ func TestParseAggregateParams(t *testing.T) { {"a key aggregate does not take", map[string]any{"field": "s", "method": "median", "output": "k", "grouping": "cluster"}, "grouping"}, + {"route_at that is not a string", + map[string]any{"field": "s", "method": "median", "output": "k", + "route_at": 3}, "route_at"}, + {"route_at that is empty", + map[string]any{"field": "s", "method": "median", "output": "k", + "route_at": ""}, "route_at"}, } for _, tc := range bad { t.Run(tc.name, func(t *testing.T) { @@ -615,3 +636,162 @@ func TestAggregateConservativeEndIsPerField(t *testing.T) { "direction and must not inherit `severity`'s", got) } } + +// routeAtParams is medianParams with a routing floor declared. +func routeAtParams(method string, hold int, floor string) AggregateParams { + p := medianParams(method, hold) + p.RouteAt = floor + return p +} + +// TestAggregateRouteAtFloor is DKT-593's boundary: a cluster whose REDUCED +// value's position is >= the floor's is emitted to the loop output, and one +// below it goes to the record. The comparison is over the reduced value — a +// cluster whose members reach `high` but whose median is `low` is a `low` +// cluster to the floor, exactly as it is to the threshold. +func TestAggregateRouteAtFloor(t *testing.T) { + // Reduced values: low (below), medium (at), high (above), and a clustered + // element whose MEDIAN is low even though a member is high (below). + const raw = `[{"severity":"low","id":"A"},{"severity":"medium","id":"B"},` + + `{"severity":"high","id":"C"},{"severity":["info","low","high"],"id":"D"}]` + + out, err := aggregateOver(t, raw, + routeAtParams(MethodMedian, 0, "medium"), severityOrder) + testsupport.Must(t, err, "Aggregate: %v", err) + + emitted := make([]string, 0, len(out.Payload)) + for _, element := range out.Payload { + emitted = append(emitted, element["id"].(string)) + } + if len(emitted) != 2 || emitted[0] != "B" || emitted[1] != "C" { + t.Errorf("emitted %v, want [B C] — at and above the floor, in input order", + emitted) + } + + recorded := make([]string, 0, len(out.Recorded)) + for _, element := range out.Recorded { + recorded = append(recorded, element["id"].(string)) + } + if len(recorded) != 2 || recorded[0] != "A" || recorded[1] != "D" { + t.Errorf("recorded %v, want [A D] — below the floor, in input order", + recorded) + } + + // A recorded cluster is fully reduced: the value, the members, and the + // demotion trail all survive into the record — routing is not erasure. + d := out.Recorded[1] + if d["severity"] != "low" || d[KeyDemotedFrom] != "high" { + t.Errorf("the recorded cluster lost its reduction: severity = %v, "+ + "demoted_from = %v", d["severity"], d[KeyDemotedFrom]) + } +} + +// TestAggregateRouteAtUnknownValueRefuses is G4's discipline applied to the +// floor: a `route_at` value the declared order does not name has no position, +// and core refuses naming it rather than guessing an end. +// +// V28a makes this unreachable through `workflow register`; it is reachable +// from a database restored from elsewhere. +func TestAggregateRouteAtUnknownValueRefuses(t *testing.T) { + _, err := aggregateOver(t, `[{"severity":"low"}]`, + routeAtParams(MethodMedian, 0, "urgent"), severityOrder) + if err == nil { + t.Fatal("an unknown route_at value was accepted; core guessed a floor") + } + for _, want := range []string{"route_at", "urgent", "severity"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q:\n%v", want, err) + } + } +} + +// TestAggregateRouteAtAbsentIsANoOp is the backward-compatibility half of +// DKT-593: with no `route_at`, the outcome is BYTE-FOR-BYTE the one the +// builtin has always produced — nothing recorded, every cluster emitted at +// its input position, and G2's identity property untouched. +func TestAggregateRouteAtAbsentIsANoOp(t *testing.T) { + // Scalars and clusters, a hold, a demotion — every feature at once, so a + // floor leaking into the absent case has the widest surface to show on. + const raw = `[{"severity":"info","id":"A"},` + + `{"severity":["low","high"],"id":"B"},` + + `{"severity":["low","medium","high"],"id":"C"}]` + + out, err := aggregateOver(t, raw, medianParams(MethodMedian, 2), severityOrder) + testsupport.Must(t, err, "Aggregate: %v", err) + + if out.Recorded != nil { + t.Errorf("Recorded = %#v with no route_at declared; the record channel "+ + "must not exist for a step that never opted in", out.Recorded) + } + if len(out.Payload) != 3 { + t.Fatalf("emitted %d elements, want one per input element", len(out.Payload)) + } + for i, want := range []string{"A", "B", "C"} { + if got := out.Payload[i]["id"]; got != want { + t.Errorf("element %d is %v, want %s — input order preserved", i, got, want) + } + } + // Held indices are input indices when nothing was routed to the record. + if len(out.Held) != 2 || out.Held[0] != 1 || out.Held[1] != 2 { + t.Errorf("Held = %v, want [1 2] — the two spread-2 clusters at their "+ + "input positions", out.Held) + } +} + +// TestAggregateRouteAtNeverRoutesAHeldCluster pins the interaction: a held +// cluster is emitted whatever its computed value, because that value is +// exactly what the hold refuses to trust — the operator resolving the hold +// may accept a different one. Held indices address the EMITTED payload, since +// that is the payload the artifact records and `-held@k#i` resolves +// against (H2a). +func TestAggregateRouteAtNeverRoutesAHeldCluster(t *testing.T) { + // Cluster 0 reduces to info, below the floor, not held -> recorded. + // Cluster 1 reduces to low (median of {low,high}), below the floor, but + // spread 2 trips the hold -> emitted and held, at EMITTED index 0. + const raw = `[{"severity":"info","id":"A"},` + + `{"severity":["low","high"],"id":"B"}]` + + out, err := aggregateOver(t, raw, + routeAtParams(MethodMedian, 2, "high"), severityOrder) + testsupport.Must(t, err, "Aggregate: %v", err) + + if len(out.Payload) != 1 || out.Payload[0]["id"] != "B" { + t.Fatalf("emitted %#v, want only the held cluster B", out.Payload) + } + if out.Payload[0][KeyHeld] != true { + t.Error("the emitted cluster is not marked held") + } + if len(out.Held) != 1 || out.Held[0] != 0 { + t.Errorf("Held = %v, want [0] — the held cluster's position in the "+ + "EMITTED payload, which is the payload the artifact records", out.Held) + } + if len(out.Recorded) != 1 || out.Recorded[0]["id"] != "A" { + t.Errorf("recorded %#v, want only the unheld below-floor cluster A", + out.Recorded) + } +} + +// TestAggregateRouteAtEmptyOutputStillSatisfiesTheShippedSchema covers the +// every-cluster-below case: the emitted payload is the EMPTY ARRAY — a valid +// `aggregate@1` document over which any threshold predicate finds nothing — +// never null and never a failure. +func TestAggregateRouteAtEmptyOutputStillSatisfiesTheShippedSchema(t *testing.T) { + out, err := aggregateOver(t, `[{"severity":"info"},{"severity":"low"}]`, + routeAtParams(MethodMedian, 0, "blocker"), severityOrder) + testsupport.Must(t, err, "Aggregate: %v", err) + + if len(out.Payload) != 0 || len(out.Recorded) != 2 { + t.Fatalf("emitted %d, recorded %d; want 0 and 2", + len(out.Payload), len(out.Recorded)) + } + encoded, err := json.Marshal(out.Payload) + testsupport.Must(t, err, "encoding: %v", err) + if string(encoded) != "[]" { + t.Errorf("the empty output encodes as %s, want []", encoded) + } + builtin, err := aggregateSchema() + testsupport.Must(t, err, "compiling the embedded document: %v", err) + err = builtin.ValidatePayload(encoded) + testsupport.Must(t, err, "an empty emitted payload does not satisfy the shipped "+ + "schema: %v", err) +} diff --git a/internal/engine/autoregister.go b/internal/engine/autoregister.go index eccaa0a0..16c20a8f 100644 --- a/internal/engine/autoregister.go +++ b/internal/engine/autoregister.go @@ -436,20 +436,39 @@ func registerScanTx(tx *sql.Tx, projectID int, scan *configScan, nowMS int64) ([ // registerConfigWorkflowTx, which already word it precisely; a second refusal // here would be the same failure reported twice, worse. func registryIdentityKey(path string, src []byte) (string, bool) { + kind, name, version, ok := configIdentity(path, src) + if !ok { + return "", false + } + return fmt.Sprintf("%s\x00%s@%d", kind, name, version), true +} + +// configIdentity is registryIdentityKey's three parts, unjoined: the KIND a +// config file registers as, and the NAME and VERSION it registers under. +// +// It exists because the cross-project audit (registry_audit.go) needs the +// version as a NUMBER to compare against a registered one, and re-deriving +// "where does a config file's identity come from" a second time is exactly how +// the audit and registration would come to disagree about what the corpus +// declares. Both callers read it from here, so they cannot. +// +// `src` is consulted ONLY for a workflow — a schema's identity is its filename +// — so a caller auditing a corpus may pass nil for a schema path rather than +// reading a document it will not parse. +func configIdentity(path string, src []byte) (kind, name string, version int, ok bool) { if isSchemaConfigPath(path) { base := strings.TrimSuffix(filepath.Base(path), filepath.Ext(path)) name, version, err := workflow.ParsePayloadRef(base) if err != nil { - return "", false + return "", "", 0, false } - return fmt.Sprintf("%s\x00%s@%d", RegistrationKindSchema, name, version), true + return RegistrationKindSchema, name, version, true } def, err := workflow.Parse(src) if err != nil { - return "", false + return "", "", 0, false } - return fmt.Sprintf("%s\x00%s@%d", - RegistrationKindWorkflow, def.Pipeline.Name, def.Pipeline.Version), true + return RegistrationKindWorkflow, def.Pipeline.Name, def.Pipeline.Version, true } // registerConfigSchemaTx registers one `.docket/config/schemas/NAME@V.json`. diff --git a/internal/engine/autoregister_test.go b/internal/engine/autoregister_test.go index 476fcd23..b3a86992 100644 --- a/internal/engine/autoregister_test.go +++ b/internal/engine/autoregister_test.go @@ -208,24 +208,30 @@ func TestAutoRegistrationWrongOrderFails(t *testing.T) { // F4-F6 — what is registered, what is pinned, what is skipped // --------------------------------------------------------------------------- -// TestAutoRegistrationPinsWhatItDoesNotUnderstand is F4, F5, and F6 together. +// TestAutoRegistrationPinsWhatItDoesNotUnderstand is F4, F5, and F6 together, +// under DKT-581's closure rule. // -// Only the two registries are REGISTERED. Everything else under -// `.docket/config/` is PINNED — which is exactly what §2 says core does with -// instance files it does not understand: "arbitrary operator-supplied file pins -// … which is how the reference instance pins its contracts, fragments, and -// policy WITHOUT CORE KNOWING WHAT THEY ARE". +// Only the two registries are REGISTERED. What else gets PINNED is the packet +// CLOSURE the bound workflows reach — each step's `packet` entries, the files +// their `packet_includes` frontmatter declares, and `policy.toml` — which is +// still exactly §2's stance ("core pins the instance's contracts, fragments, +// and policy WITHOUT KNOWING WHAT THEY ARE"), narrowed to what the run can +// actually read. A config file NOTHING reaches is not pinned at all, so a +// corpus edit to it cannot drift this run's pins. func TestAutoRegistrationPinsWhatItDoesNotUnderstand(t *testing.T) { conn, configDir := configRepo(t) - writeConfigFile(t, configDir, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/api.md\"]\n") writeConfigFile(t, configDir, "policy.toml", "spawn = \"whatever the instance means\"\n") // F5: the pinned directories are scanned RECURSIVELY — a fragment three - // levels deep is just a file to hash. + // levels deep is just a file to hash, reachable here through the + // contract's own `packet_includes`. writeConfigFile(t, configDir, "fragments/a/b/deep.md", "a fragment\n") - writeConfigFile(t, configDir, "contracts/api.md", "a contract\n") + writeConfigFile(t, configDir, "contracts/api.md", + "---\npacket_includes:\n - fragments/a/b/deep.md\n---\na contract\n") // F6: a non-matching extension inside a REGISTRY directory is skipped - // silently there and pinned like any other file. A refusal would make a - // README a run-blocker. + // silently there. A refusal would make a README a run-blocker — and since + // nothing references it, DKT-581 leaves it unpinned too. writeConfigFile(t, configDir, "workflows/README.md", "how to edit these\n") issue := createIssue(t, conn, "pinned", "body", "task", nil) @@ -238,25 +244,32 @@ func TestAutoRegistrationPinsWhatItDoesNotUnderstand(t *testing.T) { t.Fatalf("registered %d files, want only the workflow: %+v", len(result.Registered), result.Registered) } - // The four non-registry files: policy.toml, the deep fragment, the - // contract, and the README. - if result.PinsFromConfig != 4 { - t.Errorf("pinned %d config files, want 4 (policy, fragment, contract, README)", + // The closure: policy.toml, the declared contract, and the fragment its + // `packet_includes` reaches. NOT the README nothing references. + if result.PinsFromConfig != 3 { + t.Errorf("pinned %d config files, want 3 (policy, contract, included fragment)", result.PinsFromConfig) } // The pins are RECORDED, so `context.pins` can serve them and a run can say // what it reproduced against. pins := pinsByKind(t, conn, run.ID, db.PinKindFile) - var sawFragment bool + var sawFragment, sawREADME bool for _, p := range pins { if strings.HasSuffix(p.Ref, filepath.Join("fragments", "a", "b", "deep.md")) { sawFragment = true } + if strings.HasSuffix(p.Ref, "README.md") { + sawREADME = true + } } if !sawFragment { - t.Error("F5: a fragment three levels deep was not pinned; the pinned " + - "directories are scanned recursively") + t.Error("F5: a fragment three levels deep was not pinned; the closure " + + "walks `packet_includes` into recursively-scanned directories") + } + if sawREADME { + t.Error("DKT-581: a config file nothing references was pinned; a corpus " + + "edit to it would drift this run's pins for no reason") } } @@ -618,7 +631,10 @@ func TestAutoRegisterDisabledDoesNotAdoptANewVersion(t *testing.T) { // leave it needing every one of them hand-supplied via --pin. func TestAutoRegisterDisabledStillPinsTheConfigDirectory(t *testing.T) { conn, configDir := configRepo(t) - writeConfigFile(t, configDir, "workflows/auto-dev.toml", autoWorkflowSrc) + // The bound workflow DECLARES the contract, so it is in DKT-581's pin + // closure — the pinning this test proves survives the toggle. + src := autoWorkflowSrc + "packet = [\"contracts/note.md\"]\n" + writeConfigFile(t, configDir, "workflows/auto-dev.toml", src) writeConfigFile(t, configDir, "contracts/note.md", "a pinned fragment, not a registry file") err := db.SetConfig(conn, 0, db.KeyAutoRegister, "false") @@ -627,7 +643,7 @@ func TestAutoRegisterDisabledStillPinsTheConfigDirectory(t *testing.T) { // With registration off, an operator registers by hand — the same path // `workflow register` runs, exactly as F7 requires auto-registration // itself to. - registerSource(t, conn, []byte(autoWorkflowSrc), "auto-dev.toml") + registerSource(t, conn, []byte(src), "auto-dev.toml") issue := createIssue(t, conn, "pin check", "body", "task", nil) run := startRun(t, conn, issue) diff --git a/internal/engine/backfill.go b/internal/engine/backfill.go index e84a3fa1..a7f9ce70 100644 --- a/internal/engine/backfill.go +++ b/internal/engine/backfill.go @@ -174,13 +174,35 @@ type SkippedRow struct { Unit string `json:"unit"` } +// UnclaimedTarget names one back-filled step that NO WORKER EVER CLAIMED, so +// the advisory can be rendered by the caller and read by a JSON consumer +// (DKT-993). +// +// `Status` is on the row because it is the whole diagnostic: `pending` says the +// row landed on a step that has not run YET, `superseded` says it landed on one +// the loop swept before it ever could, and `skipped`/`failed-routed` say it +// landed on one that was terminated unrun. An operator reading "STEP-3146 +// (design-qa@1) is pending" knows immediately that the relay's journal +// mis-joined; a bare step id does not carry that. +type UnclaimedTarget struct { + Step string `json:"step"` + Instance string `json:"instance"` + Status string `json:"status"` +} + // BackfillOutcome is what the back-fill did: rows written, steps touched, the -// source they carry, and every row skipped as already recorded. +// source they carry, every row skipped as already recorded, and every target +// that was never claimed. type BackfillOutcome struct { Written int `json:"written"` Steps int `json:"steps"` Source string `json:"source"` Skipped []SkippedRow `json:"skipped,omitempty"` + // Unclaimed names the targets no worker ever held. The rows for them ARE + // written — see BackfillUsage — so this is an advisory, not a refusal, and + // it is on the payload because `Warn` is suppressed in JSON mode and a + // conductor runs with `--json`. + Unclaimed []UnclaimedTarget `json:"unclaimed,omitempty"` } // BackfillUsage records usage for steps whose claimant could not report it. @@ -205,6 +227,39 @@ type BackfillOutcome struct { // back-fill falling through that default would label a relay's reconstruction // as the claimant's own report and destroy the distinction the column exists // to preserve. +// +// A TARGET NO WORKER EVER CLAIMED IS RECORDED AND REPORTED, never refused +// (DKT-993). Harness RUN-66's conductor back-filled a wave journal whose join +// heuristic had misattributed probe-agent tokens, and the rows landed on +// STEP-3146 (`design-qa@1`, `pending`, never claimed) and on a superseded step +// with no refusal, no warning, and no diagnostic — so `run report` showed spend +// on steps that never ran and nothing in the store said where it came from. +// +// THE CONTRACT IS WARN-AND-RECORD, and the choice is deliberate. Trusting the +// relay stays the default: core cannot know that a step which never claimed did +// not cost anything — a pre-gate runs at claim time, a spawn can burn tokens +// before the claim lands, and core enumerating which steps are ALLOWED to have +// cost would be core holding an opinion about how a relay measures its own +// work, the same opinion `--source` and `unit` exist to avoid holding. What was +// wrong was the SILENCE. So the rows land, the ledger stays the relay's to +// write, and every unclaimed target is named on the outcome for the caller to +// warn about. +// +// "NEVER CLAIMED" IS `attempt == 0`, the same predicate D2's own exemption uses +// (missingUsage, D5's second half): `attempt` counts claims and nothing else — +// it is not bumped on failure, and `step resolve --as retry` moves +// `attempt_base` rather than zeroing it — so a zero attempt is exactly "no +// worker ever held this step", for every status at once. It is one field +// already on the row rather than an event scan, and it cannot disagree with the +// discrepancy probe that reads it three lines away. +// +// It covers the SUPERSEDED case the same evidence names, and covers it +// correctly. The loop's supersede sweep takes `pending` instances only, so a +// swept step is a never-claimed step and warns here by the same clause; a +// `fix-loop` resolution supersedes a step that HAD claimed and HAD spent, and +// that one does not warn — its usage is real, and back-filling it is D2's own +// documented resolution. Warning on status alone would have fired on the +// legitimate path and stayed quiet on neither. func (e *Engine) BackfillUsage( conn *sql.DB, runID int, rows []BackfillRow, source string, onDuplicate string, nowMS int64, @@ -227,8 +282,9 @@ func (e *Engine) BackfillUsage( } var ( - written int - skipped []SkippedRow + written int + skipped []SkippedRow + unclaimed []UnclaimedTarget ) tx, err := conn.Begin() @@ -258,6 +314,18 @@ func (e *Engine) BackfillUsage( model.FormatRunID(runID)) } steps[r.Step] = step + + // DKT-993. Collected in the SAME first-seen pass so the advisory is + // one entry per step rather than one per row, and in ROW ORDER so two + // identical batches produce identical output — the same + // golden-stability discipline the discrepancy list sorts for. + if step.Attempt == 0 { + unclaimed = append(unclaimed, UnclaimedTarget{ + Step: model.FormatStepID(step.ID), + Instance: step.Instance, + Status: step.Status, + }) + } } for _, r := range rows { @@ -321,5 +389,6 @@ func (e *Engine) BackfillUsage( } return &BackfillOutcome{ Written: written, Steps: len(steps), Source: source, Skipped: skipped, + Unclaimed: unclaimed, }, nil } diff --git a/internal/engine/backfill_test.go b/internal/engine/backfill_test.go index a01535d4..09497cd5 100644 --- a/internal/engine/backfill_test.go +++ b/internal/engine/backfill_test.go @@ -6,6 +6,7 @@ import ( "testing" "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/testsupport" ) @@ -305,6 +306,210 @@ func TestBackfillRefusesAStepOfAnotherRun(t *testing.T) { } } +// ---- DKT-993: the never-claimed target ------------------------------------- +// +// Harness RUN-66's conductor back-filled a wave journal whose join heuristic had +// misattributed probe-agent tokens. The rows landed on STEP-3146 (`design-qa@1`, +// `pending`, never claimed) and on a superseded step, and the engine took them +// in silence — so `run report` showed spend on steps that never ran, with +// nothing in the store saying where it came from. +// +// THE CONTRACT IS WARN-AND-RECORD. The row still lands (core cannot know that a +// step which never claimed cost nothing, and enumerating which steps may have +// cost would be core holding an opinion about how a relay measures its own +// work); what ends is the silence. + +// TestBackfillNamesANeverClaimedTarget is the acceptance case verbatim: a row +// against a `pending`, never-claimed step is REPORTED, naming the step and its +// status — and is still recorded. +func TestBackfillNamesANeverClaimedTarget(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + openDispatch(t, conn, runID, 0, nowMS) + + // `verify@0` is downstream of everything and has never been offered, let + // alone claimed — RUN-66's `design-qa@1` shape. The premise is ASSERTED + // rather than assumed: if the fixture ever pre-claims this step, the test + // must fail here and not silently stop testing anything. + verifyID := stepIDByInstance(t, conn, "verify@0") + step, err := db.GetStep(conn, verifyID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepPending || step.Attempt != 0 { + t.Fatalf("premise: verify@0 is %q at attempt %d, want %q at attempt 0", + step.Status, step.Attempt, db.StepPending) + } + + out, err := e.BackfillUsage(conn, runID, []BackfillRow{ + {Step: verifyID, Unit: "cache_creation", Quantity: 8724}, + }, "wave-journal:wf_68b7e3e1-abe", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + // Reported, by name and by status. + if len(out.Unclaimed) != 1 { + t.Fatalf("Unclaimed = %+v, want exactly one entry — the row landed on a "+ + "step no worker ever claimed and the engine must say so", out.Unclaimed) + } + got := out.Unclaimed[0] + if got.Step != model.FormatStepID(verifyID) { + t.Errorf("the advisory names step %q, want %q — an instance repeats "+ + "across issues and cannot be acted on alone", + got.Step, model.FormatStepID(verifyID)) + } + if got.Instance != "verify@0" { + t.Errorf("the advisory names instance %q, want %q", got.Instance, "verify@0") + } + if got.Status != db.StepPending { + t.Errorf("the advisory reports status %q, want %q — the status is the "+ + "whole diagnostic: it says the step has not run YET", + got.Status, db.StepPending) + } + + // And the row IS recorded: warn-and-record, not reject. + if out.Written != 1 { + t.Errorf("Written = %d, want 1 — the advisory does not withhold the row", + out.Written) + } + rows := usageRowsFor(t, conn, verifyID) + if len(rows) != 1 || rows[0].Quantity != 8724 { + t.Fatalf("usage_ledger rows = %+v, want the one 8724 row; the contract "+ + "is warn-and-record", rows) + } + if rows[0].Source != "wave-journal:wf_68b7e3e1-abe" { + t.Errorf("source = %q, want the given source verbatim", rows[0].Source) + } +} + +// TestBackfillIsQuietForAClaimedTarget is the "existing valid back-fills +// unchanged" half. The verb exists for a step that RAN and could not report its +// own spend; that step must never draw the advisory. +func TestBackfillIsQuietForAClaimedTarget(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + openDispatch(t, conn, runID, 0, nowMS) + + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + + out, err := e.BackfillUsage(conn, runID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 48211}, + }, "", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + if len(out.Unclaimed) != 0 { + t.Fatalf("Unclaimed = %+v on a claimed, completed step, want none — a "+ + "warning on the verb's own reason for existing is noise that "+ + "teaches conductors to ignore it", out.Unclaimed) + } +} + +// TestBackfillNamesASupersededSweptTarget covers the evidence's second step +// (STEP-3158, `verify-tribunal@1`, superseded). +// +// The loop's supersede sweep takes `pending` instances ONLY, so a swept step is +// a never-claimed step: it warns by the same clause, and its `superseded` status +// rides the advisory so an operator can tell it from a step merely waiting. +func TestBackfillNamesASupersededSweptTarget(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + // A real fix-loop entry, through the engine: `commit@0` is downstream of + // `after_loop` and unclaimed, so the sweep supersedes it. + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", "report", unmetPayload) + + commitID := stepIDByInstance(t, conn, "commit@0") + step, err := db.GetStep(conn, commitID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepSuperseded || step.Attempt != 0 { + t.Fatalf("premise: commit@0 is %q at attempt %d, want %q at attempt 0", + step.Status, step.Attempt, db.StepSuperseded) + } + + out, err := e.BackfillUsage(conn, run.ID, []BackfillRow{ + {Step: commitID, Unit: "cache_read", Quantity: 8336}, + }, "", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + if len(out.Unclaimed) != 1 || out.Unclaimed[0].Status != db.StepSuperseded { + t.Fatalf("Unclaimed = %+v, want one entry reporting %q", + out.Unclaimed, db.StepSuperseded) + } + if rows := usageRowsFor(t, conn, commitID); len(rows) != 1 { + t.Errorf("usage_ledger rows = %+v, want the one row; the contract is "+ + "warn-and-record", rows) + } +} + +// TestBackfillIsQuietForASupersededStepThatRan is the discrimination the status +// alone could not make, and the reason the predicate is the CLAIM. +// +// `step resolve --as fix-round` supersedes the step it authorizes a round for — +// a step that claimed, ran, and spent. Its usage is real, and back-filling it is +// D2's own documented resolution, so warning on `superseded` as a status would +// have fired on the legitimate path. +func TestBackfillIsQuietForASupersededStepThatRan(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", "report", unverifiablePayload) + + verifyID := stepIDByInstance(t, conn, "verify@0") + testsupport.Must(t, e.ResolveStep(conn, verifyID, ResolveFixRound, + "one more round", nowMS), "resolve --as fix-round: %v", nil) + + step, err := db.GetStep(conn, verifyID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepSuperseded || step.Attempt == 0 { + t.Fatalf("premise: verify@0 is %q at attempt %d, want %q at a nonzero "+ + "attempt", step.Status, step.Attempt, db.StepSuperseded) + } + + out, err := e.BackfillUsage(conn, run.ID, []BackfillRow{ + {Step: verifyID, Unit: "tokens", Quantity: 1200}, + }, "", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + if len(out.Unclaimed) != 0 { + t.Fatalf("Unclaimed = %+v for a superseded step that CLAIMED and ran, "+ + "want none — its spend is real and this is D2's own way out", + out.Unclaimed) + } +} + +// TestBackfillNamesEachUnclaimedTargetOnce pins the advisory's shape on a batch: +// one entry per STEP (not per row), in the order the batch named them, so two +// identical batches produce identical output. +func TestBackfillNamesEachUnclaimedTargetOnce(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + openDispatch(t, conn, runID, 0, nowMS) + + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + verifyID := stepIDByInstance(t, conn, "verify@0") + + out, err := e.BackfillUsage(conn, runID, []BackfillRow{ + {Step: verifyID, Unit: "input", Quantity: 18}, + {Step: implID, Unit: "tokens", Quantity: 4000}, + {Step: verifyID, Unit: "output", Quantity: 7}, + }, "", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + if len(out.Unclaimed) != 1 || out.Unclaimed[0].Instance != "verify@0" { + t.Fatalf("Unclaimed = %+v, want exactly one entry for verify@0 — two "+ + "rows against one step are one problem", out.Unclaimed) + } + if out.Written != 3 { + t.Errorf("Written = %d, want 3", out.Written) + } +} + // TestBackfillLeavesOtherStepsRefusing is the other half of §4.2's last bullet: // D2 goes quiet for exactly the back-filled steps, not for the run at large. func TestBackfillLeavesOtherStepsRefusing(t *testing.T) { diff --git a/internal/engine/batch_override.go b/internal/engine/batch_override.go new file mode 100644 index 00000000..4b4146b7 --- /dev/null +++ b/internal/engine/batch_override.go @@ -0,0 +1,185 @@ +package engine + +import ( + "database/sql" + "fmt" + "sort" + "strconv" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-546 — the run-scoped batch gate-override. +// +// The dominant operator toil the refit mining found was environmental gate +// parks resolved one at a time: the same "sandbox artifact, not a code defect" +// ruling re-made for every step of a run. `step resolve --as override-pass +// --batch` records that ruling ONCE, as one grant per failed gate (gate name + +// exit + reason — the failure signature), and the routing stage consults the +// run's grants before parking a later step whose failure carries the same +// signature. +// +// The gates STILL RUN. What the grant changes is the routing of a failure +// whose signature the operator already ruled on — a failure with a different +// signature parks exactly as before, which is what keeps a new real defect +// from sailing through under an old environmental ruling. +// +// THE COVERED STEP NEED NOT HAVE EXISTED WHEN THE GRANT WAS TAKEN (DKT-734). +// grantMatches compares the FAILURE SIGNATURE and nothing else — not the step, +// not the loop ordinal, not the round — because the scope rule lives one level +// up, in the grant's run_id. So a grant minted on a parked `fix@7` covers the +// `fix@8` a later fix round mints, which is what RUN-51 saw and misread as +// three authorizations. Deliberately so: an environmental failure does not +// become a code defect because the loop went around again, and expiring the +// grant per round would re-ask the identical settled question every round — +// the exact toil DKT-546 measured. The operator's protection is that the +// question stays settled only while the ANSWER does: a round whose gate fails +// with a different exit or reason parks for a fresh decision. + +// failingCompletionRows reduces a step's recorded gate rows to the last-ordinal +// completion rows that did not pass — the rows a batch override grant records, +// and the rows it must cover to route a later step past its park. The same +// reduction verdictOverRows routes on: `pre` rows excluded, last ordinal per +// gate wins, so a flaky gate that failed twice and passed third is not failing. +func failingCompletionRows(rows []db.GateResultRow) []db.GateResultRow { + last := make(map[string]db.GateResultRow) + for _, r := range rows { + if r.Pre { + continue + } + prev, seen := last[r.Gate] + if !seen || r.Ordinal >= prev.Ordinal { + last[r.Gate] = r + } + } + var failing []db.GateResultRow + for _, r := range last { + if r.Verdict != db.GateVerdictPass { + failing = append(failing, r) + } + } + // Sorted so the routing reason names the gates in the same order on every + // run over the same rows — map range order is not. + sort.Slice(failing, func(i, j int) bool { + return failing[i].Gate < failing[j].Gate + }) + return failing +} + +// grantMatches reports whether one grant covers one failing row: same gate, +// same exit, same reason classification. NULL exit matches only NULL — an +// `unmatched` gate never ran, and "no process existed" is not exit 0. +func grantMatches(g db.GateOverrideGrant, r db.GateResultRow) bool { + if g.Gate != r.Gate || g.Reason != r.Reason { + return false + } + if (g.Exit == nil) != (r.Exit == nil) { + return false + } + return g.Exit == nil || *g.Exit == *r.Exit +} + +// batchCover is the routing stage's read of the run's grants against one +// failed step: which grants cover its failing gates, or why a full cover must +// not be auto-applied. +type batchCover struct { + grantIDs []int + gates []string + // blocked, when non-empty, is the park reason for a cover that matched but + // must not auto-apply: the step's threshold interposes another step, and an + // auto-applied override-pass would silently skip it — the DKT-470 defect, + // at scale, with no operator present to read the warning. + blocked string +} + +// reason is the routing record an auto-pass carries, so the ledger shows every +// covered step and the grant(s) whose justification it rode on. +func (c *batchCover) reason() string { + return fmt.Sprintf( + "gate(s) %s failed with signature(s) covered by batch override "+ + "grant(s) %s; auto-passed per the operator's run-scoped ruling", + strings.Join(c.gates, ", "), joinGrantIDs(c.grantIDs)) +} + +// eventData is the `step-batch-overridden` payload: the covering grant id(s). +func (c *batchCover) eventData() string { + return joinGrantIDs(c.grantIDs) +} + +func joinGrantIDs(ids []int) string { + parts := make([]string, len(ids)) + for i, id := range ids { + parts[i] = strconv.Itoa(id) + } + return strings.Join(parts, ",") +} + +// batchOverrideCover reads the run's grants against a step whose gate verdict +// is `fail` and reports the cover, nil when there is none. +// +// EVERY failing gate must match a grant, or nothing applies: a step that +// failed one ruled-on gate and one new one carries information the operator +// has not seen, and auto-passing it would spend the ruling past its terms. It +// reads on the pooled connection because the routing stage calls it before its +// transaction opens, exactly as gateVerdict reads. +func batchOverrideCover( + conn *sql.DB, step *db.Step, spec *workflow.Step, +) (*batchCover, error) { + grants, err := db.GateOverrideGrantsForRun(conn, step.RunID) + if err != nil { + return nil, err + } + if len(grants) == 0 { + return nil, nil + } + rows, err := db.GateResultsForStep(conn, step.ID) + if err != nil { + return nil, err + } + failing := failingCompletionRows(rows) + if len(failing) == 0 { + return nil, nil + } + + cover := &batchCover{} + seen := make(map[int]bool) + for _, r := range failing { + matched := false + for _, g := range grants { + if grantMatches(g, r) { + matched = true + if !seen[g.ID] { + seen[g.ID] = true + cover.grantIDs = append(cover.grantIDs, g.ID) + } + break + } + } + if !matched { + return nil, nil + } + cover.gates = append(cover.gates, r.Gate) + } + + // A threshold that interposes another step forbids the auto-apply + // (DKT-470): override-pass records a generic `pass` without evaluating the + // threshold, so the interposed step would be unconditionally skipped — and + // unlike the operator's own override-pass, an auto-applied one has nobody + // present to read OverridePassSkipsInterposedTargets' warning. The step + // parks per its `on_fail` with the cover named, and the operator resolves + // it individually, warning and all. + if spec != nil { + if targets := workflow.ThresholdTargets(spec.Threshold); len(targets) > 0 { + cover.blocked = fmt.Sprintf( + "batch override grant(s) %s cover the failed gate(s) %s, but "+ + "this step's threshold interposes %s, which an auto-applied "+ + "override-pass would silently skip (DKT-470); resolve "+ + "individually", + joinGrantIDs(cover.grantIDs), strings.Join(cover.gates, ", "), + strings.Join(targets, ", ")) + } + } + return cover, nil +} diff --git a/internal/engine/batch_override_test.go b/internal/engine/batch_override_test.go new file mode 100644 index 00000000..c1439e81 --- /dev/null +++ b/internal/engine/batch_override_test.go @@ -0,0 +1,618 @@ +package engine + +import ( + "context" + "database/sql" + "encoding/json" + "fmt" + "strconv" + "strings" + "sync" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-546 — the run-scoped batch gate-override. +// +// One operator override decision covers subsequent identical environmental +// gate failures in the same run. The gates still run; what the grant changes +// is the routing of a failure whose signature (gate + exit + reason) the +// operator already ruled on. A different signature still parks, another run +// re-asks, and every auto-pass is attributed to its grant in the ledger. + +// batchOverrideSrc: two sequential executor steps sharing one gate, so a +// grant minted on the first step's park has a second, identical failure to +// cover. +const batchOverrideSrc = ` +[pipeline] +name = "batch-override" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" +gates = ["build"] +on_fail = "waiting-human" + +[[step]] +name = "package" +after = ["implement"] +executor = "package" +emits = "package-record" +gates = ["build"] +on_fail = "waiting-human" +` + +// batchInterposeSrc adds the DKT-470 shape: the second gated step's threshold +// interposes a vote step, which forbids the auto-apply. +const batchInterposeSrc = ` +[pipeline] +name = "batch-interpose" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" +gates = ["build"] +on_fail = "waiting-human" + +[[step]] +name = "audit" +after = ["implement"] +executor = "audit" +emits = "report" +gates = ["build"] +on_fail = "waiting-human" +threshold = { "tribunal" = "any(status == blocked)" } + +[[step]] +name = "tribunal" +after = ["audit"] +type = "vote" +voters = ["seat-a", "seat-b", "seat-c"] +vote_rule = "majority" +on_fail = "waiting-human" + +[[step]] +name = "finish" +after = ["tribunal"] +executor = "finish" +emits = "record" +` + +// exitGates is a GateRunner whose failure EXIT CODE is settable, so a test can +// produce two failures of the same gate with different signatures. +type exitGates struct { + mu sync.Mutex + fail bool + exit int +} + +func (g *exitGates) Run(_ context.Context, spec GateSpec, _ StepContext) (GateResult, error) { + g.mu.Lock() + defer g.mu.Unlock() + if g.fail { + return GateResult{Gate: spec.Name, Exit: g.exit, Verdict: VerdictFail}, nil + } + return GateResult{Gate: spec.Name, Exit: 0, Verdict: VerdictPass}, nil +} + +// activatedBatchRun registers src, activates a one-issue run over it, and +// returns the run id. +func activatedBatchRun(t *testing.T, conn *sql.DB, src, path string) int { + t.Helper() + registerSource(t, conn, []byte(src), path) + issue := createIssue(t, conn, "batch fixture", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + return run.ID +} + +// stepIDIn resolves an instance WITHIN one run — stepIDByInstance is ambiguous +// once a second run expands the same workflow. +func stepIDInRun(t *testing.T, conn *sql.DB, runID int, instance string) int { + t.Helper() + var id int + err := conn.QueryRow( + `SELECT id FROM steps WHERE run_id = ? AND instance = ?`, runID, instance, + ).Scan(&id) + testsupport.Must(t, err, "finding step %q in run %d: %v", instance, runID, err) + return id +} + +// parkThroughFailingGate claims and completes one step so its failing gate +// parks it, and asserts the park. +func parkThroughFailingGate(t *testing.T, conn *sql.DB, e *Engine, stepID int) { + t.Helper() + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the change summary"), NowMS: nowMS, + }) + testsupport.Must(t, err, "complete: %v", err) + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepWaitingHuman { + t.Fatalf("premise: %s = %q after a failing gate, want %q", + step.Instance, step.Status, db.StepWaitingHuman) + } +} + +func stepRoutingByID(t *testing.T, conn *sql.DB, stepID int) string { + t.Helper() + var routing string + err := conn.QueryRow( + `SELECT routing FROM steps WHERE id = ?`, stepID).Scan(&routing) + testsupport.Must(t, err, "reading routing of step %d: %v", stepID, err) + return routing +} + +// TestBatchOverrideMintsGrantsAndCoversAMatchingLaterFailure is the feature's +// whole contract in one run: the operator override-passes the first park with +// --batch, one grant per failed gate is recorded and event-logged, and the +// SECOND step failing the same gate with the same signature routes `done` +// with a `pass` naming the grant — no second park, no second operator ruling. +func TestBatchOverrideMintsGrantsAndCoversAMatchingLaterFailure(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "batch-override.toml") + + gates := &exitGates{fail: true, exit: 1} + e := testEngine() + e.Gates = gates + + implementID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, implementID) + + err := e.ResolveStepBatch(conn, implementID, ResolveOverridePass, + "sandbox artifact, not a code defect", nowMS+1) + testsupport.Must(t, err, "resolve --as override-pass --batch: %v", err) + + // ONE grant, carrying the failure signature and the shared justification. + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if len(grants) != 1 { + t.Fatalf("grants = %d, want 1 (one per failed gate)", len(grants)) + } + g := grants[0] + if g.Gate != "build" || g.Exit == nil || *g.Exit != 1 || g.Reason != "" { + t.Errorf("grant signature = (%q, %v, %q), want (build, 1, \"\")", + g.Gate, g.Exit, g.Reason) + } + if g.Note != "sandbox artifact, not a code defect" { + t.Errorf("grant note = %q; the shared justification must ride the grant", g.Note) + } + if g.OriginStepID != implementID { + t.Errorf("grant origin = %d, want %d", g.OriginStepID, implementID) + } + if got := eventKindCount(t, conn, runID, EventGateOverrideGranted); got != 1 { + t.Errorf("%s events = %d, want 1", EventGateOverrideGranted, got) + } + + // The second step fails the SAME gate with the SAME signature — and does + // not park. + packageID := stepIDInRun(t, conn, runID, "package@0") + claim, err := ClaimStep(conn, packageID, ClaimOptions{Owner: "w2", NowMS: nowMS + 2}) + testsupport.Must(t, err, "claim package: %v", err) + err = e.CompleteStep(conn, packageID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the package record"), NowMS: nowMS + 2, + }) + testsupport.Must(t, err, "complete package: %v", err) + + step, err := db.GetStep(conn, packageID) + testsupport.Must(t, err, "GetStep package: %v", err) + if step.Status != db.StepDone { + t.Fatalf("package@0 = %q after a covered failure, want %q", + step.Status, db.StepDone) + } + routing := stepRoutingByID(t, conn, packageID) + if !strings.HasPrefix(routing, RoutingPass) { + t.Errorf("package@0 routing = %q, want a %q routing", routing, RoutingPass) + } + if !strings.Contains(routing, "grant") { + t.Errorf("package@0 routing = %q; the auto-pass must name its authority", routing) + } + + // The ledger: the grant covered one step, attributably. + grants, err = db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "re-reading grants: %v", err) + if grants[0].CoveredSteps != 1 { + t.Errorf("covered_steps = %d, want 1", grants[0].CoveredSteps) + } + if got := eventKindCount(t, conn, runID, EventStepBatchOverridden); got != 1 { + t.Errorf("%s events = %d, want 1", EventStepBatchOverridden, got) + } +} + +// TestBatchGrantIgnoresADifferentSignature: the grant covers the ruled-on +// failure, not the gate. The same gate failing with a DIFFERENT exit carries +// information the operator has not seen, and it parks exactly as before. +func TestBatchGrantIgnoresADifferentSignature(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "batch-override.toml") + + gates := &exitGates{fail: true, exit: 1} + e := testEngine() + e.Gates = gates + + implementID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, implementID) + err := e.ResolveStepBatch(conn, implementID, ResolveOverridePass, + "no docker socket", nowMS+1) + testsupport.Must(t, err, "batch resolve: %v", err) + + // Same gate, different exit: a different failure. + gates.mu.Lock() + gates.exit = 2 + gates.mu.Unlock() + packageID := stepIDInRun(t, conn, runID, "package@0") + parkThroughFailingGate(t, conn, e, packageID) + + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if grants[0].CoveredSteps != 0 { + t.Errorf("covered_steps = %d after a non-matching failure, want 0", + grants[0].CoveredSteps) + } + if got := eventKindCount(t, conn, runID, EventStepBatchOverridden); got != 0 { + t.Errorf("%s events = %d, want 0", EventStepBatchOverridden, got) + } +} + +// TestBatchGrantIsRunScoped: a new run re-asks. The grant's run_id is the +// whole scope rule, so an identical failure in a SECOND run parks. +func TestBatchGrantIsRunScoped(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "batch-override.toml") + + gates := &exitGates{fail: true, exit: 1} + e := testEngine() + e.Gates = gates + + implementID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, implementID) + err := e.ResolveStepBatch(conn, implementID, ResolveOverridePass, + "sandbox artifact", nowMS+1) + testsupport.Must(t, err, "batch resolve: %v", err) + + // A second run over the same workflow, failing the same gate the same way. + issue2 := createIssue(t, conn, "second issue", "a body", "task", nil) + run2 := startRun(t, conn, issue2) + _, err = activate(conn, run2.ID) + testsupport.Must(t, err, "activate run 2: %v", err) + + implement2 := stepIDInRun(t, conn, run2.ID, "implement@0") + parkThroughFailingGate(t, conn, e, implement2) // asserts the park itself + + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if grants[0].CoveredSteps != 0 { + t.Errorf("covered_steps = %d after another RUN's failure, want 0 — "+ + "the grant must die at its run's edge", grants[0].CoveredSteps) + } +} + +// TestBatchRequiresOverridePass: --batch widens what one authorization covers, +// so it rides only the verb whose ruling it extends. +func TestBatchRequiresOverridePass(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "batch-override.toml") + + gates := &exitGates{fail: true, exit: 1} + e := testEngine() + e.Gates = gates + + implementID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, implementID) + + err := e.ResolveStepBatch(conn, implementID, ResolveSkip, "nope", nowMS+1) + if err == nil { + t.Fatal("--batch with --as skip was accepted; it must require override-pass") + } + if !strings.Contains(err.Error(), ResolveOverridePass) { + t.Errorf("refusal = %q, want it to name %s", err, ResolveOverridePass) + } +} + +// TestBatchRefusesAParkWithNoFailedGate: a park with no failed completion gate +// has no signature to grant from — a vote step waiting on its quorum is the +// reachable case — and the refusal is honest where an unmatchable grant would +// look like it did something. +func TestBatchRefusesAParkWithNoFailedGate(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchInterposeSrc, "batch-interpose.toml") + + e := testEngine() + tribunalID := stepIDInRun(t, conn, runID, "tribunal@0") + + err := e.ResolveStepBatch(conn, tribunalID, ResolveOverridePass, "moving on", nowMS) + if err == nil { + t.Fatal("--batch on a step with no failed gate was accepted") + } + if !strings.Contains(err.Error(), "no failed completion gate") { + t.Errorf("refusal = %q, want it to say there is nothing to grant from", err) + } +} + +// TestBatchCoverBlockedByAnInterposedThreshold is DKT-470 at scale: a matching +// grant must NOT auto-apply on a step whose threshold interposes another step, +// because the generic pass would silently skip it with no operator present to +// read the warning. The step parks per its on_fail, with the block named. +func TestBatchCoverBlockedByAnInterposedThreshold(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchInterposeSrc, "batch-interpose.toml") + + gates := &exitGates{fail: true, exit: 1} + e := testEngine() + e.Gates = gates + + implementID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, implementID) + err := e.ResolveStepBatch(conn, implementID, ResolveOverridePass, + "sandbox artifact", nowMS+1) + testsupport.Must(t, err, "batch resolve: %v", err) + + // audit@0 fails the same gate with the same signature — but its threshold + // interposes tribunal, so the cover is blocked and the step parks. + auditID := stepIDInRun(t, conn, runID, "audit@0") + parkThroughFailingGate(t, conn, e, auditID) // asserts the park itself + + routing := stepRoutingByID(t, conn, auditID) + if !strings.Contains(routing, "tribunal") || !strings.Contains(routing, "DKT-470") { + t.Errorf("audit@0 routing = %q, want the park reason to name the "+ + "interposed step and the DKT-470 block", routing) + } + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if grants[0].CoveredSteps != 0 { + t.Errorf("covered_steps = %d for a blocked cover, want 0", grants[0].CoveredSteps) + } + if got := eventKindCount(t, conn, runID, EventStepBatchOverridden); got != 0 { + t.Errorf("%s events = %d for a blocked cover, want 0", + EventStepBatchOverridden, got) + } +} + +// --------------------------------------------------------------------------- +// DKT-734 — the grant's reach ACROSS FIX-LOOP ROUNDS, and the audit trail that +// makes each application walk back to the operator. +// --------------------------------------------------------------------------- + +// batchLoopRoundSrc is RUN-51's shape: the gated step is the LOOP step, so the +// steps a grant covers are minted by later fix-round loop entries and did not +// exist when the operator ruled. Every test above covers steps that were +// already expanded at activation, which is the easy half of "run-scoped". +const batchLoopRoundSrc = ` +[pipeline] +name = "batch-loop-round" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +threshold = { "fix-loop" = "any(status == unmet)" } +max_fix_loops = 4 + +[[step]] +name = "fix" +executor = "fix" +emits = "change-summary" +loop = true +after_loop = "check" +gates = ["build"] +on_fail = "waiting-human" +` + +// eventDetailsOfKind returns one run's events of a kind, in sequence order, as +// the `detail` each carries — the field `docket events list` prints as +// `detail=...`, and the one an auditor walks. eventKindCount cannot see it, +// which is why every assertion above could only count. +func eventDetailsOfKind(t *testing.T, conn *sql.DB, runID int, kind string) []string { + t.Helper() + rows, err := conn.Query( + `SELECT data FROM events WHERE run_id = ? AND kind = ? ORDER BY seq`, + runID, kind) + testsupport.Must(t, err, "reading %s events: %v", kind, err) + defer rows.Close() + var out []string + for rows.Next() { + var data string + testsupport.Must(t, rows.Scan(&data), "scanning a %s event", kind) + var fields struct { + Detail string `json:"detail"` + } + testsupport.Must(t, json.Unmarshal([]byte(data), &fields), + "decoding a %s event's data", kind) + out = append(out, fields.Detail) + } + testsupport.Must(t, rows.Err(), "iterating %s events: %v", kind, rows.Err()) + return out +} + +// TestBatchGrantCoversLaterFixLoopRoundsSteps is DKT-734's question, answered +// in the engine rather than in prose: an operator grant minted on ONE round's +// parked step covers the steps LATER ROUNDS mint, because the grant's scope is +// the RUN and a fix round does not start a new one. +// +// This is BY DESIGN, not an escape: `gate_override_grants.run_id` is the whole +// scope rule (see GateOverrideGrant's doc comment), and a fix round mints new +// steps inside the same run. RUN-51 observed exactly this — one grant on +// `fix@7`, auto-passes on `fix@8` and `fix@9` — and read three authorizations +// where one standing authorization was spent three times. +// +// What keeps that honest is the signature match plus the ledger: the gates +// still run every round, a round whose failure differs still parks (the test +// below), and every application is its own `step-batch-overridden` event +// naming the grant id. +func TestBatchGrantCoversLaterFixLoopRoundsSteps(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchLoopRoundSrc, "batch-loop-round.toml") + + gates := &exitGates{fail: true, exit: 2} + e := testEngine() + e.Gates = gates + + // Round 0's check finds an unmet criterion, which enters the loop and + // mints `fix@1` — a step that did not exist at activation. + driveFixtureRound(t, 0) + claimAndComplete(t, conn, e, "check@0", "the round 0 assessment", unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: the unmet criterion must enter the loop and mint fix@1") + } + + // `fix@1`'s gate fails and it parks. The operator rules the failure + // environmental for the run. (Each round moves the stub tree, or DKT-340's + // non-convergence guard parks the loop before a second round is minted.) + driveFixtureRound(t, 1) + fix1 := stepIDInRun(t, conn, runID, "fix@1") + parkThroughFailingGate(t, conn, e, fix1) + testsupport.Must(t, e.ResolveStepBatch(conn, fix1, ResolveOverridePass, + "sandbox artifact, not a code defect", nowMS+1), + "resolve fix@1 --as override-pass --batch") + + granted := eventDetailsOfKind(t, conn, runID, EventGateOverrideGranted) + if len(granted) != 1 { + t.Fatalf("%s events = %d, want exactly 1 — one operator ruling", + EventGateOverrideGranted, len(granted)) + } + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if len(grants) != 1 { + t.Fatalf("grants = %d, want 1", len(grants)) + } + grantID := grants[0].ID + // The minting event carries `gate#grantid`, so the feed's ruling names the + // grant row every later application will cite. + if want := fmt.Sprintf("build#%d", grantID); granted[0] != want { + t.Errorf("%s data = %q, want %q — the ruling must name its grant row", + EventGateOverrideGranted, granted[0], want) + } + if grants[0].OriginStepID != fix1 { + t.Errorf("grant origin = %d, want fix@1 (%d)", grants[0].OriginStepID, fix1) + } + + // Rounds 2 and 3: each is minted by its own loop entry, AFTER the ruling, + // and each fails the same gate with the same signature. + for ordinal := 2; ordinal <= 3; ordinal++ { + prev := ordinal - 1 + claimAndComplete(t, conn, e, + fmt.Sprintf("check@%d", prev), + fmt.Sprintf("the round %d assessment", prev), unmetPayload) + + instance := fmt.Sprintf("fix@%d", ordinal) + if !stepExists(t, conn, instance) { + t.Fatalf("premise: round %d must mint %s", ordinal, instance) + } + driveFixtureRound(t, ordinal) + stepID := stepIDInRun(t, conn, runID, instance) + claim, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "w", NowMS: nowMS + int64(ordinal)}) + testsupport.Must(t, err, "claim %s: %v", instance, err) + testsupport.Must(t, e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the fix summary"), + NowMS: nowMS + int64(ordinal), + }), "complete %s", instance) + + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep %s: %v", instance, err) + if step.Status != db.StepDone { + t.Fatalf("%s = %q after a covered failure, want %q — the grant's "+ + "scope is the RUN, and a fix round does not start a new one", + instance, step.Status, db.StepDone) + } + routing := stepRoutingByID(t, conn, stepID) + if !strings.HasPrefix(routing, RoutingPass) || + !strings.Contains(routing, strconv.Itoa(grantID)) { + t.Errorf("%s routing = %q, want a %q naming grant %d", + instance, routing, RoutingPass, grantID) + } + } + + // STILL exactly one authorization: later rounds spend the standing grant, + // they do not mint new ones. + if got := eventKindCount(t, conn, runID, EventGateOverrideGranted); got != 1 { + t.Errorf("%s events = %d, want 1 — a covered round must not mint "+ + "authority the operator never granted", EventGateOverrideGranted, got) + } + // And each application is its own event, naming the grant it spent, so the + // feed distinguishes "authorized three times" from "one authorization + // spent three times". + spent := eventDetailsOfKind(t, conn, runID, EventStepBatchOverridden) + want := []string{strconv.Itoa(grantID), strconv.Itoa(grantID)} + if len(spent) != len(want) { + t.Fatalf("%s events = %d, want %d — one per covered step", + EventStepBatchOverridden, len(spent), len(want)) + } + for i, data := range spent { + if data != want[i] { + t.Errorf("%s[%d] data = %q, want %q — every application must walk "+ + "back to the grant, and the grant to the operator's ruling", + EventStepBatchOverridden, i, data, want[i]) + } + } + grants, err = db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "re-reading grants: %v", err) + if grants[0].CoveredSteps != 2 { + t.Errorf("covered_steps = %d, want 2 — the counter is the grant row's "+ + "own record of how far one ruling reached", grants[0].CoveredSteps) + } +} + +// TestBatchGrantParksALaterRoundWithADifferentSignature is the other half of +// DKT-734's answer, and the reason the run-scoped reach is safe: a standing +// grant does not make the loop step immune. A later round whose gate fails +// DIFFERENTLY parks for a fresh operator decision, exactly as an ungranted +// failure would. +func TestBatchGrantParksALaterRoundWithADifferentSignature(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchLoopRoundSrc, "batch-loop-round.toml") + + gates := &exitGates{fail: true, exit: 2} + e := testEngine() + e.Gates = gates + + driveFixtureRound(t, 0) + claimAndComplete(t, conn, e, "check@0", "the round 0 assessment", unmetPayload) + driveFixtureRound(t, 1) + fix1 := stepIDInRun(t, conn, runID, "fix@1") + parkThroughFailingGate(t, conn, e, fix1) + testsupport.Must(t, e.ResolveStepBatch(conn, fix1, ResolveOverridePass, + "sandbox artifact", nowMS+1), "batch resolve") + + // The next round's gate fails with a DIFFERENT exit: a failure the + // operator has not ruled on. + gates.mu.Lock() + gates.exit = 3 + gates.mu.Unlock() + + claimAndComplete(t, conn, e, "check@1", "the round 1 assessment", unmetPayload) + driveFixtureRound(t, 2) + fix2 := stepIDInRun(t, conn, runID, "fix@2") + parkThroughFailingGate(t, conn, e, fix2) // asserts the park itself + + grants, err := db.GateOverrideGrantsForRun(conn, runID) + testsupport.Must(t, err, "reading grants: %v", err) + if grants[0].CoveredSteps != 0 { + t.Errorf("covered_steps = %d after a differently-failing round, want 0", + grants[0].CoveredSteps) + } + if got := eventKindCount(t, conn, runID, EventStepBatchOverridden); got != 0 { + t.Errorf("%s events = %d, want 0 — a new failure signature parks for a "+ + "fresh decision no matter how many rounds a grant has covered", + EventStepBatchOverridden, got) + } +} diff --git a/internal/engine/budget.go b/internal/engine/budget.go index 9747dff3..b88ae802 100644 --- a/internal/engine/budget.go +++ b/internal/engine/budget.go @@ -3,9 +3,11 @@ package engine import ( "database/sql" "fmt" + "math" "github.com/ALT-F4-LLC/docket/internal/db" "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" ) // BUDGETS — engine-core §7, implemented clause by clause (TDD §4). @@ -270,11 +272,27 @@ func BudgetSourceOf(cap, configDefault float64) BudgetSource { } } -// RunFloorTx is §4.3, the whole of it: +// RunFloorTx is §4.3, plus DKT-584's vote-step accrual: // -// SELECT COALESCE(SUM(s.expected_cost), 0) -// FROM events e JOIN steps s ON s.id = e.step_id -// WHERE e.run_id = ? AND e.kind = 'step-claimed' +// SELECT COALESCE(SUM(s.expected_cost), 0) +// FROM events e JOIN steps s ON s.id = e.step_id +// WHERE e.run_id = ? AND e.kind = 'step-claimed' +// + SELECT COALESCE(SUM(expected_cost), 0) +// FROM steps WHERE run_id = ? AND kind = 'vote' +// +// THE SECOND SUM IS DKT-584. A vote step is unclaimable by construction +// (§6.15), so it can never write the `step-claimed` event the first sum counts +// — which made a declared `expected_cost` on a `type="vote"` step inert: the +// author stated what the panel would cost and the floor could not see it +// (RUN-34 ran 40.6% of its output tokens through seats the budget never saw). +// A vote step therefore accrues AT MATERIALIZATION — the step row's existence +// is the engine-owned fact, written at expansion (or when the engine mints a +// held ballot), exactly as the claim event is for executor work. It accrues +// once per row: a loop's next ordinal is a new row and re-accrues (B10's +// analogue), and a materialized held ballot accrues the configured +// `vote.hold.cost`. The reservation side is adjusted to match — see +// reservableCost: a vote step's cost is already IN the floor, so R7 and the +// offer must not reserve it a second time. // // Every clause of §4.3's table falls out of this query rather than needing its // own code: @@ -293,19 +311,39 @@ func BudgetSourceOf(cap, configDefault float64) BudgetSource { // - B5, the value accrued is the STEP ROW's `expected_cost`, materialized at // expansion from the pinned definition and never re-read from the live // `workflows` table. A run pins its definitions; its floor is computed from -// what it pinned. +// what it pinned. DKT-867 adds ONE exception, per claim rather than per +// step: a claim whose dispatcher declared a variant-scaled cost (`step +// claim --cost-multiplier`) accrues the scaled cost its own `step-claimed` +// event carries in `data.expected_cost`, and the step row's declaration is +// only the fallback. The event is still the engine-owned fact — the scaled +// number was committed WITH the claim, in the same append-only log the +// floor has always summed — so replay, audit, and the no-lost-update +// property (C4) are untouched. Events of non-terminal runs are unprunable +// (§3), so enforcement can never lose a live run's scaled accruals. // // It is exported because `run report` computes the same number from outside this // package and the two must not be able to disagree — a report that recomputed // the floor its own way would be a second source of truth for the number a // breach is attributed to. func RunFloorTx(tx *sql.Tx, runID int) (float64, error) { + // The `json_valid` guard follows events_read.go's discipline: SQLite's + // `json_extract` ABORTS on malformed input rather than returning NULL, and + // a hand-edited event row must degrade to the step's declared cost, not + // take the whole budget check down. A pre-DKT-867 claim event's data + // carries no `expected_cost` key, so it extracts NULL, falls through to + // `s.expected_cost`, and historical floors are byte-for-byte what they + // were. var floor float64 err := tx.QueryRow( - `SELECT COALESCE(SUM(s.expected_cost), 0) - FROM events e JOIN steps s ON s.id = e.step_id - WHERE e.run_id = ? AND e.kind = ?`, - runID, EventStepClaimed, + `SELECT COALESCE((SELECT SUM(COALESCE( + CASE WHEN json_valid(e.data) + THEN json_extract(e.data, '$.expected_cost') END, + s.expected_cost)) + FROM events e JOIN steps s ON s.id = e.step_id + WHERE e.run_id = ? AND e.kind = ?), 0) + + COALESCE((SELECT SUM(expected_cost) + FROM steps WHERE run_id = ? AND kind = ?), 0)`, + runID, EventStepClaimed, runID, workflow.TypeVote, ).Scan(&floor) if err != nil { return 0, fmt.Errorf("computing the floor for %s: %w", @@ -314,6 +352,72 @@ func RunFloorTx(tx *sql.Tx, runID int) (float64, error) { return floor, nil } +// reservableCost is the cost a budget check may RESERVE for one step — the +// number `spend() + cost <= cap` adds on top of the snapshot. +// +// A vote step's declared cost is already in the floor the moment its row +// exists (RunFloorTx's second sum, DKT-584), so reserving it again at R7 or in +// the offer would count the same declaration twice and refuse a claimable +// batch a correct arithmetic admits. Every other step's cost accrues at claim, +// so it is reserved here exactly as before. +func reservableCost(step *db.Step) float64 { + if step.Kind == workflow.TypeVote { + return 0 + } + return step.ExpectedCost +} + +// VARIANT-AWARE DECLARED COST — DKT-867, the disposition. +// +// The defect: `expected_cost` is declared once per step template, so a fix +// step that a dispatcher escalated to a far pricier executor variant accrued +// exactly what its cheap sibling did (RUN-60 STEP-2752 vs STEP-2769: a 22.51M- +// token escalated attempt and a 1.83M ordinary one priced identically), and +// `budget.spend` sat at the floor in every measured run — escalation hops were +// invisible in the cap's own currency. +// +// Of DKT-867's two named remedies, the ledger-reading one is NOT taken here, +// for two reasons the code already states: (1) its safe form ALREADY EXISTS — +// DKT-238's measured dimension (`--usage-budget` + `budget.usage.unit`) is cap +// enforcement over the reported ledger, kept as a SEPARATE cap precisely +// because declared units and token counts are incommensurable (see +// budgetSnapshot.usageCap: `max()` over 280 declared units and hundreds of +// millions of tokens only ever answers the second); (2) making the token +// ledger the DECLARED cap's currency would require core to own a +// token-to-cap-unit conversion per variant — a pricing table keyed by model +// names, which the genericity rule (docs/design/genericity.md) forbids +// outright and which redesigns the whole ledger for a defect that is narrower +// than that. +// +// What IS taken is the variant-aware declared cost, placed where genericity +// allows it: the party that RESOLVES a step to a variant is the dispatcher, at +// claim time, so the dispatcher declares the scaling (`step claim +// --cost-multiplier`) and core does with the number exactly what it does with +// every other declared cost — records it as an engine-owned fact (in the +// claim's own `step-claimed` event, same transaction, same append-only log), +// sums it (RunFloorTx), and reserves it at the cap (budgetAdmits / +// enforceBudgetTx). Core never learns what a variant is; it learns that THIS +// claim costs N times what the definition guessed, from the one party in a +// position to know. An escalated hop is therefore priced at the cap the +// moment it is claimed, and the measured dimension remains the backstop for +// spend no declaration predicted. + +// validateCostMultiplier refuses a `--cost-multiplier` that is not a positive +// finite number. Zero means UNSET (the flag was never passed) and is valid; an +// explicit zero is refused at the CLI, because a free escalated claim would +// let a dispatcher erase an accrual the workflow author declared — the same +// direction B13 forbids for reported usage. +func validateCostMultiplier(m float64) error { + if m == 0 { + return nil + } + if math.IsNaN(m) || math.IsInf(m, 0) || m < 0 { + return validationErr( + "--cost-multiplier must be a positive finite number, got %g", m) + } + return nil +} + // `--usage` — recorded, capped, opaque (§4.9). // // WHAT CORE NEVER DOES WITH THESE NUMBERS: interpret them, convert them, @@ -392,6 +496,24 @@ func cacheRunFloorTx(tx *sql.Tx, runID int) error { // and budget is the most local of all: it is the only clause whose answer // depends on the specific step's cost. func (s *Scheduler) budgetHeadroom(step *db.Step) bool { + // The cost reserved is reservableCost, not ExpectedCost: a vote step's + // declared cost already sits in the floor (DKT-584), and reserving it here + // too would refuse the step for its own accrual. + // + // R7 reserves the DECLARED cost by construction: the offer cannot know + // which variant a dispatcher will resolve the step to, so the claim path + // re-checks with the dispatcher-declared scaled cost via budgetAdmits + // (DKT-867). + return s.budgetAdmits(step, reservableCost(step)) +} + +// budgetAdmits is budgetHeadroom with the cost made explicit — the claim +// path's entry point (DKT-867). The offer reserves a step's DECLARED cost +// (budgetHeadroom above); a claim that arrives with a dispatcher-declared +// variant scaling re-runs the same arithmetic over the same snapshot with the +// cost that will actually accrue, so an escalated claim is priced at the cap +// as what it is rather than as what the definition guessed. +func (s *Scheduler) budgetAdmits(step *db.Step, cost float64) bool { // BOTH dimensions must admit (DKT-238). They are independent caps over // different quantities, so neither can vouch for the other, and a step // passes only when it crosses neither. @@ -399,13 +521,13 @@ func (s *Scheduler) budgetHeadroom(step *db.Step) bool { if s.budgetHolds == nil { s.budgetHolds = make(map[string]float64) } - s.budgetHolds[step.Instance] = step.ExpectedCost + s.budgetHolds[step.Instance] = cost return false } if s.budget.unlimited() { return true } - if s.budget.admits(step.ExpectedCost) { + if s.budget.admits(cost) { return true } // Record the withholding so the offer can SAY it (DKT-242). Denying a step @@ -415,7 +537,7 @@ func (s *Scheduler) budgetHeadroom(step *db.Step) bool { if s.budgetHolds == nil { s.budgetHolds = make(map[string]float64) } - s.budgetHolds[step.Instance] = step.ExpectedCost + s.budgetHolds[step.Instance] = cost return false } @@ -486,8 +608,14 @@ func UsageBudgetBreachReason(spend, cap float64, unit, instance string) string { // the flip and returns the error. That ordering is deliberate: the pause and the // refusal are one fact, and a refusal that rolled back its own pause would leave // a run that refuses every claim while reporting itself active. -func enforceBudgetTx(tx *sql.Tx, sched *Scheduler, step *db.Step, nowMS int64) error { - if sched.budgetHeadroom(step) { +// +// `cost` is the amount THIS claim would accrue — the step's reservable cost on +// an ordinary claim, or the dispatcher-declared scaled cost when the claim +// carries a `--cost-multiplier` (DKT-867). It is a parameter rather than +// re-derived from the step because the step row cannot know which variant the +// dispatcher resolved. +func enforceBudgetTx(tx *sql.Tx, sched *Scheduler, step *db.Step, cost float64, nowMS int64) error { + if sched.budgetAdmits(step, cost) { return nil } diff --git a/internal/engine/claim.go b/internal/engine/claim.go index 1d70646b..20bdd96e 100644 --- a/internal/engine/claim.go +++ b/internal/engine/claim.go @@ -40,7 +40,41 @@ type ClaimOptions struct { // TTLOverride is an explicit `--ttl`. Zero means resolve from the // workflow's [limits] then config, per §6.4's precedence. TTLOverride int64 - NowMS int64 + // Metadata is `--metadata`, the opaque KV bag the DISPATCHER knows at + // claim time, merged onto the step's own in the claim's own transaction + // (docs/tdd/completion-metadata.md §1.7, DKT-592). + // + // It exists because the facts a dispatcher knows when it hands a step out + // are exactly the facts a step that never completes would otherwise never + // record. Deferring them to `complete` means the steps that failed or + // crashed — the ones an operator most wants to characterize — are the only + // steps with no bag at all, and a rollup over those keys goes blind at + // precisely the rows that motivated it. + // + // Core reads no key here, as everywhere: the bag is stored and never + // interpreted. + Metadata string + // CostMultiplier is `--cost-multiplier`: the DISPATCHER's declaration that + // this claim will cost this many times the step's declared `expected_cost` + // (DKT-867). Zero means unset — the claim accrues the declared cost, + // byte-for-byte the pre-DKT-867 behavior. + // + // It exists because `expected_cost` is variant-blind: the definition + // declares one number per step template, and the party that resolves a + // step to an executor variant — routing an escalated retry to something + // far pricier — is the dispatcher, at claim time. The multiplier is that + // resolution priced in the cap's own currency, declared by the only party + // in a position to know it. Core stores, sums, and reserves the number and + // never interprets it (genericity.md): what varied and by how much is the + // dispatcher's business, exactly as `--usage`'s units are the claimant's. + // + // The scaled cost participates in the SAME facts the declared cost always + // has: it is checked against the cap in the claim's own transaction and + // recorded in the claim's `step-claimed` event, so each attempt of a step + // accrues what ITS claim declared — an escalated re-claim accrues more + // than the cheap first attempt did (B9's re-accrual, made variant-aware). + CostMultiplier float64 + NowMS int64 } // ClaimStep takes a lease on a step and returns the token AND the context @@ -94,6 +128,67 @@ func (e *Engine) ClaimStepWithGates( return claimStepWithGates(conn, stepID, opts, e) } +// ClaimStepRendered is ClaimStepWithGates for `step claim --render`: the claim +// and the packet render as ONE saga (DKT-804). +// +// The render used to run strictly AFTER the claim's commit, and a render-time +// refusal — an unpinned packet file, a template that does not parse — landed on +// a step already `claimed` with no token ever delivered. RUN-56 stranded eight +// steps this way in one dispatch: the lease was held, every recovery verb +// requires the token nobody received, and the only exits were the full TTL or +// an operator-gated `step reap`. +// +// So the render is VALIDATED FIRST, pre-claim, on the same rule §1.7 places the +// metadata-bag validation before `Begin()`: a refusal must write NOTHING — +// no lease, no attempt, no event — and leave the step exactly as claimable as +// it was. The render cannot simply move inside the claim's transaction, because +// it reads through the connection while the claim's transaction would hold the +// pool's single connection (the same deadlock loadTTLConfig's placement +// avoids). +// +// The AUTHORITATIVE render then runs post-claim, so the returned packet reports +// the step as its claimant will see it — `context.step` claimed, this claim's +// `attempt`, the merged metadata bag. Every failure the preflight can catch is +// deterministic over the run's pins and the template's bytes, so a post-claim +// refusal requires the filesystem to change between two reads milliseconds +// apart; that residual race keeps today's disposition — the error surfaces and +// the lease stands — because rolling the claim back is not a rollback at all: +// the claim is COMMITTED, and the `step-claimed` event it wrote is the budget +// floor's accrual (§4.3, a SUM over those events), so un-taking the lease means +// un-writing a ledger entry the log is append-only about. What the refusal does +// instead is tell its caller how to end the lease NOW — `step reap` is the +// relay's channel for exactly this (DKT-83, DKT-820) — so the cost is one verb +// rather than the full TTL. Every deterministic failure still refuses before +// anything is written. +func (e *Engine) ClaimStepRendered( + conn *sql.DB, stepID int, opts ClaimOptions, templatePath, executor string, +) (*ClaimResult, *RenderResult, error) { + if _, err := RenderStepAs(conn, stepID, templatePath, executor, opts.NowMS); err != nil { + return nil, nil, err + } + + result, err := claimStepWithGates(conn, stepID, opts, e) + if err != nil { + return nil, nil, err + } + + packet, err := RenderStepAs(conn, stepID, templatePath, executor, opts.NowMS) + if err != nil { + // The one refusal that can still leave a lease standing, and the + // caller is the only party who can act on it: the token went nowhere, + // so every token-bearing verb is closed to it, and the relay that + // spawned this claim is exactly the party `step reap` is the channel + // for (DKT-83). Naming the remedy here is what saves it the TTL — + // DKT-820's RUN-59 executors had the verb and did not know it applied. + return nil, nil, fmt.Errorf( + "%w; the lease is held and no token was issued — run "+ + "`docket step reap %s --reason ...` to return the step to the "+ + "pool without waiting out the lease", + err, model.FormatStepID(stepID)) + } + return result, packet, nil +} + func claimStepWithGates( conn *sql.DB, stepID int, opts ClaimOptions, e *Engine, ) (*ClaimResult, error) { @@ -105,6 +200,32 @@ func claimStepWithGates( return nil, err } + // ---- The claim bag's validation, PRE-TRANSACTION (§1.7). --------------- + // + // The cap and the shape check sit here, before `conn.Begin()`, for the + // reason §6.9 places `complete`'s there: a refusal writes NOTHING and + // leaves `row_version` where it was. On this path that property has an + // extra edge — the transaction below performs the LAZY REAP, so a + // validation that ran after `Begin()` would let a malformed bag consume an + // attempt on its way to being refused. + // + // Same order as stage zero's, and for C5's reason: an operator debugging + // one input must not get a different message depending on which check + // happened to run first. + if err := validateClaimMetadataSize(opts.Metadata); err != nil { + return nil, err + } + if _, err := DecodeMetadataBag(opts.Metadata, "metadata"); err != nil { + return nil, err + } + // The multiplier's validation sits with the bag's, for the same reason: a + // refusal must write nothing, and this path's transaction performs the + // lazy reap, so a malformed input validated any later would consume an + // attempt on its way to being refused (DKT-867). + if err := validateCostMultiplier(opts.CostMultiplier); err != nil { + return nil, err + } + defs, err := StepDefinitions(conn, step.RunID) if err != nil { return nil, err @@ -189,13 +310,36 @@ func claimStepWithGates( fresh.Instance, fresh.Kind, unclaimableReason(fresh.Kind)) } + // ---- The claim's EFFECTIVE cost (DKT-867). ------------------------------ + // + // What this claim will accrue: the step's reservable cost, scaled by the + // dispatcher's `--cost-multiplier` when one was declared. The offer (R7) + // necessarily reserved the DECLARED cost — `next` cannot know which + // variant a dispatcher will resolve a step to — so the claim, which is the + // engine's resolve moment, is where the actual cost becomes knowable and + // is therefore where it is enforced and accrued. + claimCost := reservableCost(fresh) + if opts.CostMultiplier > 0 { + claimCost = fresh.ExpectedCost * opts.CostMultiplier + } + // ---- R8: claim enforces readiness ITSELF. ------------------------------- // // It does not trust that the caller ran `next`. A dispatcher racing a scope // conflict would otherwise claim a step `next` would never have offered — // and the refusal names WHICH condition failed, so a stalled dispatcher can // diagnose itself instead of retrying blind. - if ready, cond := sched.Ready(fresh); !ready { + ready, cond := sched.Ready(fresh) + if !ready && cond == CondBudget && sched.budgetAdmits(fresh, claimCost) { + // The DECLARED cost would cross the cap but the cost this claim + // actually accrues does not — a dispatcher routing to a CHEAPER + // variant (`--cost-multiplier` under 1). Budget is the LAST clause of + // §6.3's conjunction, so CondBudget means every other condition held, + // and admitting on the real cost is the same arithmetic the refusal + // below runs, answered with the honest number (DKT-867). + ready, cond = true, "" + } + if !ready { // ---- B14/B15/B20: the budget refusal is not an ordinary one. -------- // // R7 IS the claim-time check — the same arithmetic over the same @@ -215,7 +359,7 @@ func claimStepWithGates( // rolling that back because a later clause refused would make the reap // depend on which step happened to be claimed next. if cond == CondBudget { - refusal := enforceBudgetTx(tx, sched, fresh, opts.NowMS) + refusal := enforceBudgetTx(tx, sched, fresh, claimCost, opts.NowMS) if err := tx.Commit(); err != nil { return nil, fmt.Errorf("committing the budget pause: %w", err) } @@ -235,6 +379,22 @@ func claimStepWithGates( "step %s is not ready to claim: %s", fresh.Instance, cond) } + // ---- The ESCALATED claim's own budget check (DKT-867). ------------------ + // + // Readiness above admitted the DECLARED cost; a dispatcher that declared a + // multiplier above 1 is about to accrue more than that, and the cap must + // see the real number BEFORE the accrual commits — this is precisely the + // hop DKT-867 observed sailing past the cap: an escalated fix attempt + // priced at the cap as though it were the cheap variant. Same arithmetic, + // same snapshot, same pause-and-refuse contract as the CondBudget branch. + if opts.CostMultiplier > 0 && !sched.budgetAdmits(fresh, claimCost) { + refusal := enforceBudgetTx(tx, sched, fresh, claimCost, opts.NowMS) + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("committing the budget pause: %w", err) + } + return nil, refusal + } + // The TTL is resolved from the ALREADY-LOADED config (read before the // transaction opened), never by querying the pool from in here: the // connection pool is capped at one connection, so a pool read while this @@ -258,10 +418,56 @@ func claimStepWithGates( if err := db.SetStepStatusTx(tx, fresh.ID, db.StepClaimed, opts.NowMS, opts.NowMS); err != nil { return nil, err } - if err := recordEvent(tx, eventRecord{ + + // ---- The dispatcher's bag lands WITH THE CLAIM (§1.7, DKT-592). -------- + // + // In transaction A, with the CAS that awarded the claim, and before the + // event that records the claim happened — the same ordering stage zero + // uses for the completion bag, and for the same reason: the step's own row + // reaches its final shape for this transition before the transition is + // recorded as having occurred. A crash between them rolls back both. + // + // This is what makes the bag survive every non-completing end. The reap + // (`ReapStepTx`) and the failure paths write status, lease and counters + // and never touch `metadata`, so a step that crashes after this commit + // still carries what the dispatcher knew when it handed the step out. + // + // It MERGES rather than assigns, on the same last-write-wins-per-key rule + // as every other writer of this column: a definition-side bag survives, a + // re-claim overlays the previous attempt's, and the completion bag later + // overlays this one. `mergeMetadata` cannot fail on caller input here — + // the shape was decoded before the transaction opened — so the only error + // left is a serialization one. + if opts.Metadata != "" { + merged, err := mergeMetadata(fresh.Metadata, opts.Metadata) + if err != nil { + return nil, err + } + if err := db.SetStepMetadataTx(tx, fresh.ID, merged, opts.NowMS); err != nil { + return nil, err + } + // The snapshot carries the merge forward so THIS claim's own context + // bundle reports the bag it just recorded: a worker reading + // `context.metadata` sees what it was dispatched with, not the state + // before its own claim. + fresh.Metadata = merged + } + + // The claim event IS the accrual (§4.3), so a claim whose dispatcher + // declared a variant scaling records the SCALED cost on the event itself + // (DKT-867): RunFloorTx sums `data.expected_cost` when present and the + // step row's declaration otherwise. The resulting number is recorded + // rather than the multiplier, so the ledger entry is self-contained — what + // this claim accrued does not change if the step row is ever different. + // An unscaled claim writes exactly the event it always has. + claimEvent := eventRecord{ Kind: EventStepClaimed, RunID: fresh.RunID, Instance: fresh.Instance, IssueID: fresh.IssueID, - }); err != nil { + } + if opts.CostMultiplier > 0 { + claimEvent.Data = fmt.Sprintf(`{"expected_cost":%g}`, claimCost) + } + if err := recordEvent(tx, claimEvent); err != nil { return nil, err } diff --git a/internal/engine/context.go b/internal/engine/context.go index ca992e4e..6d235f56 100644 --- a/internal/engine/context.go +++ b/internal/engine/context.go @@ -90,6 +90,16 @@ type Context struct { // makes easy. Read from `steps.routing`, which holds the CURRENT routing // record for this step alone. // + // TWO WRITERS PUT A RULING HERE, and together they are the ONLY reliable + // operator-steering channels into a packet (DKT-725): `step resolve --as + // retry|rerun-gates -m` leaves the note on the SAME row it re-executes, + // and `step resolve --as fix-round -m` stamps it onto the NEW round's + // instantiated rows (stampEntryRouting), because the parked row it was + // issued against is superseded and renders nothing again. Issue COMMENTS + // are not a context source at all — §6.6's five-source rule, by design — + // and a mid-run `description` edit never renders either, because + // `body_snapshot` froze at activation (§9 item 5). + // // nil when the step carries no routing record, so a first-round packet is // byte-identical to what it always was. Resolution *ContextResolution `json:"resolution,omitempty"` @@ -240,7 +250,9 @@ func AssembleContext( } // Sources 2 and 3: the snapshots. NOTHING below reads the `issues` table. - issue, err := contextIssue(tx, step.RunID, step.IssueID) + // `linked` is the activation-pinned cross-issue artifact ids (DKT-547), + // carried in the issue snapshot and read back with it. + issue, linked, err := contextIssue(tx, step.RunID, step.IssueID) if err != nil { return nil, err } @@ -257,7 +269,7 @@ func AssembleContext( return nil, err } } - inputs, err := resolveInputs(tx, sched, step, spec, issue.BodySnapshot, artifacts) + inputs, err := resolveInputs(tx, sched, step, spec, issue.BodySnapshot, artifacts, linked) if err != nil { return nil, err } @@ -405,7 +417,13 @@ func resolveTarget(inputs []ContextInput) (sha, worktree string) { // `id` is the run_issues key — the one field that is not snapshot-derived, and // it cannot drift because an issue's id never changes. Every other field comes // from `issue_snapshot`, which activation froze (§5.1.1). -func contextIssue(tx *sql.Tx, runID, issueID int) (*ContextIssue, error) { +// +// The second return is the snapshot's `linked` pin set (DKT-547): declared +// `issue.linked..` suffix -> artifact ids, as activation +// resolved and froze them. It rides the same read because it lives in the same +// column; it is not a member of the wire-shape ContextIssue, because §11.4's +// consumers read artifacts through `inputs`, never through `issue`. +func contextIssue(tx *sql.Tx, runID, issueID int) (*ContextIssue, map[string][]int, error) { var ( body sql.NullString snapshot sql.NullString @@ -415,11 +433,11 @@ func contextIssue(tx *sql.Tx, runID, issueID int) (*ContextIssue, error) { WHERE run_id = ? AND issue_id = ?`, runID, issueID, ).Scan(&body, &snapshot) if err == sql.ErrNoRows { - return nil, fmt.Errorf("issue %s is not in run %s", + return nil, nil, fmt.Errorf("issue %s is not in run %s", model.FormatID(issueID), model.FormatRunID(runID)) } if err != nil { - return nil, fmt.Errorf("reading the issue snapshot: %w", err) + return nil, nil, fmt.Errorf("reading the issue snapshot: %w", err) } out := &ContextIssue{ @@ -429,15 +447,17 @@ func contextIssue(tx *sql.Tx, runID, issueID int) (*ContextIssue, error) { Scope: []string{}, } + var linked map[string][]int if snapshot.String != "" { var frozen struct { - Title string `json:"title"` - Kind string `json:"kind"` - Labels []string `json:"labels"` - Scope []string `json:"scope"` + Title string `json:"title"` + Kind string `json:"kind"` + Labels []string `json:"labels"` + Scope []string `json:"scope"` + Linked map[string][]int `json:"linked"` } if err := json.Unmarshal([]byte(snapshot.String), &frozen); err != nil { - return nil, fmt.Errorf("reading the issue snapshot for %s: %w", + return nil, nil, fmt.Errorf("reading the issue snapshot for %s: %w", model.FormatID(issueID), err) } out.Title, out.Kind = frozen.Title, frozen.Kind @@ -447,8 +467,9 @@ func contextIssue(tx *sql.Tx, runID, issueID int) (*ContextIssue, error) { if frozen.Scope != nil { out.Scope = frozen.Scope } + linked = frozen.Linked } - return out, nil + return out, linked, nil } // contextPins reads the pin LIST — paths and hashes. It never opens a pinned @@ -632,9 +653,11 @@ func loopEntryOf(step *db.Step) *LoopEntry { // says so for fanout joins. "Ordered by artifact id" is trivially satisfiable // by accident when insertion order happens to match, so the ordering here is // applied explicitly and a test shuffles insertion order to prove it. +// `linked` is the issue snapshot's activation-pinned cross-issue artifact ids +// (DKT-547), consulted only for `issue.linked..` entries. func resolveInputs( tx *sql.Tx, sched *Scheduler, step *db.Step, spec *workflow.Step, - bodySnapshot string, artifacts []*db.Artifact, + bodySnapshot string, artifacts []*db.Artifact, linked map[string][]int, ) ([]ContextInput, error) { if len(spec.Inputs) == 0 { return []ContextInput{}, nil @@ -667,6 +690,22 @@ func resolveInputs( continue } + // `issue.linked..` (DKT-547): an artifact recorded + // under ANOTHER issue, resolved through this issue's relations and + // pinned by artifact id at activation — assembly only reads the + // pinned rows back, never the live cross-issue question (§6.6). + // Checked BEFORE the producer-addressed forms below: the whole + // declaration is an engine namespace, and `splitInput` would + // otherwise read `issue.linked.` as a producer step name. + if _, _, ok := workflow.LinkedInput(declared); ok { + resolved, err := resolveLinkedPinned(tx, step, linked, declared) + if err != nil { + return nil, err + } + out = append(out, resolved...) + continue + } + // `.gate-results` (DKT-77): the producer's RECORDED gate // results, addressable as an input. Before this form existed, no // syntax exposed what the engine had already recorded — so every @@ -682,6 +721,22 @@ func resolveInputs( continue } + // `.vote-record` (DKT-545): the named vote step's RECORDED + // proposal — tally outcome, casts, and rationales — addressable as an + // input. The gate-results reasoning applied to the vote machinery: + // before this form existed, what a panel actually said reached no + // downstream step, so a revise loop entered on concerns had to shell + // out to `docket vote show` to learn what to revise. + if stepName, kind, ok := splitInput(declared); ok && + kind == workflow.VoteRecordKind { + resolved, err := resolveVoteRecords(tx, sched, step, stepName) + if err != nil { + return nil, err + } + out = append(out, resolved...) + continue + } + // `issue.latest.` (DKT-492): the issue's latest recorded round // of this kind, whoever produced it. Engine-produced like the issue // forms — no producer step is named, so it resolves here rather than @@ -736,6 +791,11 @@ func ResolveInputArtifacts( } def := sched.defs[step.WorkflowID] + // The activation-pinned cross-issue ids (DKT-547), read lazily: only a + // definition declaring the form pays the snapshot read. + var linked map[string][]int + linkedLoaded := false + var out []*db.Artifact for _, declared := range spec.Inputs { if declared == inputIssueBody || declared == inputIssueDiff { @@ -745,6 +805,21 @@ func ResolveInputArtifacts( out = append(out, resolveLatestOfKind(artifacts, producers, step, kind)...) continue } + if _, _, ok := workflow.LinkedInput(declared); ok { + if !linkedLoaded { + _, linked, err = contextIssue(tx, step.RunID, step.IssueID) + if err != nil { + return nil, err + } + linkedLoaded = true + } + matched, err := linkedArtifacts(tx, step, linked, declared) + if err != nil { + return nil, err + } + out = append(out, matched...) + continue + } matched, err := resolveDeclaredInput(artifacts, producers, def, step, declared) if err != nil { return nil, err @@ -920,7 +995,24 @@ func latestPerProducer(matched []*db.Artifact) []*db.Artifact { // KIND at exactly this ordinal — the redirect matches by kind, not by // name, because that is the only thing the two producers share: `fix` // does not know it is standing in for `implement`, it only knows it -// `emits = "change-summary"` too. +// `emits = "change-summary"` too; +// 5. the DEFINITION names exactly ONE loop body the kind match could mean — +// one `loop = true` step that emits the kind AND whose own `after_loop` +// chain (the closure of its declared re-entry root) contains the +// consuming step. Kind is all conditions 1–4 look at, so when SEVERAL +// bodies of the kind re-enter this consumer the match cannot tell which +// body stands in for which named producer, and it must not guess: it +// refuses, and ordinalScoped's answer stands — the named step's own +// earlier artifact, or nothing at all for a step that never ran. RUN-39 +// measured the guess (DKT-591): spec-doc's review declared six per-lane +// `.doc` inputs, one revise body per lane all emitting `doc`, +// and the one body that ran satisfied every entry's kind scan — five +// SKIPPED authors' inputs each resolved to a step none of them named, +// six byte-identical copies per judge. The per-body `after_loop` chain +// is what keeps DKT-544's serves-scoped clusters redirecting: two +// same-kind bodies whose clusters re-enter DISJOINT chains (`prd-fix` +// at `prd-gate`, `design-fix` at `design-gate`) are unambiguous for any +// one consumer, because only one body's chain contains it. // // A LOOP BODY'S OWN INPUTS THEREFORE NEVER REDIRECT, doubly so (DKT-492). // Condition 1 excludes the body by construction — a `loop = true` step cannot @@ -955,6 +1047,32 @@ func loopProducerRedirect( return nil } + // Condition 5: resolve WHICH body the kind match could mean from the + // definition alone. Run state cannot answer this — the body that happened + // to record at this ordinal is exactly the wrong oracle when several + // bodies share the kind and only one ran (DKT-591) — so the candidate set + // is definitional: bodies of the kind whose own `after_loop` chain + // contains this consumer. Anything but exactly one candidate is a refusal. + var body *workflow.Step + for _, s := range def.Steps { + if !s.Loop || s.AfterLoop == "" { + continue + } + if kind != "*" && workflow.ArtifactKind(s) != kind { + continue + } + if !downstreamClosure(def, []string{s.AfterLoop})[step.StepName] { + continue + } + if body != nil { + return nil + } + body = s + } + if body == nil { + return nil + } + var out []*db.Artifact for _, a := range artifacts { if kind != "*" && a.Kind != kind { @@ -967,8 +1085,7 @@ func loopProducerRedirect( if !recordedProducer(producer.Status) || producer.Ordinal != step.Ordinal { continue } - body := workflow.StepByName(def, producer.StepName) - if body == nil || !body.Loop { + if producer.StepName != body.Name { continue } out = append(out, a) diff --git a/internal/engine/dispatch.go b/internal/engine/dispatch.go index 08773d17..c8c7bbe0 100644 --- a/internal/engine/dispatch.go +++ b/internal/engine/dispatch.go @@ -84,6 +84,14 @@ type StaleTarget struct { TargetSHA string `json:"target_sha"` SharedHead string `json:"shared_head"` Reason string `json:"reason"` + // Absent distinguishes DKT-742's harder case machine-readably: the + // recorded target does not resolve as a commit object from the shared + // checkout AT ALL (`git cat-file -e ^{commit}` fails), as opposed to + // resolving but sitting off HEAD's history. A consumer planning a wave + // reads it as "no seat can reconstruct this tree", not merely "confirm + // the tree before reviewing". `omitempty` keeps every divergence-shaped + // advisory byte-identical to what it always was. + Absent bool `json:"absent,omitempty"` } // FormatDispatchID renders a dispatch's display identity. @@ -883,8 +891,36 @@ func consumesIssueDiff(spec *workflow.Step) bool { // that is provably not an ancestor of the shared HEAD is stale — the // conductor integrated something other than the recorded commit, so a packet // rendered from it reviews a tree the branch no longer carries. An -// unanswerable question (missing git, GC'd object, no ancestry seam) warns -// about nothing: absence of evidence is not staleness. +// unanswerable question (missing git, no ancestry seam) warns about nothing: +// absence of evidence is not staleness. +// +// AN ABSENT OBJECT IS EVIDENCE, NOT ABSENCE OF IT (DKT-742). IsAncestorFn's +// `known = false` covers both "git could not answer" and "the object is not +// in the shared store at all", and skipping both silently was exactly +// backwards for the second: a target that cannot even be resolved is the +// WORST state a packet can render from, not the safest. RUN-52's DKT-V253 +// dispatched a three-seat vote whose packet named a sha `git cat-file -t` +// found in no checkout — each seat discovered that independently, mid-wave, +// with no warning having fired. So an unanswerable ancestry now asks the +// narrower question the seats asked (ObjectExistsFn, `git cat-file -e +// ^{commit}`): a DEFINITIVE "no such object" warns with its own reason; +// anything short of definitive stays silent exactly as before. The recorded +// sha itself is never rewritten — the round record is the producer's frozen +// evidence, and DKT-725/DKT-741 both settled that a recorded reference is +// surfaced when it goes bad, not silently re-derived. +// +// AN ADJUDICATED WARNING DOES NOT RE-FIRE (DKT-742's companion half). An +// operator who investigated a warning and ruled it acceptable records a +// run-scoped waiver for the (step instance, target sha) pair — the +// gate_override_grants shape (DKT-546) applied to this advisory — and rows +// matching a waiver are dropped from the answer. The signature is the pair +// alone: the same pair re-fires nothing however often HEAD moves, while a +// different sha on the same row (RUN-52's DISPATCH-301, target rotated to +// 7e072f54) or the same sha on an unnamed row is a fresh question and still +// warns. The filter is READ-ONLY because this judge serves `dispatch verify`, +// which writes nothing by contract; the audit trail is the waiver's own +// `stale-target-waived` event at recording time. A waiver read that fails +// leaves every warning standing — the safe direction for an advisory. // // ANCESTRY ALONE IS NOT THE TEST (DKT-424). The sanctioned integration flow // cherry-picks an isolated worktree's commit onto the shared branch, which @@ -926,11 +962,29 @@ func (e *Engine) staleTargets( if sharedHead == "" { return nil } + // The run's standing waivers (DKT-742), loaded lazily: only a judge about + // to warn pays the read, and a failed read waives nothing. + var waivers []db.StaleTargetWaiver + waiversLoaded := false + waived := func(instance, sha string) bool { + if !waiversLoaded { + waivers, _ = db.StaleTargetWaiversForRun(conn, runID) + waiversLoaded = true + } + for _, w := range waivers { + if waiverCovers(w, instance, sha) { + return true + } + } + return false + } + // One verdict per SHA, not per row: a review fanout's siblings all render // from the same recorded target, and the git questions are identical. type verdict struct { ancestor, known bool treeMatch, treeKnown bool + absent bool } cache := make(map[string]verdict, len(candidates)) var out []StaleTarget @@ -941,26 +995,56 @@ func (e *Engine) staleTargets( if v.known && !v.ancestor && e.TreeMatchFn != nil { v.treeMatch, v.treeKnown = e.TreeMatchFn(execRoot, c.sha) } + if !v.known && e.ObjectExistsFn != nil { + // DKT-742: the ancestry question was unanswerable — ask the + // narrower one a packet consumer will ask anyway. Only a + // definitive "no such object" accuses; every genuinely + // unanswerable state stays silent. + exists, existsKnown := e.ObjectExistsFn(execRoot, c.sha) + v.absent = existsKnown && !exists + } cache[c.sha] = v } - if !v.known || v.ancestor { + if v.known && v.ancestor { + continue + } + if !v.known && !v.absent { continue } - if v.treeKnown && v.treeMatch { + if v.known && v.treeKnown && v.treeMatch { // DKT-424: the sha was rewritten (cherry-pick integration), the // tree was not. The branch carries exactly what the packet // renders on the paths this work touched — nothing to warn about. continue } + if waived(c.instance, c.sha) { + // DKT-742: this exact (step, target) pair was adjudicated by an + // operator; the standing ruling is engine-visible now. + continue + } + reason := staleTargetReason(c.sha, sharedHead, v.treeKnown) + if v.absent { + reason = staleTargetAbsentReason(c.sha, sharedHead) + } out = append(out, StaleTarget{ Instance: c.instance, Issue: c.issue, TargetSHA: c.sha, SharedHead: sharedHead, - Reason: staleTargetReason(c.sha, sharedHead, v.treeKnown), + Reason: reason, Absent: v.absent, }) } return out } +// waiverCovers reports whether one waiver covers one would-be warning: the +// SAME step instance, and a target sha the waiver's recorded sha is a +// case-insensitive prefix of (>= 7 hex chars, enforced at the verb). Prefix +// rather than equality because the advisory the operator copies the sha from +// renders it at 12 characters; case-insensitive because git itself is. +func waiverCovers(w db.StaleTargetWaiver, instance, sha string) bool { + return w.StepInstance == instance && + strings.HasPrefix(strings.ToLower(sha), strings.ToLower(w.TargetSHA)) +} + // staleTargetReason renders the advisory's sentence, in the two shapes the // evidence actually supports (DKT-424). // @@ -986,11 +1070,38 @@ func staleTargetReason(sha, sharedHead string, treeChecked bool) string { "divergence; if integration did diverge, a packet rendered from " + "this sha reviews a tree the branch no longer carries" } - return evidence + - ". A claim does not re-derive the target from HEAD: the packet renders " + - "from whichever recorded diff artifact resolves at claim time, so this " + - "row stays on this sha unless an upstream step records a newer diff for " + - "the issue first" + return evidence + staleTargetClaimSemantics +} + +// staleTargetClaimSemantics is DKT-415's claim-time tail, shared by every +// advisory shape: the decision the warning informs is "is dispatching through +// this safe", and that turns on what a claim will actually render. +const staleTargetClaimSemantics = ". A claim does not re-derive the target " + + "from HEAD: the packet renders from whichever recorded diff artifact " + + "resolves at claim time, so this row stays on this sha unless an upstream " + + "step records a newer diff for the issue first" + +// staleTargetAbsentReason renders DKT-742's advisory shape: the recorded +// target does not resolve as a commit object from the shared checkout at all. +// +// It is a THIRD sentence rather than a variant of staleTargetReason's two, +// because the conductor's next move differs again: a divergence is compared, +// an unanswered tree is hand-checked, but an absent object can be NEITHER — +// no seat can `git cat-file` or `git archive` this sha from the shared +// checkout, so the packet's target is unusable as recorded and the work needs +// its integrated successor (or the change-summary artifact) judged instead, +// which is exactly the workaround RUN-52's three vote seats each re-derived +// alone. +func staleTargetAbsentReason(sha, sharedHead string) string { + return fmt.Sprintf( + "the recorded target sha %.12s does not resolve as a commit from the "+ + "shared checkout at all (`git cat-file -e %.12s^{commit}` fails "+ + "against HEAD %.12s's checkout) — the producing worktree's commit "+ + "never reached the shared object store, or was pruned after "+ + "integration removed its worktree and branch. No consumer can "+ + "reconstruct this tree from the shared checkout; judge the step's "+ + "integrated successor or its recorded change-summary instead", + sha, sha, sharedHead) + staleTargetClaimSemantics } // noOpenDispatchErr maps the storage sentinel onto the taxonomy once. diff --git a/internal/engine/dispatch_reconcile.go b/internal/engine/dispatch_reconcile.go new file mode 100644 index 00000000..d8c0e709 --- /dev/null +++ b/internal/engine/dispatch_reconcile.go @@ -0,0 +1,161 @@ +package engine + +import ( + "database/sql" + "fmt" + + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// One-verb wave reconcile (DKT-580). +// +// A conductor closing a wave typed the same pipeline every time — back-fill +// what the wave measured, verify the manifest still matches, close it — once +// per wave, five times in harness RUN-44 alone. Nothing about that pipeline is +// a decision: the ORDER is forced (usage must land before the close's probe +// reads it, and a close over a drifted manifest is the thing verify exists to +// stop), and every retype is a chance to name the wrong journal, pipe a stale +// file, or skip the verify entirely — the three ways a ledger silently starts +// lying. +// +// THIS ADDS NO NEW SEMANTICS. It is the same three engine calls in the same +// order, and each stage is the EXACT function the standalone verb calls — +// BackfillUsage, VerifyDispatch, CloseDispatch — never a re-implementation. +// That is what makes the discrepancy probe see a reconciled back-fill exactly +// as it sees a hand-typed one: `usage_recorded` is set by the same +// MarkStepUsageRecordedTx inside the same BackfillUsage transaction, and +// `missingUsage` reads that column. There is no second codepath that could +// drift from the first. +// +// WHAT "ATOMIC" MEANS HERE, precisely, because it is not one transaction and +// cannot be: +// +// - Each stage is individually all-or-nothing. BackfillUsage is one +// transaction over the whole batch; VerifyDispatch writes nothing at all +// and is rolled back by construction; CloseDispatch is one transaction. +// - A stage that fails STOPS THE SEQUENCE. Verify never runs on a failed +// back-fill, close never runs on a failed verify, and the run is left in +// whatever state the last SUCCESSFUL stage produced — never half-closed. +// - Every failure NAMES ITS STAGE (StageError), so an operator reading a +// refusal knows which of the three verbs to reach for. +// +// The three could not share one transaction even if that were wanted: +// VerifyDispatch's stale-target pass shells out to git and must therefore run +// with no transaction open (§6, no subprocess inside a transaction), and its +// read-only-by-rollback property is the mechanism that makes it a verify at +// all. Nesting it inside a writing transaction would destroy the very property +// P11 is. +const ( + // StageBackfill is stage 1: the usage rows the wave measured. + StageBackfill = "backfill" + // StageVerify is stage 2: the manifest still renders as it was stored. + StageVerify = "verify" + // StageClose is stage 3: the manifest is reconciled and retired. + StageClose = "close" +) + +// StageError names WHICH stage of a reconcile refused. +// +// It wraps rather than replaces: Unwrap reaches the stage's own error, so +// CodeOf still finds the taxonomy code the standalone verb would have exited +// with and errors.Is still reaches any db sentinel underneath. A reconcile +// therefore exits with the SAME code for the same failure — the message gains +// a stage name and nothing else changes. +type StageError struct { + // Stage is one of StageBackfill, StageVerify, StageClose. + Stage string + // Err is the stage's own refusal, unmodified. + Err error + // Verify and Mismatch are set ONLY on a verify-stage row mismatch, which + // is the one refusal a caller cannot render from the error text alone: + // P9's diagnostic is the differing BYTES of both renderings plus DKT-243's + // per-row verdict block, and those live on the result, not in a message. + // Carrying them here is what lets the reconcile path print the identical + // refusal `dispatch verify` prints rather than a summary of it. + Verify *VerifyResult + Mismatch *RowMismatch +} + +func (e *StageError) Error() string { + return fmt.Sprintf("%s stage: %s", e.Stage, e.Err) +} + +func (e *StageError) Unwrap() error { return e.Err } + +// ReconcileOutcome is what all three stages did, in the order they ran. +// +// A stage that never ran is absent (`omitempty`), so a reader can tell "verify +// passed with nothing to say" from "verify never happened" — the distinction +// the per-stage errors exist to preserve. +type ReconcileOutcome struct { + Run string `json:"run"` + Backfill *BackfillOutcome `json:"backfill,omitempty"` + Verify *VerifyResult `json:"verify,omitempty"` + Close *CloseOutcome `json:"close,omitempty"` +} + +// ReconcileDispatch performs back-fill, then verify, then close, refusing at +// whichever stage fails (DKT-580). +// +// ON FAILURE IT RETURNS BOTH the partial outcome and a *StageError. The +// outcome is deliberately not discarded: after a verify-stage refusal the +// back-fill HAS landed and is a fact the operator needs — re-running the whole +// pipeline would hit the ledger's (step, attempt, unit) key on the rows that +// are already in — so a caller that reports the partial work reports the +// truth. The error is always non-nil when a stage failed; nothing about the +// partial outcome should be read as success. +// +// AN EMPTY BATCH IS REFUSED, at the back-fill stage, by BackfillUsage's own +// rule. That is the intended behaviour rather than a skip-to-verify: a wave +// whose usage journal produced no rows is a measurement that went missing, and +// closing quietly over it is exactly the silent lie this verb exists to make +// harder. +func (e *Engine) ReconcileDispatch( + conn *sql.DB, runID int, rows []BackfillRow, source string, + onDuplicate string, acceptMissingUsage bool, nowMS int64, +) (*ReconcileOutcome, error) { + out := &ReconcileOutcome{Run: model.FormatRunID(runID)} + + // ---- stage 1: back-fill ------------------------------------------------- + // + // THE SAME CALL `dispatch backfill-usage` MAKES. Criterion 3 holds by + // construction rather than by testing: there is one BackfillUsage, it + // writes the ledger rows and sets `usage_recorded` in one transaction, and + // the discrepancy probe stage 3 runs reads that column. Nothing here + // reimplements any of it. + backfill, err := e.BackfillUsage(conn, runID, rows, source, onDuplicate, nowMS) + if err != nil { + return out, &StageError{Stage: StageBackfill, Err: err} + } + out.Backfill = backfill + + // ---- stage 2: verify ---------------------------------------------------- + // + // It runs AFTER the back-fill, which is the order the manual pipeline used + // and the only order that makes sense: the back-fill writes usage rows and + // marks steps, neither of which touches the ready set, so it cannot + // invalidate a verify — but a verify placed first would be stale by the + // time the close ran, which is precisely the window this verb closes. + verify, mismatch, err := e.VerifyDispatch(conn, runID, nowMS) + if err != nil { + return out, &StageError{Stage: StageVerify, Err: err} + } + if mismatch != nil { + return out, &StageError{ + Stage: StageVerify, Verify: verify, Mismatch: mismatch, + Err: conflictErr( + "%s does not match its current rendering (manifest row %d)", + verify.Dispatch, mismatch.Position), + } + } + out.Verify = verify + + // ---- stage 3: close ----------------------------------------------------- + closed, err := e.CloseDispatch(conn, runID, acceptMissingUsage, nowMS) + if err != nil { + return out, &StageError{Stage: StageClose, Err: err} + } + out.Close = closed + + return out, nil +} diff --git a/internal/engine/dispatch_reconcile_test.go b/internal/engine/dispatch_reconcile_test.go new file mode 100644 index 00000000..bd5fa3e3 --- /dev/null +++ b/internal/engine/dispatch_reconcile_test.go @@ -0,0 +1,293 @@ +package engine + +import ( + "errors" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-580's one-verb wave reconcile. +// +// RUN-44 retyped back-fill / verify / close five times, once per wave. The +// tests below are the three acceptance criteria, plus the two orderings the +// word "atomically" actually forbids: a failed back-fill must not verify or +// close, and a failed verify must not close. + +// TestReconcileRunsAllThreeStages is criterion 1's happy path: one call +// back-fills, verifies, and closes, and every stage reports what it did. +func TestReconcileRunsAllThreeStages(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + + openDispatch(t, conn, runID, 0, nowMS) + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + + // The premise: without the back-fill this close REFUSES. If it did not, + // the test below would prove nothing about the back-fill stage running. + if _, err := e.CloseDispatch(conn, runID, false, nowMS); err == nil { + t.Fatal("premise: `close` must refuse over the missing usage that " + + "the back-fill stage exists to record") + } + + out, err := e.ReconcileDispatch(conn, runID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 48211}, + }, "wave-journal:wf-7", "", false, nowMS) + testsupport.Must(t, err, "reconcile: %v", err) + + if out.Backfill == nil || out.Backfill.Written != 1 || out.Backfill.Steps != 1 { + t.Fatalf("backfill stage reported %+v, want 1 row across 1 step", out.Backfill) + } + if out.Backfill.Source != "wave-journal:wf-7" { + t.Errorf("source = %q, want the one passed — a reconcile must not "+ + "rewrite who measured the work", out.Backfill.Source) + } + if out.Verify == nil || out.Verify.Dispatch == "" { + t.Fatalf("verify stage reported %+v, want a named dispatch", out.Verify) + } + if out.Close == nil || out.Close.Status != db.DispatchClosed { + t.Fatalf("close stage reported %+v, want status %q", out.Close, db.DispatchClosed) + } + + // And the manifest really is closed, not merely reported as such. + status, reason := dispatchStatus(t, conn, runID) + if status != db.DispatchClosed || reason != db.CloseReasonReconciled { + t.Fatalf("stored dispatch is (%s, %s), want (%s, %s)", + status, reason, db.DispatchClosed, db.CloseReasonReconciled) + } +} + +// TestReconcileMatchesTheManualOrdering is criterion 3, asserted the only way +// that means anything: run the SAME wave twice, once through the three verbs by +// hand and once through the reconcile, and compare what the discrepancy probe +// sees afterwards. +// +// The ledger row, its source, its attempt, and the `usage_recorded` fast path +// the probe actually reads must be identical, and `next` — which runs the probe +// — must answer in both. +func TestReconcileMatchesTheManualOrdering(t *testing.T) { + // The manual ordering, in its own database. + manualConn := mustDB(t) + manualEngine := testEngine() + manualRun := dispatchRun(t, manualConn) + openDispatch(t, manualConn, manualRun, 0, nowMS) + manualStep := stepIDByInstance(t, manualConn, "implement@0") + completeWithoutUsage(t, manualConn, manualEngine, manualStep) + + _, err := manualEngine.BackfillUsage(manualConn, manualRun, []BackfillRow{ + {Step: manualStep, Unit: "tokens", Quantity: 48211}, + }, "wave-journal:wf-7", "", nowMS) + testsupport.Must(t, err, "manual backfill-usage: %v", err) + _, mismatch, err := manualEngine.VerifyDispatch(manualConn, manualRun, nowMS) + testsupport.Must(t, err, "manual verify: %v", err) + if mismatch != nil { + t.Fatalf("manual verify found a mismatch at row %d", mismatch.Position) + } + _, err = manualEngine.CloseDispatch(manualConn, manualRun, false, nowMS) + testsupport.Must(t, err, "manual close: %v", err) + + // The reconcile, in another. + oneConn := mustDB(t) + oneEngine := testEngine() + oneRun := dispatchRun(t, oneConn) + openDispatch(t, oneConn, oneRun, 0, nowMS) + oneStep := stepIDByInstance(t, oneConn, "implement@0") + completeWithoutUsage(t, oneConn, oneEngine, oneStep) + + _, err = oneEngine.ReconcileDispatch(oneConn, oneRun, []BackfillRow{ + {Step: oneStep, Unit: "tokens", Quantity: 48211}, + }, "wave-journal:wf-7", "", false, nowMS) + testsupport.Must(t, err, "reconcile: %v", err) + + // What the ledger holds, row for row. + manualRows := usageRowsFor(t, manualConn, manualStep) + oneRows := usageRowsFor(t, oneConn, oneStep) + if len(manualRows) != 1 || len(oneRows) != 1 { + t.Fatalf("ledger rows: manual %d, reconciled %d, want 1 each", + len(manualRows), len(oneRows)) + } + m, o := manualRows[0], oneRows[0] + if m.Unit != o.Unit || m.Quantity != o.Quantity || + m.Attempt != o.Attempt || m.Source != o.Source { + t.Fatalf("the reconciled row differs from the hand-typed one:\n"+ + " manual: %+v\n reconciled: %+v", m, o) + } + + // And what the PROBE reads — the fast-path column, not the rows. + manualStepRow, err := db.GetStep(manualConn, manualStep) + testsupport.Must(t, err, "GetStep: %v", err) + oneStepRow, err := db.GetStep(oneConn, oneStep) + testsupport.Must(t, err, "GetStep: %v", err) + if manualStepRow.UsageRecorded != oneStepRow.UsageRecorded { + t.Fatalf("usage_recorded: manual %v, reconciled %v — D2's probe reads "+ + "this column and the two orderings must leave it identical", + manualStepRow.UsageRecorded, oneStepRow.UsageRecorded) + } + if !oneStepRow.UsageRecorded { + t.Fatal("usage_recorded is false after a reconcile; the probe would " + + "keep refusing") + } + + // The probe itself, run end to end, on both. + if _, err := manualEngine.NextSteps(manualConn, manualRun, 0, nowMS); err != nil { + t.Fatalf("`next` refuses after the manual ordering: %v", err) + } + if _, err := oneEngine.NextSteps(oneConn, oneRun, 0, nowMS); err != nil { + t.Fatalf("`next` refuses after the reconcile but not after the manual "+ + "ordering: %v", err) + } +} + +// TestReconcileStopsAtAFailedBackfill is the first half of "atomically": a +// back-fill that refuses runs no verify and no close, and leaves the run in the +// state it found it — dispatch still open, ledger still empty. +func TestReconcileStopsAtAFailedBackfill(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + + openDispatch(t, conn, runID, 0, nowMS) + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + + // A batch naming a step that does not exist. BackfillUsage resolves every + // step before writing anything, so this refuses with nothing written. + out, err := e.ReconcileDispatch(conn, runID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 48211}, + {Step: 999999, Unit: "tokens", Quantity: 1}, + }, "", "", false, nowMS) + if err == nil { + t.Fatal("the reconcile succeeded over a step that does not exist") + } + + var stage *StageError + if !errors.As(err, &stage) { + t.Fatalf("error %v is not a *StageError; a reconcile failure must name "+ + "its stage", err) + } + if stage.Stage != StageBackfill { + t.Fatalf("stage = %q, want %q", stage.Stage, StageBackfill) + } + if !strings.Contains(err.Error(), StageBackfill+" stage") { + t.Errorf("the message %q does not name the stage that failed", err) + } + if code, ok := CodeOf(err); !ok || code != CodeNotFound { + t.Errorf("code = %q (found=%v), want %q — StageError must unwrap to "+ + "the stage's own taxonomy code", code, ok, CodeNotFound) + } + + // NO LATER STAGE RAN. + if out.Backfill != nil || out.Verify != nil || out.Close != nil { + t.Fatalf("a failed back-fill reported later stages: %+v", out) + } + // THE RUN IS AS IT WAS. Nothing written, nothing closed. + if rows := usageRowsFor(t, conn, implID); len(rows) != 0 { + t.Fatalf("the refused batch still wrote %d ledger row(s): %+v", + len(rows), rows) + } + if status, _ := dispatchStatus(t, conn, runID); status != db.DispatchOpen { + t.Fatalf("the dispatch is %q after a failed back-fill, want %q — a "+ + "failed stage must never leave a partial close", status, db.DispatchOpen) + } +} + +// TestReconcileStopsAtAFailedVerify is the second half: a verify that finds a +// drifted manifest runs no close. The back-fill it already committed STAYS +// committed — it is a measurement, not a lock — but the dispatch is untouched. +func TestReconcileStopsAtAFailedVerify(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + + manifest := openDispatch(t, conn, runID, 0, nowMS) + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + + // Drift the manifest under the reconcile: claim a still-ready row, so it is + // non-terminal and no longer offerable — verify's `genuinely-missing`. + var claimed int + for _, row := range manifest.Rows { + id, err := model.ParseStepID(row.Step) + testsupport.Must(t, err, "parsing %q: %v", row.Step, err) + if id == implID { + continue + } + if _, err := ClaimStep(conn, id, ClaimOptions{Owner: "w2", NowMS: nowMS}); err == nil { + claimed = id + break + } + } + if claimed == 0 { + t.Skip("the fixture manifest has no second claimable row to drift") + } + + out, err := e.ReconcileDispatch(conn, runID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 48211}, + }, "", "", false, nowMS) + if err == nil { + t.Fatal("the reconcile closed over a drifted manifest") + } + + var stage *StageError + if !errors.As(err, &stage) { + t.Fatalf("error %v is not a *StageError", err) + } + if stage.Stage != StageVerify { + t.Fatalf("stage = %q, want %q", stage.Stage, StageVerify) + } + if stage.Mismatch == nil || stage.Verify == nil { + t.Fatalf("a verify-stage refusal carries no diagnostic: %+v", stage) + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("code = %q (found=%v), want %q", code, ok, CodeConflict) + } + + // THE CLOSE DID NOT RUN. + if out.Close != nil { + t.Fatalf("the close stage reported %+v after a failed verify", out.Close) + } + if status, _ := dispatchStatus(t, conn, runID); status != db.DispatchOpen { + t.Fatalf("the dispatch is %q after a failed verify, want %q", + status, db.DispatchOpen) + } + // The back-fill that already landed is reported rather than hidden: an + // operator who re-runs the pipeline would otherwise hit the ledger's + // (step, attempt, unit) key with no idea why. + if out.Backfill == nil || out.Backfill.Written != 1 { + t.Fatalf("the committed back-fill is not reported: %+v", out.Backfill) + } + if rows := usageRowsFor(t, conn, implID); len(rows) != 1 { + t.Fatalf("the back-fill wrote %d ledger row(s), want 1 — the stage "+ + "committed before verify refused", len(rows)) + } +} + +// TestReconcileLeavesTheStandaloneVerbsAlone is criterion 2 at the engine +// level: the three functions the reconcile calls are the SAME ones the verbs +// call, so a plain CloseDispatch on a plain run behaves exactly as before. +func TestReconcileLeavesTheStandaloneVerbsAlone(t *testing.T) { + conn := mustDB(t) + e := testEngine() + runID := dispatchRun(t, conn) + + openDispatch(t, conn, runID, 0, nowMS) + implID := stepIDByInstance(t, conn, "implement@0") + completeWithoutUsage(t, conn, e, implID) + + _, err := e.BackfillUsage(conn, runID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 7}, + }, "", "", nowMS) + testsupport.Must(t, err, "backfill-usage: %v", err) + + outcome, err := e.CloseDispatch(conn, runID, false, nowMS) + testsupport.Must(t, err, "close: %v", err) + if outcome.Status != db.DispatchClosed || outcome.Reason != db.CloseReasonReconciled { + t.Fatalf("close outcome = %+v, want (%s, %s)", + outcome, db.DispatchClosed, db.CloseReasonReconciled) + } +} diff --git a/internal/engine/dispatch_stale_test.go b/internal/engine/dispatch_stale_test.go index efebcb26..0ab682fc 100644 --- a/internal/engine/dispatch_stale_test.go +++ b/internal/engine/dispatch_stale_test.go @@ -29,6 +29,10 @@ func staleFixture(t *testing.T, conn *sql.DB, e *Engine) { // left to shell out to whatever checkout the test process happens to sit // in: these cases are about the ancestry verdict alone. e.TreeMatchFn = func(string, string) (match, known bool) { return false, false } + // Existence unanswerable by default, for the tree probe's exact reason: + // these cases are about the ancestry verdict alone, and the DKT-742 + // absence probe accuses only on a DEFINITIVE "no such object". + e.ObjectExistsFn = func(string, string) (exists, known bool) { return false, false } implementID := stepIDByInstance(t, conn, "implement@0") claim, err := ClaimStep(conn, implementID, ClaimOptions{Owner: "w", NowMS: nowMS}) testsupport.Must(t, err, "claim implement: %v", err) @@ -111,8 +115,10 @@ func TestDispatchStaysQuietWhenTargetIsAncestor(t *testing.T) { } } -// The unanswerable case must not warn: git absent, a GC'd object, a tree that -// is not a repository. Absence of evidence is not staleness. +// The unanswerable case must not warn: git absent, a tree that is not a +// repository. Absence of evidence is not staleness — but only while it stays +// ABSENCE of evidence: a definitive "no such object" is evidence, and the +// DKT-742 cases below prove it warns. func TestDispatchStaysQuietWhenAncestryUnknown(t *testing.T) { conn := mustDB(t) run, _ := activatedRun(t, conn) diff --git a/internal/engine/dkt470_test.go b/internal/engine/dkt470_test.go index afde6d32..6a851b55 100644 --- a/internal/engine/dkt470_test.go +++ b/internal/engine/dkt470_test.go @@ -111,10 +111,11 @@ func TestOverridePassSkipsInterposedTargetsEmptyWithoutAThreshold(t *testing.T) } // TestOverridePassStillBypassesTheThreshold pins CURRENT behavior after the -// fix: override-pass records a generic `pass` and the interposed vote is -// still skipped — the fix is the warning (above), not a threshold -// evaluation. If a future change teaches override-pass to evaluate the -// threshold instead, this test is the one to revisit. +// fix: an ACKNOWLEDGED override-pass (DKT-861's --drop-interposed) records a +// generic `pass` and the interposed vote is still skipped — the fix is the +// warning (above) and the pre-mutation refusal (dkt861_test.go), not a +// threshold evaluation. If a future change teaches override-pass to evaluate +// the threshold instead, this test is the one to revisit. func TestOverridePassStillBypassesTheThreshold(t *testing.T) { conn := mustDB(t) e := testEngine() @@ -126,7 +127,8 @@ func TestOverridePassStillBypassesTheThreshold(t *testing.T) { stepID := parkVerifyWaitingHuman(t, conn, e) - testsupport.Must(t, e.ResolveStep(conn, stepID, ResolveOverridePass, "accepted", nowMS), + testsupport.Must(t, e.ResolveStepDropInterposed( + conn, stepID, ResolveOverridePass, "accepted", false, nowMS), "override-pass: %v", nil) if got := stepRouting(t, conn, "verify@0"); got != RoutingPass { @@ -157,7 +159,8 @@ func TestUnroutedVoteAfterRetryReportsBlockedReason(t *testing.T) { testsupport.Must(t, err, "activate: %v", err) stepID := parkVerifyWaitingHuman(t, conn, e) - testsupport.Must(t, e.ResolveStep(conn, stepID, ResolveOverridePass, "accepted", nowMS), + testsupport.Must(t, e.ResolveStepDropInterposed( + conn, stepID, ResolveOverridePass, "accepted", false, nowMS), "override-pass: %v", nil) if got := stepStatus(t, conn, "tribunal@0"); got != db.StepSkipped { t.Fatalf("premise: tribunal@0 = %q, want %q", got, db.StepSkipped) diff --git a/internal/engine/dkt545_test.go b/internal/engine/dkt545_test.go new file mode 100644 index 00000000..d4165c8a --- /dev/null +++ b/internal/engine/dkt545_test.go @@ -0,0 +1,412 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-545: concern-aware routing on vote steps, and the `.vote-record` +// input form. Every read-gate tally in the corpus APPROVED and none was clean +// (DKT-V34 passed 2-1 over a security dissent; DKT-V140/V160 each passed with +// two approve-with-concerns casts), and a workflow had no way to act on it: +// only the binary tally reached routing, and the concern text lived where no +// input form could address it. + +// concernLoopSrc: a vote gate whose APPROVED-with-concerns tally routes +// `fix-loop`, with a loop body to instantiate — the revise loop a rejection +// already entered, now reachable by a concerned approval. +const concernLoopSrc = ` +[pipeline] +name = "concern-loop" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "seed" +after = [] +executor = "x" +emits = "findings" + +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["seat-a", "seat-b", "seat-c"] +vote_rule = "majority" +on_fail = "waiting-human" +threshold = { "fix-loop" = "count>=2(vote == approve-with-concerns)" } + +[[step]] +name = "fix" +executor = "x" +emits = "findings" +loop = true +after_loop = "gate" +inputs = ["gate.vote-record"] +` + +// openGateProposal drives a run up to its vote step's open proposal: seed +// completes, the lifecycle drive opens the ballot, and the proposal id comes +// back for the seats to cast against. +func openGateProposal(t *testing.T, conn *sql.DB, e *Engine, runID int) int { + t.Helper() + claimAndComplete(t, conn, e, "seed@0", "the findings", "") + err := e.DriveRunLifecycles(conn, runID, nowMS) + testsupport.Must(t, err, "driving after the record: %v", err) + + gate, err := db.GetStep(conn, stepIDByInstance(t, conn, "gate@0")) + testsupport.Must(t, err, "reading gate@0: %v", err) + proposalID, err := findVoteProposal(conn, gate) + testsupport.Must(t, err, "finding gate@0's proposal: %v", err) + if proposalID == 0 { + t.Fatal("no proposal opened for gate@0") + } + return proposalID +} + +// castSeat casts one verdict with the weights every seat in these tests +// shares, so a unanimous approval (concerned or clean) tallies above the 0.5 +// rule and the routing under test is genuinely the threshold's, never the +// tally's. +func castSeat(t *testing.T, conn *sql.DB, proposalID int, seat string, verdict model.Verdict, summary string) { + t.Helper() + _, err := db.CastVote(conn, &model.Vote{ + ProposalID: proposalID, VoterName: seat, Verdict: verdict, + Confidence: 0.9, DomainRelevance: 0.8, Summary: summary, + }) + testsupport.Must(t, err, "CastVote(%s): %v", seat, err) +} + +// stepCount reports how many step rows hold an instance — the non-fatal +// sibling of stepIDByInstance, for asserting a step was NOT created. +func stepCount(t *testing.T, conn *sql.DB, instance string) int { + t.Helper() + var n int + err := conn.QueryRow( + `SELECT COUNT(*) FROM steps WHERE instance = ?`, instance).Scan(&n) + testsupport.Must(t, err, "counting %s: %v", instance, err) + return n +} + +// TestConcernedApprovalRoutesFixLoop is the issue's central scenario: an +// APPROVED tally carried by two approve-with-concerns casts enters the same +// revise loop a rejection does — counter, loop body, routing record naming +// the matched predicate — instead of the concerns evaporating. +func TestConcernedApprovalRoutesFixLoop(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + registerSource(t, conn, []byte(concernLoopSrc), "concern-loop.toml") + issue := createIssue(t, conn, "concerned", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + castSeat(t, conn, proposalID, "seat-a", model.VerdictApproveWithConcerns, "auth check is thin") + castSeat(t, conn, proposalID, "seat-b", model.VerdictApproveWithConcerns, "no rollback path") + castSeat(t, conn, proposalID, "seat-c", model.VerdictApprove, "") + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the concerned approval: %v", err) + + // The tally itself APPROVED — the routing under test is the threshold's. + proposal, err := db.GetProposal(conn, proposalID) + testsupport.Must(t, err, "GetProposal: %v", err) + if proposal.Status != model.ProposalStatusApproved { + t.Fatalf("proposal status = %q, want approved — the fixture must "+ + "exercise the threshold, not the tally", proposal.Status) + } + + if got := loopCount(t, conn, run.ID, issue); got != 1 { + t.Errorf("loop_count = %d after a concerned approval matched the "+ + "fix-loop threshold, want 1", got) + } + stepIDByInstance(t, conn, "fix@1") // fatals if the fix step was never created + + gate, err := db.GetStep(conn, stepIDByInstance(t, conn, "gate@0")) + testsupport.Must(t, err, "re-reading gate@0: %v", err) + if !strings.HasPrefix(gate.Routing, workflow.OnFailFixLoop) { + t.Errorf("gate@0 routing = %q, want a %q routing", gate.Routing, workflow.OnFailFixLoop) + } + if !strings.Contains(gate.Routing, "approve-with-concerns") { + t.Errorf("gate@0 routing record %q does not name the matched predicate", gate.Routing) + } +} + +// TestCleanApprovalPassesConcernThreshold: three clean approvals do not match +// the concern predicate, so the step routes pass exactly as an un-thresholded +// vote step does — no loop entry, no fix step. +func TestCleanApprovalPassesConcernThreshold(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + registerSource(t, conn, []byte(concernLoopSrc), "concern-loop.toml") + issue := createIssue(t, conn, "clean", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + for _, seat := range []string{"seat-a", "seat-b", "seat-c"} { + castSeat(t, conn, proposalID, seat, model.VerdictApprove, "") + } + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the clean approval: %v", err) + + if got := stepStatus(t, conn, "gate@0"); got != db.StepDone { + t.Errorf("gate@0 = %q after a clean approval, want %q", got, db.StepDone) + } + gate, err := db.GetStep(conn, stepIDByInstance(t, conn, "gate@0")) + testsupport.Must(t, err, "reading gate@0: %v", err) + if !strings.HasPrefix(gate.Routing, RoutingPass) { + t.Errorf("gate@0 routing = %q, want a %q routing", gate.Routing, RoutingPass) + } + if got := loopCount(t, conn, run.ID, issue); got != 0 { + t.Errorf("loop_count = %d after a clean approval, want 0", got) + } + if n := stepCount(t, conn, "fix@1"); n != 0 { + t.Errorf("%d fix@1 step(s) exist after a clean approval, want none", n) + } +} + +// TestRejectedTallyIgnoresConcernThreshold: a REJECTED tally still routes per +// `on_fail`, threshold or no threshold — the threshold asks "was the approval +// clean", which is not a question about a rejection. +func TestRejectedTallyIgnoresConcernThreshold(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + registerSource(t, conn, []byte(concernLoopSrc), "concern-loop.toml") + issue := createIssue(t, conn, "rejected", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + for _, seat := range []string{"seat-a", "seat-b", "seat-c"} { + castSeat(t, conn, proposalID, seat, model.VerdictReject, "no") + } + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the rejection: %v", err) + + // The declared on_fail is waiting-human; the fix-loop THRESHOLD must not + // capture a rejection. + if got := stepStatus(t, conn, "gate@0"); got != db.StepWaitingHuman { + t.Errorf("gate@0 = %q after a rejected tally, want %q — a rejection "+ + "routes per on_fail, never through the threshold", got, db.StepWaitingHuman) + } + if got := loopCount(t, conn, run.ID, issue); got != 0 { + t.Errorf("loop_count = %d after a rejected tally with on_fail=waiting-human, want 0", got) + } +} + +// TestConcernedApprovalWithoutThresholdPasses is backward compatibility: the +// exact cast set that routes fix-loop under a threshold routes PASS on a step +// declaring none — every pre-existing vote step behaves as it always did. +func TestConcernedApprovalWithoutThresholdPasses(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + src := strings.Replace(concernLoopSrc, + "threshold = { \"fix-loop\" = \"count>=2(vote == approve-with-concerns)\" }\n", "", 1) + if src == concernLoopSrc { + t.Fatal("fixture bug: the threshold line was not removed") + } + registerSource(t, conn, []byte(src), "no-threshold.toml") + issue := createIssue(t, conn, "unthresholded", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + castSeat(t, conn, proposalID, "seat-a", model.VerdictApproveWithConcerns, "worry one") + castSeat(t, conn, proposalID, "seat-b", model.VerdictApproveWithConcerns, "worry two") + castSeat(t, conn, proposalID, "seat-c", model.VerdictApprove, "") + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the concerned approval: %v", err) + + if got := stepStatus(t, conn, "gate@0"); got != db.StepDone { + t.Errorf("gate@0 = %q on a step with no threshold, want %q", got, db.StepDone) + } + if got := loopCount(t, conn, run.ID, issue); got != 0 { + t.Errorf("loop_count = %d on a step with no threshold, want 0", got) + } +} + +// TestConcernThresholdWaitingHumanLeg: the waiting-human routing parks the +// step for an operator, with the matched predicate in the routing record. +func TestConcernThresholdWaitingHumanLeg(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + src := strings.Replace(concernLoopSrc, + `threshold = { "fix-loop" = "count>=2(vote == approve-with-concerns)" }`, + `threshold = { "waiting-human" = "any(verdict == approve-with-concerns)" }`, 1) + if src == concernLoopSrc { + t.Fatal("fixture bug: the threshold line was not replaced") + } + registerSource(t, conn, []byte(src), "concern-park.toml") + issue := createIssue(t, conn, "parked", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + castSeat(t, conn, proposalID, "seat-a", model.VerdictApproveWithConcerns, "one worry") + castSeat(t, conn, proposalID, "seat-b", model.VerdictApprove, "") + castSeat(t, conn, proposalID, "seat-c", model.VerdictApprove, "") + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the concerned approval: %v", err) + + if got := stepStatus(t, conn, "gate@0"); got != db.StepWaitingHuman { + t.Errorf("gate@0 = %q, want %q", got, db.StepWaitingHuman) + } + gate, err := db.GetStep(conn, stepIDByInstance(t, conn, "gate@0")) + testsupport.Must(t, err, "reading gate@0: %v", err) + if !strings.HasPrefix(gate.Routing, workflow.OnFailWaitingHuman) { + t.Errorf("gate@0 routing = %q, want a %q routing", gate.Routing, workflow.OnFailWaitingHuman) + } + if !strings.Contains(gate.Routing, "approve-with-concerns") { + t.Errorf("gate@0 routing record %q does not name the matched predicate", gate.Routing) + } + if got := loopCount(t, conn, run.ID, issue); got != 0 { + t.Errorf("loop_count = %d for a waiting-human routing, want 0", got) + } +} + +// voteRecordSrc: a downstream executor reading the vote step's record through +// declared inputs — the piece that lets a consumer see WHAT the panel said +// without shelling out to `docket vote show`. +const voteRecordSrc = ` +[pipeline] +name = "vote-record-wf" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "seed" +after = [] +executor = "x" +emits = "findings" + +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["seat-a", "seat-b"] +vote_rule = "majority" +on_fail = "skip" + +[[step]] +name = "report" +after = ["gate"] +executor = "reporter" +emits = "record" +inputs = ["gate.vote-record"] +` + +// TestVoteRecordResolvesAsAnInput: the record arrives in the consumer's +// context bundle — tally outcome, score, and every cast with its rationale. +func TestVoteRecordResolvesAsAnInput(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.5", "") + registerSource(t, conn, []byte(voteRecordSrc), "vote-record-wf.toml") + issue := createIssue(t, conn, "read the record", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + proposalID := openGateProposal(t, conn, e, run.ID) + castSeat(t, conn, proposalID, "seat-a", model.VerdictApproveWithConcerns, + "tighten the auth check") + castSeat(t, conn, proposalID, "seat-b", model.VerdictApprove, "ship it") + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the approval: %v", err) + + claim, err := ClaimStep(conn, stepIDByInstance(t, conn, "report@0"), + ClaimOptions{Owner: "reporter", NowMS: nowMS}) + testsupport.Must(t, err, "claim report@0: %v", err) + + var body string + for _, input := range claim.Context.Inputs { + if input.Kind == workflow.VoteRecordKind { + body = input.Body + if input.ProducerStep != "gate@0" { + t.Errorf("producer = %q, want gate@0", input.ProducerStep) + } + } + } + if body == "" { + t.Fatal("report's context carries no vote-record input") + } + + var record struct { + Proposal string `json:"proposal"` + Status string `json:"status"` + WeightedScore *float64 `json:"weighted_score"` + Casts []struct { + Voter string `json:"voter"` + Verdict string `json:"verdict"` + Rationale string `json:"rationale"` + } `json:"casts"` + } + testsupport.Must(t, json.Unmarshal([]byte(body), &record), "parsing the record: %v", err) + + if record.Proposal != model.FormatProposalID(proposalID) { + t.Errorf("record.proposal = %q, want %q", record.Proposal, model.FormatProposalID(proposalID)) + } + if record.Status != string(model.ProposalStatusApproved) { + t.Errorf("record.status = %q, want approved", record.Status) + } + if record.WeightedScore == nil { + t.Error("record carries no weighted_score for a tallied proposal") + } + if len(record.Casts) != 2 { + t.Fatalf("record carries %d casts, want 2: %s", len(record.Casts), body) + } + byVoter := map[string]struct{ verdict, rationale string }{} + for _, c := range record.Casts { + byVoter[c.Voter] = struct{ verdict, rationale string }{c.Verdict, c.Rationale} + } + if got := byVoter["seat-a"]; got.verdict != string(model.VerdictApproveWithConcerns) || + got.rationale != "tighten the auth check" { + t.Errorf("seat-a's cast = %+v, want its concerned verdict and rationale", got) + } + if got := byVoter["seat-b"]; got.verdict != string(model.VerdictApprove) || + got.rationale != "ship it" { + t.Errorf("seat-b's cast = %+v, want its approval and rationale", got) + } +} + +// TestVoteCastPayloadKeysMatchTheValidator pins the drift V36 and the payload +// builder must not develop: the keys the engine builds are EXACTLY the fields +// the validator admits, both read from workflow.VoteCastFields. +func TestVoteCastPayloadKeysMatchTheValidator(t *testing.T) { + payloads := voteCastPayloads([]*model.Vote{{ + VoterName: "seat-a", Verdict: model.VerdictApproveWithConcerns, + }}) + if len(payloads) != 1 { + t.Fatalf("%d payloads for one cast", len(payloads)) + } + if len(payloads[0]) != len(workflow.VoteCastFields) { + t.Errorf("cast payload has %d keys, validator admits %d fields", + len(payloads[0]), len(workflow.VoteCastFields)) + } + for _, field := range workflow.VoteCastFields { + if _, ok := payloads[0][field]; !ok { + t.Errorf("cast payload is missing validated field %q", field) + } + } +} diff --git a/internal/engine/dkt547_test.go b/internal/engine/dkt547_test.go new file mode 100644 index 00000000..7ab02c5f --- /dev/null +++ b/internal/engine/dkt547_test.go @@ -0,0 +1,417 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-547 — the `issue.linked..` input form: a step consuming +// an artifact recorded under ANOTHER issue, reached through the consuming +// issue's declared relations, pinned at activation, and enforced loudly. +// +// The incident: ui-change@12 claimed its changes were "bound to an accepted +// ux-spec" produced by a spec-doc run on a different issue, and no input form +// reached it — every legal form is same-run and same-issue — so the spec +// reached executors only when the issue body happened to cite it (33 design-qa +// instances across 3 runs, all relying on prose). These tests pin the whole +// contract: resolution through relations, activation-time pinning, the loud +// refusals when the relation or the artifact is missing, and the pin's +// immunity to artifacts recorded after activation. + +// specDocMiniSrc is the producing side, minimized: one step on a `spike` +// issue that drafts and records the ux-spec. +const specDocMiniSrc = ` +[pipeline] +name = "spec-doc-mini" +version = 1 + +[match] +kind = ["spike"] + +[[step]] +name = "draft-spec" +after = [] +executor = "author" +emits = "ux-spec" +inputs = ["issue.body"] +` + +// uiChangeMiniSrc is the consuming side: design-qa on a `task` issue reads +// the accepted spec of the issue(s) this issue depends on. No step of this +// workflow produces `ux-spec` — the producer is another issue's run, which is +// exactly the relaxation V11 grants the form. +const uiChangeMiniSrc = ` +[pipeline] +name = "ui-change-mini" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "design-qa" +after = [] +executor = "qa" +emits = "qa-report" +inputs = ["issue.body", "issue.linked.depends_on.ux-spec"] +` + +// blockedUiMiniSrc consumes through the INVERSE token: the spec issue BLOCKS +// this one, so the consumer reads its `blocked-by` issues' spec — the same +// binding declared from the relation's other end, hyphenated to prove the +// spelling normalizes. +const blockedUiMiniSrc = ` +[pipeline] +name = "blocked-ui-mini" +version = 1 + +[match] +kind = ["chore"] + +[[step]] +name = "design-qa-b" +after = [] +executor = "qa" +emits = "qa-report" +inputs = ["issue.linked.blocked-by.ux-spec"] +` + +const specDocV1 = "the accepted ux spec, v1" + +// produceSpec drives a spec-doc-mini run over a fresh spike issue to done, +// leaving the issue holding one recorded `ux-spec` artifact, and returns the +// issue id. +func produceSpec(t *testing.T, conn *sql.DB, e *Engine, title, doc string) int { + t.Helper() + issue := createIssue(t, conn, title, "the spec request", "spike", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activating the spec run: %v", err) + claimAndCompleteInRun(t, conn, e, run.ID, "draft-spec@0", doc) + return issue +} + +// claimAndCompleteInRun is claimAndComplete scoped to one run's instance +// namespace — these tests hold several runs whose step instances collide +// (every spec run has a `draft-spec@0`), so the global instance lookup the +// shared helper uses would be ambiguous. +func claimAndCompleteInRun( + t *testing.T, conn *sql.DB, e *Engine, runID int, instance, artifact string, +) { + t.Helper() + var stepID int + err := conn.QueryRow( + `SELECT id FROM steps WHERE run_id = ? AND instance = ?`, + runID, instance).Scan(&stepID) + testsupport.Must(t, err, "finding %s in run %d: %v", instance, runID, err) + + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim %s: %v", instance, err) + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte(artifact), NowMS: nowMS, + }) + testsupport.Must(t, err, "complete %s: %v", instance, err) +} + +// linkIssues records one relation through the ordinary db path. +func linkIssues(t *testing.T, conn *sql.DB, source, target int, rt model.RelationType) { + t.Helper() + _, err := db.CreateRelation(conn, &model.Relation{ + SourceIssueID: source, TargetIssueID: target, RelationType: rt, + }) + testsupport.Must(t, err, "linking %d -> %d (%s): %v", source, target, rt, err) +} + +// specInputs filters a bundle's inputs down to the `ux-spec` entries. +func specInputs(inputs []ContextInput) []ContextInput { + var out []ContextInput + for _, in := range inputs { + if in.Kind == "ux-spec" { + out = append(out, in) + } + } + return out +} + +// TestLinkedInputPinsAndServesTheDoc is the happy path end to end: the spec +// run records the doc under its own issue; the consumer links `depends_on` +// and activates; design-qa's bundle carries the doc as a declared input, with +// the producer named as `/` and the artifact id the spec +// run's ledger attributes to it. +func TestLinkedInputPinsAndServesTheDoc(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(specDocMiniSrc), "spec-doc-mini.toml") + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + specIssue := produceSpec(t, conn, e, "write the spec", specDocV1) + + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + linkIssues(t, conn, consumer, specIssue, model.RelationDependsOn) + runB := startRun(t, conn, consumer) + _, err := activate(conn, runB.ID) + testsupport.Must(t, err, "activating the consumer run: %v", err) + + // The pin is in the issue snapshot — the same column every other frozen + // fact of the issue lives in. + var snapshot string + err = conn.QueryRow( + `SELECT issue_snapshot FROM run_issues WHERE run_id = ? AND issue_id = ?`, + runB.ID, consumer).Scan(&snapshot) + testsupport.Must(t, err, "reading the consumer snapshot: %v", err) + if !strings.Contains(snapshot, `"linked"`) || + !strings.Contains(snapshot, `"depends_on.ux-spec"`) { + t.Errorf("the issue snapshot carries no linked pin: %s", snapshot) + } + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "design-qa@0"), nowMS) + testsupport.Must(t, err, "assembling design-qa@0's bundle: %v", err) + + specs := specInputs(bundle.Inputs) + if len(specs) != 1 { + t.Fatalf("design-qa@0 binds %d ux-spec inputs, want 1: %+v", len(specs), specs) + } + if specs[0].Body != specDocV1 { + t.Errorf("the bound spec is %q, want %q", specs[0].Body, specDocV1) + } + wantProducer := model.FormatID(specIssue) + "/draft-spec@0" + if specs[0].ProducerStep != wantProducer { + t.Errorf("producer = %q, want %q — the cross-issue provenance", + specs[0].ProducerStep, wantProducer) + } + var artifactID int + err = conn.QueryRow( + `SELECT a.id FROM artifacts a JOIN steps s ON s.id = a.step_id + WHERE s.issue_id = ? AND a.kind = 'ux-spec'`, specIssue).Scan(&artifactID) + testsupport.Must(t, err, "reading the spec artifact id: %v", err) + if want := fmt.Sprintf("ARTIFACT-%d", artifactID); specs[0].Artifact != want { + t.Errorf("artifact = %q, want %q", specs[0].Artifact, want) + } +} + +// TestLinkedInputActivationFailsWithoutRelation is the first loud refusal: the +// workflow declares the binding and the issue never linked anything, so +// activation refuses inside the fat transaction and writes nothing — the +// binding is enforced rather than an issue-body convention. +func TestLinkedInputActivationFailsWithoutRelation(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + run := startRun(t, conn, consumer) + + _, err := activate(conn, run.ID) + if err == nil { + t.Fatal("activation succeeded with no depends_on relation to resolve") + } + for _, want := range []string{ + model.FormatID(consumer), "issue.linked.depends_on.ux-spec", + "no depends_on relation", "docket issue link", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q: %v", want, err) + } + } + + // The fat transaction rolled back whole: no steps, no snapshot. + var steps int + err = conn.QueryRow( + `SELECT COUNT(*) FROM steps WHERE run_id = ?`, run.ID).Scan(&steps) + testsupport.Must(t, err, "counting steps: %v", err) + if steps != 0 { + t.Errorf("the refused activation left %d steps behind", steps) + } +} + +// TestLinkedInputActivationFailsWithoutDoc is the second refusal: the relation +// exists but the linked issue has recorded no artifact of the kind — the spec +// issue's run has not produced it yet, so the consumer cannot bind and must +// not start. +func TestLinkedInputActivationFailsWithoutDoc(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + specIssue := createIssue(t, conn, "write the spec", "the spec request", "spike", nil) + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + linkIssues(t, conn, consumer, specIssue, model.RelationDependsOn) + run := startRun(t, conn, consumer) + + _, err := activate(conn, run.ID) + if err == nil { + t.Fatal("activation succeeded with no ux-spec artifact on the linked issue") + } + for _, want := range []string{ + model.FormatID(consumer), model.FormatID(specIssue), + "issue.linked.depends_on.ux-spec", + `has a recorded artifact of kind "ux-spec"`, + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q: %v", want, err) + } + } +} + +// TestLinkedInputPinsTheLatestAndOnlyTheLatest pins both halves of "latest, +// deterministically, at activation". Before activation, the linked issue holds +// two recorded spec artifacts and the pin takes the newest (highest id — +// latestPerProducer's own rule). After activation, a THIRD artifact lands on +// the linked issue and the bundle must not move: the pin froze at activation +// exactly as every other input does, so a mid-run spec revision cannot change +// what the run's steps were dispatched against. +func TestLinkedInputPinsTheLatestAndOnlyTheLatest(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(specDocMiniSrc), "spec-doc-mini.toml") + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + specIssue := produceSpec(t, conn, e, "write the spec", specDocV1) + + // A superseding emit from the same producer — the DKT-103 multi-emit + // shape — leaves the issue holding v1 and v2, v2 newest. + const specDocV2 = "the accepted ux spec, v2" + recordExtraSpec(t, conn, specIssue, specDocV2) + + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + linkIssues(t, conn, consumer, specIssue, model.RelationDependsOn) + runB := startRun(t, conn, consumer) + _, err := activate(conn, runB.ID) + testsupport.Must(t, err, "activating the consumer run: %v", err) + + // A revision recorded AFTER activation must not reach the bundle. + recordExtraSpec(t, conn, specIssue, "the ux spec, revised mid-run") + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "design-qa@0"), nowMS) + testsupport.Must(t, err, "assembling design-qa@0's bundle: %v", err) + + specs := specInputs(bundle.Inputs) + if len(specs) != 1 { + t.Fatalf("design-qa@0 binds %d ux-spec inputs, want 1: %+v", len(specs), specs) + } + if specs[0].Body != specDocV2 { + t.Errorf("the bound spec is %q, want the pre-activation latest %q", + specs[0].Body, specDocV2) + } +} + +// recordExtraSpec records one more ux-spec artifact under the issue's done +// draft-spec step, through the artifact writer completion uses. +func recordExtraSpec(t *testing.T, conn *sql.DB, specIssue int, body string) { + t.Helper() + var stepID, runID int + err := conn.QueryRow( + `SELECT id, run_id FROM steps + WHERE issue_id = ? AND step_name = 'draft-spec' AND status = ?`, + specIssue, db.StepDone).Scan(&stepID, &runID) + testsupport.Must(t, err, "finding the done draft-spec step: %v", err) + + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + _, err = db.InsertArtifactTx(tx, db.Artifact{ + RunID: runID, StepID: stepID, Kind: "ux-spec", Body: body, + SHA256: workflow.SHA256([]byte(body)), + }, nowMS) + testsupport.Must(t, err, "recording the extra spec: %v", err) + testsupport.Must(t, tx.Commit(), "Commit: %v", err) +} + +// TestLinkedInputResolvesEveryLinkedIssueWithTheKind pins the multi-link +// rules. Relations are overloaded — `depends_on` orders scheduling as well as +// binding specs — so a consumer depending on two spec issues AND an +// implementation issue with no spec must bind both specs, in issue-id order, +// and must not refuse over the implementation issue. +func TestLinkedInputResolvesEveryLinkedIssueWithTheKind(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(specDocMiniSrc), "spec-doc-mini.toml") + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + specOne := produceSpec(t, conn, e, "spec one", "spec one's doc") + specTwo := produceSpec(t, conn, e, "spec two", "spec two's doc") + impl := createIssue(t, conn, "an implementation dependency", "code", "spike", nil) + + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + linkIssues(t, conn, consumer, specTwo, model.RelationDependsOn) + linkIssues(t, conn, consumer, specOne, model.RelationDependsOn) + linkIssues(t, conn, consumer, impl, model.RelationDependsOn) + + runB := startRun(t, conn, consumer) + _, err := activate(conn, runB.ID) + testsupport.Must(t, err, "activating the consumer run: %v", err) + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "design-qa@0"), nowMS) + testsupport.Must(t, err, "assembling design-qa@0's bundle: %v", err) + + specs := specInputs(bundle.Inputs) + if len(specs) != 2 { + t.Fatalf("design-qa@0 binds %d ux-spec inputs, want 2: %+v", len(specs), specs) + } + // Issue-id order, not link-creation order: specOne was linked second and + // still comes first, because the pin order is a pure function of the + // store. + if specs[0].Body != "spec one's doc" || specs[1].Body != "spec two's doc" { + t.Errorf("bound specs out of issue order: %q then %q", + specs[0].Body, specs[1].Body) + } +} + +// TestLinkedInputInverseRelationToken proves the form addresses BOTH +// directions of a relation: the spec issue `blocks` the consumer, and the +// consumer's `issue.linked.blocked-by.ux-spec` — the inverse token, in its +// hyphenated spelling — resolves through the relation's other end. +func TestLinkedInputInverseRelationToken(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(specDocMiniSrc), "spec-doc-mini.toml") + registerSource(t, conn, []byte(blockedUiMiniSrc), "blocked-ui-mini.toml") + + specIssue := produceSpec(t, conn, e, "write the spec", specDocV1) + + consumer := createIssue(t, conn, "the blocked change", "the change", "chore", nil) + linkIssues(t, conn, specIssue, consumer, model.RelationBlocks) + + runB := startRun(t, conn, consumer) + _, err := activate(conn, runB.ID) + testsupport.Must(t, err, "activating the consumer run: %v", err) + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "design-qa-b@0"), nowMS) + testsupport.Must(t, err, "assembling design-qa-b@0's bundle: %v", err) + + specs := specInputs(bundle.Inputs) + if len(specs) != 1 { + t.Fatalf("design-qa-b@0 binds %d ux-spec inputs, want 1: %+v", len(specs), specs) + } + if specs[0].Body != specDocV1 { + t.Errorf("the bound spec is %q, want %q", specs[0].Body, specDocV1) + } +} + +// TestLinkedInputPacketCarriesTheDoc is the happy path at the rendered layer — +// the packet a worker actually reads carries the spec's bytes, so the binding +// reaches the executor rather than stopping at the ledger. +func TestLinkedInputPacketCarriesTheDoc(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(specDocMiniSrc), "spec-doc-mini.toml") + registerSource(t, conn, []byte(uiChangeMiniSrc), "ui-change-mini.toml") + + specIssue := produceSpec(t, conn, e, "write the spec", specDocV1) + consumer := createIssue(t, conn, "apply the ui change", "the change", "task", nil) + linkIssues(t, conn, consumer, specIssue, model.RelationDependsOn) + runB := startRun(t, conn, consumer) + _, err := activate(conn, runB.ID) + testsupport.Must(t, err, "activating the consumer run: %v", err) + + packet, err := RenderStep(conn, stepIDByInstance(t, conn, "design-qa@0"), "", nowMS) + testsupport.Must(t, err, "rendering design-qa@0: %v", err) + + if got := strings.Count(packet.Packet, specDocV1); got != 1 { + t.Errorf("the spec is inlined %d times, want 1:\n%s", got, packet.Packet) + } +} diff --git a/internal/engine/dkt584_test.go b/internal/engine/dkt584_test.go new file mode 100644 index 00000000..d7f669c9 --- /dev/null +++ b/internal/engine/dkt584_test.go @@ -0,0 +1,305 @@ +package engine + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-584 — vote-step usage was invisible to the budget: a declared +// `expected_cost` on a `type="vote"` step was inert (votes are unclaimable, so +// no `step-claimed` event ever carried it into the floor), engine-minted +// reconcile-held ballots carried expected_cost 0 by construction, the report's +// `vote_usage` was silently excluded from `reported`/`spend` with nothing +// saying so, and a conversational-gate proposal that named a run (an +// activation panel, a reap-ack ballot) appeared in no run section at all. + +// voteCostSrc declares a vote step WITH an expected_cost — the declaration +// the issue found inert. +const voteCostSrc = ` +[pipeline] +name = "dkt584-vote-cost" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "worker" +emits = "diff" +expected_cost = 1.5 + +[[step]] +name = "tribunal" +after = ["implement"] +type = "vote" +voters = ["seat-a", "seat-b"] +vote_rule = "majority" +expected_cost = 2.0 +on_fail = "waiting-human" +` + +// TestFloorAccruesDeclaredVoteStepCostAtMaterialization: the floor includes a +// vote step's declared expected_cost from the moment the row exists — no +// claim event required, because none can ever exist for it. +func TestFloorAccruesDeclaredVoteStepCostAtMaterialization(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(voteCostSrc), "dkt584.toml") + issue := createIssue(t, conn, "vote cost", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // Before ANY claim: the floor is exactly the vote step's declaration. + if got := runFloor(t, conn, run.ID); got != 2.0 { + t.Fatalf("floor after activation = %g, want 2.0 (the vote step's "+ + "declared expected_cost, accrued at materialization)", got) + } + + // A claim accrues on top, exactly as before. + claimInstance(t, conn, "implement@0", nowMS) + if got := runFloor(t, conn, run.ID); got != 3.5 { + t.Errorf("floor after one claim = %g, want 3.5 (1.5 claimed + 2.0 vote)", got) + } +} + +// TestVoteStepCostIsNotReservedTwice: R7 must not add a vote step's cost on +// top of a floor that already carries it — spend()+0, not spend()+cost. +func TestVoteStepCostIsNotReservedTwice(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(voteCostSrc), "dkt584.toml") + issue := createIssue(t, conn, "vote cost", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // Cap exactly equals the run's whole declared cost: 1.5 + 2.0. + execSQL(t, conn, `UPDATE runs SET budget = ? WHERE id = ?`, 3.5, run.ID) + + // The executor step fits: spend (2.0 vote floor) + 1.5 = 3.5 <= 3.5. + loadScheduler(t, conn, run.ID, nowMS, func(sched *Scheduler) { + if ready, cond := sched.Ready(stepNamed(t, sched, "implement@0")); !ready { + t.Fatalf("implement@0 not ready under an exactly-fitting cap: %s "+ + "(the vote step's cost is being counted against the cap twice)", + cond) + } + }) + + claimAndComplete(t, conn, e, "implement@0", "a diff", "") + + // The vote step's own turn: spend is now 3.5 (1.5 claimed + 2.0 vote). + // Its readiness must reserve 0 — the 2.0 is ALREADY in the floor — so the + // budget clause passes at an exactly-spent cap. + loadScheduler(t, conn, run.ID, nowMS, func(sched *Scheduler) { + if ready, cond := sched.Ready(stepNamed(t, sched, "tribunal@0")); !ready { + t.Errorf("tribunal@0 not ready at an exactly-spent cap: %s (its "+ + "declared cost is in the floor and must not be reserved again)", + cond) + } + }) +} + +// TestHeldVoteStepMintsConfiguredCost: `vote.hold.cost` lands on the +// engine-minted held ballot's row and reaches the floor — the configurable +// default for the steps that carried 0 by construction. +func TestHeldVoteStepMintsConfiguredCost(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + configureHoldTally(t, conn, "dkt584-panel", "seat-a,seat-b") + err := db.SetConfig(conn, 0, db.KeyVoteHoldCost, "2.5") + testsupport.Must(t, err, "setting %s: %v", db.KeyVoteHoldCost, err) + + floorBefore := runFloor(t, conn, run.ID) + driveToReconcile(t, conn, e, clusteredPayload) + held := heldInstances(t, conn) + if len(held) == 0 { + t.Fatal("nothing held") + } + + for _, instance := range held { + step := heldStep(t, conn, instance) + if step.Kind != workflow.TypeVote { + t.Fatalf("%s minted %q, want %q", instance, step.Kind, workflow.TypeVote) + } + if step.ExpectedCost != 2.5 { + t.Errorf("%s expected_cost = %g, want the configured 2.5", + instance, step.ExpectedCost) + } + } + + // The floor rose by the drive's own claims PLUS 2.5 per minted ballot. + var claimed float64 + err = conn.QueryRow( + `SELECT COALESCE(SUM(s.expected_cost), 0) + FROM events e JOIN steps s ON s.id = e.step_id + WHERE e.run_id = ? AND e.kind = ?`, run.ID, EventStepClaimed, + ).Scan(&claimed) + testsupport.Must(t, err, "summing claim events: %v", err) + + want := floorBefore + claimed + 2.5*float64(len(held)) + if got := runFloor(t, conn, run.ID); got != want { + t.Errorf("floor = %g, want %g (claims %g + %d held ballots at 2.5)", + got, want, claimed, len(held)) + } +} + +// TestHumanHoldIgnoresConfiguredCost: a hold minted `human` (no tally +// configured) stays at 0 even with `vote.hold.cost` set — an operator's +// decision is not a panel's spend. +func TestHumanHoldIgnoresConfiguredCost(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + err := db.SetConfig(conn, 0, db.KeyVoteHoldCost, "2.5") + testsupport.Must(t, err, "setting %s: %v", db.KeyVoteHoldCost, err) + + driveToReconcile(t, conn, e, clusteredPayload) + held := heldInstances(t, conn) + if len(held) == 0 { + t.Fatal("nothing held") + } + for _, instance := range held { + step := heldStep(t, conn, instance) + if step.Kind != workflow.TypeHuman { + t.Fatalf("%s minted %q, want %q", instance, step.Kind, workflow.TypeHuman) + } + if step.ExpectedCost != 0 { + t.Errorf("%s expected_cost = %g, want 0 on a human hold", + instance, step.ExpectedCost) + } + } +} + +// TestRunReportNotesVoteUsageExclusion: a run whose panels cast carries the +// explicit exclusion note in its budget section; a run with no panels does +// not. The silent omission — vote_usage disjoint from reported/spend with +// nothing saying so — is the reported bug. +func TestRunReportNotesVoteUsageExclusion(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + configureHoldTally(t, conn, "note-panel", "seat-a,seat-b") + driveToReconcile(t, conn, e, clusteredPayload) + nextRun(t, conn, e) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + if report.Budget.VoteUsageNote != "" { + t.Errorf("a run with no casts carries a note: %q", report.Budget.VoteUsageNote) + } + + held := heldInstances(t, conn) + if len(held) == 0 { + t.Fatal("nothing held") + } + proposalID := heldProposalID(t, conn, e, held[0]) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: proposalID, VoterName: "seat-a", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + Usage: map[string]float64{"tokens": 100}, + }) + testsupport.Must(t, err, "CastVote: %v", err) + + report, err = LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + if report.Budget.VoteUsageNote != VoteUsageExcludedNote { + t.Errorf("vote_usage_note = %q, want the exclusion note once a seat cast", + report.Budget.VoteUsageNote) + } +} + +// TestConversationalProposalNamingRunJoinsVoteUsage: a proposal that NAMES +// the run in its description — an activation panel opened with `vote create`, +// no idempotency key, no step — is attributed to that run's vote_usage +// rollup and its coverage line. A proposal naming a DIFFERENT run whose +// rendered id merely contains this one's ("RUN-1" inside "RUN-17") is not. +func TestConversationalProposalNamingRunJoinsVoteUsage(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + cast := func(description string, usage float64) { + t.Helper() + id, err := db.CreateProposal(conn, &model.Proposal{ + ProjectID: 1, Description: description, + Rationale: "conversational gate", + Criticality: model.CriticalityMedium, + Threshold: 0.5, RequiredVoters: 3, + Status: model.ProposalStatusOpen, CreatedBy: "conductor", + }) + testsupport.Must(t, err, "CreateProposal(%q): %v", description, err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: id, VoterName: "seat-a", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + Usage: map[string]float64{"tokens": usage}, + }) + testsupport.Must(t, err, "CastVote on %q: %v", description, err) + } + + token := model.FormatRunID(run.ID) + cast("activation panel for "+token+": proceed with the batch?", 42) + // The boundary case: this proposal names a run whose id CONTAINS ours. + cast("activation panel for "+token+"7: unrelated run", 999) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + var tokens *db.UnitTotal + for i := range report.VoteUsage { + if report.VoteUsage[i].Unit == "tokens" { + tokens = &report.VoteUsage[i] + } + } + if tokens == nil { + t.Fatalf("vote_usage carries no tokens rollup: %+v — the "+ + "run-naming conversational proposal was not attributed", + report.VoteUsage) + } + if tokens.Quantity != 42 || tokens.Rows != 1 { + t.Errorf("tokens rollup = %+v, want 42 across 1 seat report — 999 "+ + "belongs to the OTHER run", tokens) + } + if c := report.VoteUsageCoverage; c.Casts != 1 || c.Reported != 1 { + t.Errorf("coverage = %+v, want 1 cast / 1 reported", c) + } +} + +// TestReapAckBallotJoinsVoteUsage: a reap-ack ballot — keyed to the run by +// ReapAckProposalKey but outside the vote-step family — is attributed too. +func TestReapAckBallotJoinsVoteUsage(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + id, err := db.CreateProposalIdempotent(conn, &model.Proposal{ + ProjectID: 1, Description: "accept the reap at seq 42?", + Criticality: model.CriticalityMedium, + Threshold: 0.5, RequiredVoters: 3, + Status: model.ProposalStatusOpen, CreatedBy: "conductor", + }, ReapAckProposalKey(run.ID, 42)) + testsupport.Must(t, err, "CreateProposalIdempotent: %v", err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: id, VoterName: "seat-a", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + Usage: map[string]float64{"tokens": 7}, + }) + testsupport.Must(t, err, "CastVote: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + var tokens *db.UnitTotal + for i := range report.VoteUsage { + if report.VoteUsage[i].Unit == "tokens" { + tokens = &report.VoteUsage[i] + } + } + if tokens == nil || tokens.Quantity != 7 || tokens.Rows != 1 { + t.Errorf("vote_usage tokens = %+v, want 7 across 1 seat report from "+ + "the reap-ack ballot", tokens) + } +} diff --git a/internal/engine/dkt586_test.go b/internal/engine/dkt586_test.go new file mode 100644 index 00000000..6096a20f --- /dev/null +++ b/internal/engine/dkt586_test.go @@ -0,0 +1,223 @@ +package engine + +import ( + "encoding/json" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-586 — RUN-30's spurious resume, pinned end to end. +// +// RUN-30's trail: seq 3053 an operator `run pause` (active -> waiting-human); +// seq 3057 a `run-resumed` with EMPTY data, one millisecond after synthesize@1 +// recorded (seq 3055/3056), with no operator verb in between; seq 3083 a second +// operator pause whose reason complains the run "read back as active even after +// the dispatch (DISPATCH-129) was reconciled and closed"; seq 3124 the real +// resume, carrying {from, to, reason}. +// +// The defect was reconcileRun's default branch: a run-level pause parks no +// step, so the rollup read `parked == 0` and flipped the run back to `active` +// the moment an in-flight step's record routed through it. That is DKT-305's +// defect exactly (RUN-31 was the other victim), fixed by the `pause_origin` +// guard, and DKT-304 additionally made every rollup-written lifecycle event +// self-identifying ({from, to, reason: "rollup"}) so an empty-data resume can +// no longer be produced by ANY path. +// +// This test drives RUN-30's full shape — both incidents in one run — against +// the current code as a regression pin: +// +// 1. step claimed under a dispatch, operator pauses, the step RECORDS: the +// run must stay `waiting-human` and no `run-resumed` may appear (the +// acceptance criterion's first clause). +// 2. the dispatch is then reconciled and closed (backfill + verify + close, +// DISPATCH-129's pipeline): the run must STILL read `waiting-human` — +// nothing in the dispatch pipeline touches the run row, so the pause +// stands across the close. +// 3. the only way back to `active` is the resume transition — `run resume` +// via MoveRun — and its event carries {from, to, reason} (the criterion's +// second clause, and seq 3124's shape). +func TestRun30PauseSurvivesStepRecordAndDispatchClose(t *testing.T) { + conn := mustDB(t) + e := testEngine() + run, _ := activatedRun(t, conn) + + // The wave: a manifest is open and its step is claimed — the state RUN-30 + // was in when the operator typed the pause. Limit 1 so the manifest holds + // only the claimed step: RUN-30's wave had recorded every manifest row + // before the reconcile ran, and VerifyDispatch (correctly) reports a + // mismatch for a manifest whose UNRECORDED rows are absent from a paused + // run's ready set — R1 empties it — which is a different fact than the one + // this test pins. + openDispatch(t, conn, run.ID, 1, nowMS) + implID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement@0: %v", err) + + // seq 3053: the operator pauses. Run-level — no step parks. + paused, _, err := MoveRun(conn, run.ID, "pause", model.RunWaitingHuman, + []model.RunStatus{model.RunActive}, "operator paused mid-wave", nowMS+1) + testsupport.Must(t, err, "MoveRun pause: %v", err) + if paused.Status != model.RunWaitingHuman { + t.Fatalf("premise: run is %s after the pause, want %s", + paused.Status, model.RunWaitingHuman) + } + + // seq 3055/3056: the in-flight step RECORDS, one tick later. Its routing + // transaction runs reconcileRun with parked == 0 (no step is + // waiting-human) and unfinished > 0 (the review fan-out is pending) — the + // exact branch that resumed RUN-30 at seq 3057. + err = e.CompleteStep(conn, implID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the change summary"), NowMS: nowMS + 2, + }) + testsupport.Must(t, err, "recording the in-flight step: %v", err) + + after, err := db.GetRun(conn, run.ID) + testsupport.Must(t, err, "GetRun: %v", err) + if after.Status != model.RunWaitingHuman { + t.Fatalf("run is %s after a step recorded into the pause, want %s — "+ + "the step-record path wrote the run row back toward active (DKT-586)", + after.Status, model.RunWaitingHuman) + } + if n := countRunEvents(t, conn, run.ID, EventRunResumed); n != 0 { + t.Fatalf("%d run-resumed event(s) after a step recorded into the pause, "+ + "want 0 — seq 3057's spurious resume", n) + } + + // DISPATCH-129's pipeline: backfill, verify, close — while the run is + // paused, exactly as RUN-30's wave was reconciled. None of the three + // stages may move the run row. + _, err = e.ReconcileDispatch(conn, run.ID, []BackfillRow{ + {Step: implID, Unit: "tokens", Quantity: 48211}, + }, "wave-journal:wf-30", "", false, nowMS+3) + testsupport.Must(t, err, "reconciling the dispatch: %v", err) + if status, _ := dispatchStatus(t, conn, run.ID); status != db.DispatchClosed { + t.Fatalf("premise: dispatch is %s after the reconcile, want %s", + status, db.DispatchClosed) + } + + closed, err := db.GetRun(conn, run.ID) + testsupport.Must(t, err, "GetRun after the close: %v", err) + if closed.Status != model.RunWaitingHuman { + t.Fatalf("run reads %s after the dispatch was reconciled and closed, "+ + "want %s — RUN-30's second pause complained of exactly this", + closed.Status, model.RunWaitingHuman) + } + if n := countRunEvents(t, conn, run.ID, EventRunResumed); n != 0 { + t.Fatalf("%d run-resumed event(s) after the dispatch closed, want 0", n) + } + + // seq 3124: the legitimate resume, through the resume transition and + // nothing else. It moves the run and its event carries {from, to, reason}. + resumed, _, err := MoveRun(conn, run.ID, "resume", model.RunActive, + []model.RunStatus{model.RunWaitingHuman}, "operator carried on", nowMS+4) + testsupport.Must(t, err, "MoveRun resume: %v", err) + if resumed.Status != model.RunActive { + t.Fatalf("run is %s after `run resume`, want %s", resumed.Status, model.RunActive) + } + + page, err := ListEvents(conn, EventQuery{RunID: run.ID}) + testsupport.Must(t, err, "ListEvents: %v", err) + var resumes []Event + for _, ev := range page.Events { + if ev.Kind == EventRunResumed { + resumes = append(resumes, ev) + } + } + if len(resumes) != 1 { + t.Fatalf("%d run-resumed event(s) in the feed, want exactly 1 — the "+ + "operator's own", len(resumes)) + } + var data lifecycleData + testsupport.Must(t, json.Unmarshal(resumes[0].Data, &data), + "decoding the resume's data: %v", err) + if data.From != string(model.RunWaitingHuman) || data.To != string(model.RunActive) { + t.Errorf("resume data = %s -> %s, want waiting-human -> active", data.From, data.To) + } + if data.Reason != "operator carried on" { + t.Errorf("resume reason = %q, want the operator's — an empty or absent "+ + "reason is seq 3057's unattributable shape", data.Reason) + } +} + +// TestNoRunResumedEventEverCarriesEmptyData pins DKT-304's half of the fix +// from the ledger side: the ROLLUP's own resume — the legitimate, step-park +// kind — identifies itself. RUN-30's seq 3057 was diagnosable only because its +// data was empty; with every writer of `run-resumed` now naming {from, to, +// reason}, an empty-data resume cannot be produced by any path, and this test +// fails if one ever is. +func TestNoRunResumedEventEverCarriesEmptyData(t *testing.T) { + const src = ` +[pipeline] +name = "parks586" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "flaky" +after = [] +executor = "w" +emits = "out" +max_attempts = 1 +on_fail = "waiting-human" + +[[step]] +name = "wrap" +after = ["flaky"] +executor = "w" +emits = "out2" +` + conn := mustDB(t) + registerSource(t, conn, []byte(src), "parks586.toml") + issue := createIssue(t, conn, "parked then resumed", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + id := stepIDByInstance(t, conn, "flaky@0") + + // Park the run at the STEP level, then resolve the park: the rollup's + // automatic resume — the one `run-resumed` no operator types. + claim, err := ClaimStep(conn, id, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + testsupport.Must(t, e.FailStep(conn, id, claim.Token, "gave up", "", nowMS), + "fail: %v", err) + testsupport.Must(t, e.ResolveStep(conn, id, ResolveSkip, "not worth it", nowMS+1), + "resolve: %v", err) + + after, err := db.GetRun(conn, run.ID) + testsupport.Must(t, err, "GetRun: %v", err) + if after.Status != model.RunActive { + t.Fatalf("premise: run is %s after its only park resolved, want %s "+ + "(the rollup's automatic resume)", after.Status, model.RunActive) + } + + page, err := ListEvents(conn, EventQuery{RunID: run.ID}) + testsupport.Must(t, err, "ListEvents: %v", err) + seen := 0 + for _, ev := range page.Events { + if ev.Kind != EventRunResumed { + continue + } + seen++ + var data lifecycleData + if err := json.Unmarshal(ev.Data, &data); err != nil { + t.Fatalf("run-resumed data %q does not decode: %v", ev.Data, err) + } + if data.From == "" || data.To == "" || data.Reason == "" { + t.Errorf("run-resumed carries empty data %s — RUN-30's seq 3057 "+ + "shape; every resume must name {from, to, reason}", ev.Data) + } + if data.Reason != runRollupReason { + t.Errorf("the rollup's resume reason = %q, want %q — the value no "+ + "operator verb can produce", data.Reason, runRollupReason) + } + } + if seen != 1 { + t.Fatalf("%d run-resumed event(s), want exactly 1 (the rollup's)", seen) + } +} diff --git a/internal/engine/dkt587_test.go b/internal/engine/dkt587_test.go new file mode 100644 index 00000000..cfa5cbb8 --- /dev/null +++ b/internal/engine/dkt587_test.go @@ -0,0 +1,403 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-587: RUN-34 (security-load-bearing@10, max_fix_loops = 2) showed three +// completed loop ordinals — fix@1, fix@2, fix@3 — while RUN-39 (spec-doc, +// max_fix_loops = 2) exhausted at ordinal 2, and the retro read that as the +// counter differing by entry path (reconcile threshold vs verify on_fail) or +// being off by one on one of them. +// +// These tests pin the opposite finding: at HEAD there is ONE counter and ONE +// definition. Every routing source — a threshold resolving `fix-loop`, an +// `on_fail = "fix-loop"` from a gate failure or a rejected human gate, a +// rejected vote tally, a quorum miss — reaches enterLoop through the same +// applyFixLoop bridge with the same workflow definition and `authorized = +// false`, and the arithmetic is post-increment: the admitted entry's count IS +// the issue's new 1-indexed ordinal, `count > max` refuses and restores the +// counter. `max_fix_loops = N` therefore admits exactly N entries from ANY mix +// of routing sources and parks the N+1th `waiting-human`. +// +// The one sanctioned way to a third ordinal under `max_fix_loops = 2` is an +// operator's `step resolve --as fix-round` (DKT-237), which records a grant +// and enters the round in one transaction — leaving loop_grants = 1, a +// step-resolved event, and the parked step superseded. That is the state +// RUN-34's ledger shape ("fix@3 done, no park left standing") matches, and the +// last test here reproduces it deliberately. +// +// Each round's routing step records its OWN artifact (roundReport) alongside +// the shared `unmetPayload`, for driveFixtureRound's reason one guard over: +// DKT-589 parks a loop whose routing step repeats the BYTE-IDENTICAL verdict at +// two consecutive ordinals, so a fixture reusing one constant body would park +// at entry 2 and never reach the bound these tests are about. The payload — the +// thing the threshold actually reads — is unchanged, so every entry still +// routes `fix-loop` from the same predicate it always did. + +// dkt587ThresholdSrc drives every loop entry through a THRESHOLD routing — +// the reconcile-threshold entry shape, reduced: `check` aggregates nothing +// and simply records a payload its own threshold reads. +const dkt587ThresholdSrc = ` +[pipeline] +name = "dkt587-threshold" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +threshold = { "fix-loop" = "any(status == unmet)" } +max_fix_loops = 2 + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +after_loop = "check" +` + +// dkt587OnFailSrc drives every loop entry through an ON_FAIL routing — a +// rejected human gate, the verify-on_fail entry shape's decision path. +const dkt587OnFailSrc = ` +[pipeline] +name = "dkt587-onfail" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "seed" +executor = "seed" +emits = "doc" + +[[step]] +name = "gate" +after = ["seed"] +type = "human" +on_fail = "fix-loop" +max_fix_loops = 2 + +[[step]] +name = "fix" +executor = "fix" +emits = "doc" +loop = true +after_loop = "gate" +` + +// dkt587MixedSrc lets ONE issue enter its loop through BOTH routing sources: +// `check`'s threshold and `gate`'s on_fail both feed the same (serves-free, +// single-cluster) loop. +const dkt587MixedSrc = ` +[pipeline] +name = "dkt587-mixed" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +threshold = { "fix-loop" = "any(status == unmet)" } +max_fix_loops = 2 + +[[step]] +name = "gate" +after = ["check"] +type = "human" +on_fail = "fix-loop" + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +after_loop = "check" +` + +// dkt587ClusterSrc declares the bound on a `serves`-SCOPED loop body — the +// shape under which a `max_fix_loops = 2` is the CLUSTER's round budget +// (clusterMaxFixLoops/clusterRoundsUsed) rather than the issue ceiling. The +// arithmetic must land on the same answer: two rounds admitted, the third +// refused. +const dkt587ClusterSrc = ` +[pipeline] +name = "dkt587-cluster" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +threshold = { "fix-loop" = "any(status == unmet)" } + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +serves = ["check"] +after_loop = "check" +max_fix_loops = 2 +` + +func TestThresholdEntriesBoundAtExactlyMaxFixLoops(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, dkt587ThresholdSrc) + e := testEngine() + + // Entry 1 and entry 2: both admitted at max_fix_loops = 2, each minting + // the issue's next ordinal. + claimAndComplete(t, conn, e, "check@0", roundReport(0), unmetPayload) + if got := loopCount(t, conn, runID, issue); got != 1 { + t.Fatalf("loop_count = %d after the first threshold entry, want 1", got) + } + if !stepExists(t, conn, "fix@1") { + t.Fatal("fix@1 was not instantiated by the first threshold entry") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after the second threshold entry, want 2", got) + } + if !stepExists(t, conn, "fix@2") { + t.Fatal("fix@2 was not instantiated by the second threshold entry") + } + + // Entry 3: refused. count = 3 > max = 2; the counter is restored, nothing + // is instantiated, and the routing step parks naming the way out. + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + if stepExists(t, conn, "fix@3") { + t.Error("fix@3 exists; max_fix_loops = 2 must refuse the third threshold entry") + } + if got := stepStatus(t, conn, "check@2"); got != db.StepWaitingHuman { + t.Errorf("check@2 = %q after the bound, want %q", got, db.StepWaitingHuman) + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Errorf("loop_count = %d after the refusal, want 2 — a refusal is not an ordinal", got) + } + raw := stepRoutingRaw(t, conn, "check@2") + if !strings.Contains(raw, "max_fix_loops = 2") || !strings.Contains(raw, "fix-round") { + t.Errorf("check@2 routing = %q, want it to name the bound and the fix-round way out", raw) + } +} + +func TestOnFailEntriesBoundAtExactlyMaxFixLoops(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, dkt587OnFailSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "seed@0", "the doc", "") + + // Entry 1 and entry 2: rejected human gates, both admitted. + err := e.DecideStep(conn, stepIDByInstance(t, conn, "gate@0"), false, "no", nowMS) + testsupport.Must(t, err, "rejecting gate@0: %v", err) + if got := loopCount(t, conn, runID, issue); got != 1 { + t.Fatalf("loop_count = %d after the first on_fail entry, want 1", got) + } + if !stepExists(t, conn, "fix@1") { + t.Fatal("fix@1 was not instantiated by the first on_fail entry") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + err = e.DecideStep(conn, stepIDByInstance(t, conn, "gate@1"), false, "still no", nowMS) + testsupport.Must(t, err, "rejecting gate@1: %v", err) + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after the second on_fail entry, want 2", got) + } + if !stepExists(t, conn, "fix@2") { + t.Fatal("fix@2 was not instantiated by the second on_fail entry") + } + + // Entry 3: refused, identically to the threshold path. + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + err = e.DecideStep(conn, stepIDByInstance(t, conn, "gate@2"), false, "third no", nowMS) + testsupport.Must(t, err, "rejecting gate@2: %v", err) + + if stepExists(t, conn, "fix@3") { + t.Error("fix@3 exists; max_fix_loops = 2 must refuse the third on_fail entry") + } + if got := stepStatus(t, conn, "gate@2"); got != db.StepWaitingHuman { + t.Errorf("gate@2 = %q after the bound, want %q", got, db.StepWaitingHuman) + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Errorf("loop_count = %d after the refusal, want 2", got) + } + raw := stepRoutingRaw(t, conn, "gate@2") + if !strings.Contains(raw, "max_fix_loops = 2") || !strings.Contains(raw, "fix-round") { + t.Errorf("gate@2 routing = %q, want it to name the bound and the fix-round way out", raw) + } +} + +// TestBothRoutingSourcesShareOneCounter is the acceptance criterion stated +// directly: an entry from the on_fail path and an entry from the threshold +// path move the SAME issue counter, so `max_fix_loops = 2` admits exactly two +// entries from any mix of sources — no divergence between them. +func TestBothRoutingSourcesShareOneCounter(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, dkt587MixedSrc) + e := testEngine() + + // check@0 passes; the human gate rejects — entry 1 arrives via ON_FAIL. + claimAndComplete(t, conn, e, "check@0", "findings", metPayload) + err := e.DecideStep(conn, stepIDByInstance(t, conn, "gate@0"), false, "no", nowMS) + testsupport.Must(t, err, "rejecting gate@0: %v", err) + if got := loopCount(t, conn, runID, issue); got != 1 { + t.Fatalf("loop_count = %d after the on_fail entry, want 1", got) + } + + // Entry 2 arrives via the THRESHOLD — and continues the same ordinal + // sequence, not a second counter of its own. + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after the threshold entry, want 2 — "+ + "both routing sources must move one counter", got) + } + if !stepExists(t, conn, "fix@2") { + t.Fatal("fix@2 was not instantiated by the threshold entry") + } + + // The third attempt — from either source — is refused. Here the threshold + // fires it; the two prior entries, one per source, have spent the budget. + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + if stepExists(t, conn, "fix@3") { + t.Error("fix@3 exists; two entries from mixed sources must exhaust max_fix_loops = 2") + } + if got := stepStatus(t, conn, "check@2"); got != db.StepWaitingHuman { + t.Errorf("check@2 = %q after the mixed-source bound, want %q", got, db.StepWaitingHuman) + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Errorf("loop_count = %d after the refusal, want 2", got) + } + + // The ledger names each entry's trigger: first the gate, then the check. + triggers := loopEnteredTriggers(t, conn, runID) + if len(triggers) != 2 || + !strings.Contains(triggers[0], `"gate"`) || + !strings.Contains(triggers[1], `"check"`) { + t.Errorf("loop-entered events = %v, want exactly two entries, the first "+ + "triggered by gate and the second by check", triggers) + } +} + +// TestClusterScopedBoundAdmitsExactlyItsRounds pins the SECOND arithmetic — +// clusterMaxFixLoops/clusterRoundsUsed, the mechanism that bounds a +// `max_fix_loops` declared on a `serves`-scoped body — to the same answer: +// bound 2 admits rounds 1 and 2 and refuses round 3, with no issue-level +// ceiling in the workflow at all. +func TestClusterScopedBoundAdmitsExactlyItsRounds(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, dkt587ClusterSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", roundReport(0), unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("fix@1 was not instantiated by the cluster's first round") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + if !stepExists(t, conn, "fix@2") { + t.Fatal("fix@2 was not instantiated by the cluster's second round") + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after two cluster rounds, want 2", got) + } + + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + if stepExists(t, conn, "fix@3") { + t.Error("fix@3 exists; a cluster bound of 2 must refuse its third round") + } + if got := stepStatus(t, conn, "check@2"); got != db.StepWaitingHuman { + t.Errorf("check@2 = %q after the cluster bound, want %q", got, db.StepWaitingHuman) + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Errorf("loop_count = %d after the cluster refusal, want 2", got) + } + raw := stepRoutingRaw(t, conn, "check@2") + if !strings.Contains(raw, "cluster") || !strings.Contains(raw, "max_fix_loops = 2") { + t.Errorf("check@2 routing = %q, want it to name the cluster's bound of 2", raw) + } +} + +// TestFixRoundGrantIsTheOnlyWayToAThirdOrdinal reproduces the ONE sanctioned +// path to RUN-34's observed shape — fix@1, fix@2, fix@3 all done under +// `max_fix_loops = 2`: an operator resolves the parked step `--as fix-round`, +// which records a grant and mints the third ordinal in one transaction. The +// durable evidence it leaves (loop_grants = 1, the park superseded) is exactly +// what distinguishes an authorized third round from a counting defect. +func TestFixRoundGrantIsTheOnlyWayToAThirdOrdinal(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, dkt587ThresholdSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", roundReport(0), unmetPayload) + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + // Bounded, as the tests above pin. + if got := stepStatus(t, conn, "check@2"); got != db.StepWaitingHuman { + t.Fatalf("check@2 = %q, want %q before the grant", got, db.StepWaitingHuman) + } + + // The operator authorizes one more round. + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "check@2"), + ResolveFixRound, "one more round", nowMS) + testsupport.Must(t, err, "resolving --as fix-round: %v", err) + + if !stepExists(t, conn, "fix@3") { + t.Fatal("fix@3 was not minted by the fix-round grant") + } + if got := loopCount(t, conn, runID, issue); got != 3 { + t.Errorf("loop_count = %d after the granted round, want 3", got) + } + if got := stepStatus(t, conn, "check@2"); got != db.StepSuperseded { + t.Errorf("check@2 = %q after the grant, want %q — the park's question is "+ + "answered by the new round's work", got, db.StepSuperseded) + } + // The grant is durable evidence: a ledger reader can tell this third + // ordinal from a counting defect. + var grants int + err = conn.QueryRow( + `SELECT loop_grants FROM run_issues WHERE run_id = ? AND issue_id = ?`, + runID, issue).Scan(&grants) + testsupport.Must(t, err, "reading loop_grants: %v", err) + if grants != 1 { + t.Errorf("loop_grants = %d after one fix-round resolution, want 1", grants) + } +} diff --git a/internal/engine/dkt588_test.go b/internal/engine/dkt588_test.go new file mode 100644 index 00000000..967678ff --- /dev/null +++ b/internal/engine/dkt588_test.go @@ -0,0 +1,262 @@ +package engine + +import ( + "database/sql" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-588: a fix round whose hand-back sha equals the prior round's re-ran the +// whole fanout. RUN-34 round 2: fix@2 recorded "Empty work list this round ... +// hand-back sha is unchanged HEAD 64d3336b3d71", judge-testing confirmed +// identical trees, 4 of 5 judges emitted zero findings — and a full 5-judge + +// synthesize + verify round (~2.9 budget units) still ran over a zero-byte +// delta, because DKT-340's non-convergence guard fires only at the NEXT loop's +// entry, after the wasted round has already been paid for. +// +// The fix parks the round AT ITS SOURCE: when a loop body at ordinal k > 0 +// hands back the same commit the same body recorded at an earlier ordinal, the +// body's own completion routes `waiting-human` naming the unchanged sha, and +// the downstream chain — already withheld from every offer while its +// same-ordinal loop body is non-terminal (DKT-48/DKT-61) — never runs. +// +// These tests drive the diff AND the head through the engine's own seams +// (DiffFn/HeadFn), for loop_convergence_test.go's stated reason: hand-inserted +// artifacts defeat the suppression rules the real ledger obeys. + +// handBackEngine is convergenceEngine with the hand-back head also under the +// test's control: HeadFn answers whatever the test last assigned, modelling a +// checkout whose HEAD moves only when a round actually commits something. +func handBackEngine(tree *treeState, head *string) *Engine { + e := convergenceEngine(tree) + e.HeadFn = func(string) string { return *head } + return e +} + +// enterSecondRound drives rounds 0 and 1 with real movement — the tree and the +// head both advance — so round 2 is genuinely entered and fix@2 exists. It +// returns the head fix@1 recorded, which is the sha the guard compares +// against. +func enterSecondRound(t *testing.T, conn *sql.DB, e *Engine, tree *treeState, head *string) string { + t.Helper() + tree.body, *head = "the original change", "aaaa1111bbbb2222cccc3333dddd4444" + driveRound(t, conn, e, 0) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: round 0 must have entered the loop") + } + tree.body, *head = "the original change, plus fix 1", "eeee5555ffff6666aaaa7777bbbb8888" + driveRound(t, conn, e, 1) + if !stepExists(t, conn, "fix@2") { + t.Fatal("premise: round 1 must have entered round 2") + } + return *head +} + +// TestUnchangedHandBackParksTheRoundBeforeTheFanout is the verbatim RUN-34 +// shape: fix@2 completes handing back the same commit fix@1 recorded, and the +// round short-circuits to `waiting-human` before a single judge is offered. +func TestUnchangedHandBackParksTheRoundBeforeTheFanout(t *testing.T) { + conn := mustDB(t) + run, issue := activatedRun(t, conn) + tree := &treeState{} + head := "" + e := handBackEngine(tree, &head) + + prior := enterSecondRound(t, conn, e, tree, &head) + + // fix@2 does nothing: the tree does not move and HEAD stays where fix@1 + // left it. + claimAndComplete(t, conn, e, "fix@2", "no work this round", "") + + if got := stepStatus(t, conn, "fix@2"); got != db.StepWaitingHuman { + t.Errorf("fix@2 = %q, want the unchanged-hand-back park %q", + got, db.StepWaitingHuman) + } + + // The park NAMES THE UNCHANGED SHA and the way out, like every other + // refusal beside it. + routing := stepRoutingRaw(t, conn, "fix@2") + if !strings.Contains(routing, prior[:12]) { + t.Errorf("the park does not name the unchanged sha: %q", routing) + } + if !strings.Contains(routing, "override-pass") { + t.Errorf("the park names no way out: %q", routing) + } + + // THE FANOUT NEVER RUNS. The run itself parks with the step, so nothing at + // all is offered — this is the round DKT-340 could only refuse after it + // had already been paid for. + if offered := readyInstances(t, conn); len(offered) != 0 { + t.Errorf("a parked round still offers work: %v", offered) + } + var runStatus string + testsupport.Must(t, + conn.QueryRow(`SELECT status FROM runs WHERE id = ?`, run.ID).Scan(&runStatus), + "reading run status: %v", nil) + if model.RunStatus(runStatus) != model.RunWaitingHuman { + t.Errorf("run = %q, want %q", runStatus, model.RunWaitingHuman) + } + + // The chain is PENDING, not superseded: `superseded` is terminal, and a + // swept chain would let the issue complete without its judges the moment + // the park resolves. The park holds it; the resolution releases it. + if got := stepStatus(t, conn, "review@2#0"); got != db.StepPending { + t.Errorf("review@2#0 = %q, want %q — the chain waits, it is not swept", + got, db.StepPending) + } + + // The counter is untouched: unlike a refused ENTRY, this round exists — + // its body ran and parked — so ordinal 2 is genuinely the issue's current + // ordinal. + if got := loopCount(t, conn, run.ID, issue); got != 2 { + t.Errorf("loop_count = %d, want 2 — the round was entered, then parked", got) + } + + // THE WAY OUT WORKS: `--as override-pass` accepts the hand-back anyway, + // and the chain the park was holding becomes offerable again. + testsupport.Must(t, e.ResolveStep(conn, stepIDByInstance(t, conn, "fix@2"), + ResolveOverridePass, "run the chain anyway", nowMS), + "resolving fix@2: %v", nil) + offered := readyInstances(t, conn) + found := false + for _, instance := range offered { + if strings.HasPrefix(instance, "review@2") { + found = true + } + } + if !found { + t.Errorf("override-pass did not release the round's chain; offered %v", offered) + } +} + +// TestChangedHandBackProceedsDownstream is the regression bound: a round that +// hands back a NEW commit is not the repeat case, and the chain proceeds +// exactly as before. +func TestChangedHandBackProceedsDownstream(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + tree := &treeState{} + head := "" + e := handBackEngine(tree, &head) + + enterSecondRound(t, conn, e, tree, &head) + + tree.body, head = "the original change, plus fixes 1 and 2", + "9999aaaa8888bbbb7777cccc6666dddd" + claimAndComplete(t, conn, e, "fix@2", "the fix summary", "") + + if got := stepStatus(t, conn, "fix@2"); got != db.StepDone { + t.Errorf("fix@2 = %q, want %q — a moved hand-back is a real round", got, db.StepDone) + } + offered := readyInstances(t, conn) + found := false + for _, instance := range offered { + if strings.HasPrefix(instance, "review@2") { + found = true + } + } + if !found { + t.Errorf("a real round's chain was not offered; offered %v", offered) + } +} + +// TestUnresolvableHandBackNeverParksTheRound: a head that cannot be resolved +// ("" — no commit to name) is not evidence of an unchanged tree. The guard +// answers "changed" and the round proceeds, exactly as roundMovedNothing +// refuses to park on a degenerate diff — a broken head resolution must not +// become a stalled run. +func TestUnresolvableHandBackNeverParksTheRound(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + tree := &treeState{} + head := "" + e := handBackEngine(tree, &head) + + enterSecondRound(t, conn, e, tree, &head) + + // The tree does not move AND the head cannot be resolved: no measurement, + // no park. + head = "" + claimAndComplete(t, conn, e, "fix@2", "no work this round", "") + + if got := stepStatus(t, conn, "fix@2"); got != db.StepDone { + t.Errorf("fix@2 = %q, want %q — an unresolvable head is not evidence "+ + "of an unchanged tree", got, db.StepDone) + } +} + +// TestMissingPriorHandBackNeverParksTheRound is the other degenerate side: the +// prior round recorded no head at all (its checkout's HEAD was unresolvable at +// the time), so there is nothing real to compare against and the round +// proceeds. +func TestMissingPriorHandBackNeverParksTheRound(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + tree := &treeState{} + head := "" + e := handBackEngine(tree, &head) + + // Rounds 0 and 1 move the tree but never resolve a head, so no round + // record carries one. + tree.body, head = "the original change", "" + driveRound(t, conn, e, 0) + tree.body = "the original change, plus fix 1" + driveRound(t, conn, e, 1) + if !stepExists(t, conn, "fix@2") { + t.Fatal("premise: round 1 must have entered round 2") + } + + // fix@2 resolves a head for the first time; there is no prior hand-back to + // equal, whatever the tree did. + head = "1111eeee2222ffff3333aaaa4444bbbb" + claimAndComplete(t, conn, e, "fix@2", "no work this round", "") + + if got := stepStatus(t, conn, "fix@2"); got != db.StepDone { + t.Errorf("fix@2 = %q, want %q — no prior hand-back means no evidence", + got, db.StepDone) + } +} + +// TestHeadMovementDoesNotSatisfyTheConvergenceGuard pins the two mechanisms +// apart (DKT-340 vs DKT-588). A round whose HEAD moved — an out-of-scope +// commit — while the scoped diff stayed identical is invisible to the +// hand-back guard (the sha moved) and still caught by roundMovedNothing at the +// next loop's entry, on its own later, differently-scoped trigger. +func TestHeadMovementDoesNotSatisfyTheConvergenceGuard(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + tree := &treeState{} + head := "" + e := handBackEngine(tree, &head) + + tree.body, head = "the original change", "aaaa1111bbbb2222cccc3333dddd4444" + driveRound(t, conn, e, 0) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: round 0 must have entered the loop") + } + + // Round 1: HEAD moves, the scoped tree does not. + head = "eeee5555ffff6666aaaa7777bbbb8888" + driveRound(t, conn, e, 1) + + // DKT-588 stayed silent — fix@1 has no prior hand-back and its sha moved — + // so the round ran; DKT-340 then refused the NEXT entry. + if got := stepStatus(t, conn, "fix@1"); got != db.StepDone { + t.Errorf("fix@1 = %q, want %q — the hand-back guard has no business here", + got, db.StepDone) + } + if stepExists(t, conn, "fix@2") { + t.Error("a round that changed nothing in scope minted the next round") + } + if got := stepStatus(t, conn, "reconcile@1"); got != db.StepWaitingHuman { + t.Errorf("reconcile@1 = %q, want the non-convergence park %q", + got, db.StepWaitingHuman) + } + if routing := stepRoutingRaw(t, conn, "reconcile@1"); !strings.Contains(routing, "changed nothing") { + t.Errorf("DKT-340's own park lost its reason: %q", routing) + } +} diff --git a/internal/engine/dkt589_test.go b/internal/engine/dkt589_test.go new file mode 100644 index 00000000..02dd002f --- /dev/null +++ b/internal/engine/dkt589_test.go @@ -0,0 +1,482 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-589 extends DKT-340's non-convergence guard to the signal DKT-340 cannot +// see: a round that MOVED BYTES and still changed nothing that matters. +// +// RUN-31 (DKT-294) is the shape. verify@0 reported AC2 and AC5 unmet; fix@1 +// committed an 82,742-byte diff; verify@1 reported AC2 and AC5 unmet with AC9 +// newly regressed; fix@2 committed 2,571 bytes; verify@2 came back +// BYTE-IDENTICAL to verify@1. Both rounds moved real bytes, so DKT-340's +// byte-based guard correctly stayed silent — and 342,490 output tokens, 45.8% +// of the run, closed zero acceptance criteria. RUN-34 is the same shape with +// four byte-identical ac-reports. +// +// WHAT CORE COMPARES IS BYTES, NOT CRITERIA. The issue describes the symptom as +// "the same criterion ids at the same statuses", but `criterion`, `id` and +// `status` are the workflow author's vocabulary — core no more parses them than +// it parses `severity` (genericity.md, roundMovedNothing's own reasoning). The +// engine reads the kind the routing step's OWN `emits` declares and compares +// the recorded artifact fingerprints. These tests therefore drive the CANONICAL +// fixture, whose `verify` emits `ac-report`, and one interposed fixture whose +// `check` emits `findings`: the same guard fires on both, because neither kind +// is known to it. + +// driveRoundToVerify completes one ordinal's chain up to but NOT including its +// `verify`, leaving the stub tree exactly where the caller put it. +// +// It is driveToVerify without the driveFixtureRound move, which is the whole +// point: the tests below need to control the tree and the verdict +// INDEPENDENTLY, because the two guards read one each and the interesting cases +// are the ones where they disagree. +func driveRoundToVerify(t *testing.T, conn *sql.DB, e *Engine, ordinal int) { + t.Helper() + if ordinal == 0 { + claimAndComplete(t, conn, e, "implement@0", "the change summary", "") + } else { + claimAndComplete(t, conn, e, fmt.Sprintf("fix@%d", ordinal), "the fix summary", "") + } + completeReviewFanout(t, conn, e, ordinal) + claimAndComplete(t, conn, e, fmt.Sprintf("synthesize@%d", ordinal), "the synthesis", "") + driveAction(t, conn, e, fmt.Sprintf("reconcile@%d", ordinal)) +} + +// theSameACReport is RUN-31's verify@1 and verify@2: one report, recorded +// twice. A constant is the SUBJECT here, not a fixture shortcut. +const theSameACReport = `AC2 unmet: the retry budget is still unbounded +AC5 unmet: no test covers the exhausted path` + +// TestIdenticalVerdictAtTwoOrdinalsParksTheLoop is the verbatim RUN-31 shape. +// +// Both rounds MOVE THE TREE — driveToVerify advances the stub diff per ordinal, +// exactly as a fix round that commits 82,742 bytes does — so DKT-340's own +// guard sees genuine movement and stays silent. The only thing that did not +// change is the verdict, and that is enough. +func TestIdenticalVerdictAtTwoOrdinalsParksTheLoop(t *testing.T) { + conn := mustDB(t) + run, issue := activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", theSameACReport, unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: the first verdict must enter the loop") + } + if got := loopCount(t, conn, run.ID, issue); got != 1 { + t.Fatalf("premise: loop_count = %d after the first entry, want 1", got) + } + + // Round 1 does real work — the tree moves — and reaches the IDENTICAL + // verdict. + driveToVerify(t, conn, e, 1) + claimAndComplete(t, conn, e, "verify@1", theSameACReport, unmetPayload) + + if stepExists(t, conn, "fix@2") { + t.Error("a third round was minted after `verify` recorded the identical " + + "verdict twice; the next round is handed the same verdict the last " + + "one already failed to change") + } + if got := stepStatus(t, conn, "verify@1"); got != db.StepWaitingHuman { + t.Errorf("verify@1 = %q, want the repeated-verdict park %q", + got, db.StepWaitingHuman) + } + + routing := stepRoutingRaw(t, conn, "verify@1") + + // The park NAMES THE UNCHANGED VERDICT — which step repeated itself, and at + // which two ordinals — because that is what the operator has to decide + // about. + if !strings.Contains(routing, "identical verdict") || + !strings.Contains(routing, `"verify"`) { + t.Errorf("the park does not name the unchanged verdict: %q", routing) + } + if !strings.Contains(routing, "ordinals 0 and 1") { + t.Errorf("the park does not name the two ordinals: %q", routing) + } + + // AND IT IS THE NEW SIGNAL THAT FIRED, not DKT-340's. roundMovedNothing is + // evaluated first and its clause would have won; the tree moved, so it + // correctly said nothing. + if strings.Contains(routing, "changed nothing") { + t.Errorf("DKT-340's byte guard claimed this park; the tree moved in both "+ + "rounds and only the verdict repeated: %q", routing) + } + + // The way out is the EXISTING verb, not a new one. + if !strings.Contains(routing, "--as fix-round") { + t.Errorf("the park names no way out: %q", routing) + } + + // AND THE COUNTER IS PUT BACK, exactly as the bound and DKT-340 refusals do: + // a refusal minted no ordinal, and a counter left above the highest + // instantiated one declares that ordinal stale in its entirety (DKT-78). + if got := loopCount(t, conn, run.ID, issue); got != 1 { + t.Errorf("loop_count = %d after a refused entry, want 1 — the refusal "+ + "minted no round", got) + } +} + +// TestChangedVerdictAtTwoOrdinalsEntersTheLoop is the lower bound, and the half +// that matters most: a guard that fired on a loop still learning something +// would break every workflow that uses one. +// +// RUN-31's round 1 is exactly this case — AC2 and AC5 still unmet but AC9 newly +// regressed — and it must not park. Only the round after it, whose report was +// byte-identical, is the repeat. +func TestChangedVerdictAtTwoOrdinalsEntersTheLoop(t *testing.T) { + conn := mustDB(t) + run, issue := activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", theSameACReport, unmetPayload) + + driveToVerify(t, conn, e, 1) + claimAndComplete(t, conn, e, "verify@1", + theSameACReport+"\nAC9 regressed: the reap path lost its narration", + unmetPayload) + + if !stepExists(t, conn, "fix@2") { + t.Error("a round whose verdict changed did not mint the next one; the " + + "loop must keep running while it is still learning something") + } + if got := stepStatus(t, conn, "verify@1"); got != db.StepDone { + t.Errorf("verify@1 = %q, want %q — a changed verdict is a real round", + got, db.StepDone) + } + if got := loopCount(t, conn, run.ID, issue); got != 2 { + t.Errorf("loop_count = %d, want 2", got) + } +} + +// TestFixRoundOverridesTheRepeatedVerdictPark is the escape hatch, and the +// reason parking is the right refusal rather than a hard stop. It is the SAME +// verb that gets past the bound (DKT-237) and past DKT-340's own park — the +// issue is explicit that no new override verb is invented. +func TestFixRoundOverridesTheRepeatedVerdictPark(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", theSameACReport, unmetPayload) + driveToVerify(t, conn, e, 1) + claimAndComplete(t, conn, e, "verify@1", theSameACReport, unmetPayload) + + if got := stepStatus(t, conn, "verify@1"); got != db.StepWaitingHuman { + t.Fatalf("premise: verify@1 = %q, want the park", got) + } + + testsupport.Must(t, e.ResolveStep(conn, stepIDByInstance(t, conn, "verify@1"), + ResolveFixRound, "AC2 is unmeetable; one more round to document it", nowMS), + "resolve --as fix-round: %v", nil) + + if !stepExists(t, conn, "fix@2") { + t.Error("the operator authorized a round past the repeated-verdict park " + + "and none was minted") + } +} + +// TestUnmovedTreeStillParksAsNonConvergence is DKT-340's own guard, unchanged. +// +// This addition is a SIBLING signal, not a replacement: the two are OR'd into +// one refusal and roundMovedNothing is still evaluated first, so a genuinely +// unmoved tree still parks with its own reason even when the verdicts differ +// and the new check would have said nothing. +func TestUnmovedTreeStillParksAsNonConvergence(t *testing.T) { + conn := mustDB(t) + run, issue := activatedRun(t, conn) + e := testEngine() + + // NO driveFixtureRound anywhere: the stub tree never moves, so the engine's + // own DKT-258 suppression records one diff for the whole run. + driveRoundToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", roundReport(0), unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: round 0 must have entered the loop") + } + + driveRoundToVerify(t, conn, e, 1) + claimAndComplete(t, conn, e, "verify@1", roundReport(1), unmetPayload) + + if stepExists(t, conn, "fix@2") { + t.Error("a round was minted after one that changed nothing in scope") + } + if got := stepStatus(t, conn, "verify@1"); got != db.StepWaitingHuman { + t.Errorf("verify@1 = %q, want DKT-340's park %q", got, db.StepWaitingHuman) + } + routing := stepRoutingRaw(t, conn, "verify@1") + if !strings.Contains(routing, "changed nothing") || + !strings.Contains(routing, "reaches the same verdict") { + t.Errorf("DKT-340's own park lost its reason: %q", routing) + } + if !strings.Contains(routing, "--as fix-round") { + t.Errorf("DKT-340's park lost its way out: %q", routing) + } + if got := loopCount(t, conn, run.ID, issue); got != 1 { + t.Errorf("loop_count = %d after DKT-340's refusal, want 1", got) + } +} + +// TestFirstVerdictIsNeverRefusedAsRepeated is the ordinal-0 case: a routing step +// that has run ONCE has no previous verdict of its own, and absence of evidence +// is never evidence of a repeat. +func TestFirstVerdictIsNeverRefusedAsRepeated(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", theSameACReport, unmetPayload) + + if !stepExists(t, conn, "fix@1") { + t.Error("the FIRST round was refused as a repeat; ordinal 0 has no " + + "previous verdict to equal") + } +} + +// TestEmptyVerdictNeverParksTheLoop is the measurement-must-be-real discipline, +// roundMovedNothing's degenerate-diff rule one guard over. +// +// Two artifacts with no bytes in either channel agree with each other trivially +// — not because the step reached the same conclusion twice, but because it +// recorded no conclusion at all. Parking on that would turn a misconfigured +// executor into a stalled run, a far worse failure than the wasted round this +// exists to prevent. +func TestEmptyVerdictNeverParksTheLoop(t *testing.T) { + conn := mustDB(t) + activateInterposed(t, conn, dkt589EmptyVerdictSrc) + e := testEngine() + e.Gates = &exitGates{fail: true, exit: 1} + + // `check` records an artifact with NO BYTES IN EITHER CHANNEL and its gate + // fails, so `on_fail = "fix-loop"` enters the loop off a step that recorded + // no conclusion at all — at both ordinals. + claimAndComplete(t, conn, e, "check@0", "", "") + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: the failed gate's on_fail must enter the loop") + } + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", "", "") + + if !stepExists(t, conn, "fix@2") { + t.Error("the loop parked on two empty verdicts agreeing; an artifact " + + "with no bytes is an absent measurement, not a repeated one, and " + + "parking on it turns a misconfigured executor into a stalled run") + } +} + +// dkt589EmptyVerdictSrc reaches the one state where a routing step records an +// artifact carrying NO BYTES and still routes `fix-loop`: a failing gate on a +// step whose single attempt is spent, with `on_fail = "fix-loop"`. The artifact +// lands at stage 1, before the gate runs, so the row exists and is empty. +const dkt589EmptyVerdictSrc = ` +[pipeline] +name = "dkt589-empty" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +gates = ["build"] +max_attempts = 1 +on_fail = "fix-loop" + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +after_loop = "check" +` + +// dkt589OnFailSrc gives ONE issue two different loop-entry triggers: `check`'s +// threshold, and `gate`'s on_fail — the rejected human gate. `check` also +// carries `max_attempts = 1`, so its exhausted-budget entry is reachable too. +// +// The three misfire tests below all need this: the new signal must fire only +// when a routing step's OWN emitted verdict repeated, and must be invisible to +// every other way a loop is entered. +const dkt589OnFailSrc = ` +[pipeline] +name = "dkt589-triggers" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +max_attempts = 1 +on_fail = "fix-loop" +threshold = { "fix-loop" = "any(status == unmet)" } + +[[step]] +name = "gate" +after = ["check"] +type = "human" +on_fail = "fix-loop" + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +after_loop = "check" +` + +// TestRejectedGateEntryIsUnaffected is the first misfire bound. A `type = +// "human"` gate records no artifact at all — workflow.ArtifactKind is "" for +// its class — so a gate rejected identically at two consecutive ordinals has +// nothing for this check to compare and enters the loop exactly as it always +// did, whatever the loop BODY's artifacts look like. +func TestRejectedGateEntryIsUnaffected(t *testing.T) { + conn := mustDB(t) + _, _ = activateInterposed(t, conn, dkt589OnFailSrc) + e := testEngine() + + // `check` passes at both ordinals, with the IDENTICAL findings — and it is + // not the trigger, so its repetition is not this guard's business either. + claimAndComplete(t, conn, e, "check@0", "no findings", metPayload) + err := e.DecideStep(conn, stepIDByInstance(t, conn, "gate@0"), false, "no", nowMS) + testsupport.Must(t, err, "rejecting gate@0: %v", err) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: the first rejection must enter the loop") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", "no findings", metPayload) + err = e.DecideStep(conn, stepIDByInstance(t, conn, "gate@1"), false, "still no", nowMS) + testsupport.Must(t, err, "rejecting gate@1: %v", err) + + if !stepExists(t, conn, "fix@2") { + t.Error("a rejected human gate's second entry was refused as a repeated " + + "verdict; a gate records no artifact, so this check has nothing to " + + "compare and must stay out of its way") + } +} + +// TestExhaustedAttemptEntryIsUnaffected is the second misfire bound, and it is +// what fixes the CURRENT side of the comparison at the step's exact ordinal +// rather than "at or below" it. +// +// `check@1` exhausts its single attempt and routes `fix-loop` from on_fail +// having recorded NO artifact of its own. Read "at or below ordinal 1", both +// sides of the comparison would resolve to check@0's row — the guard would +// compare one artifact with itself, find it equal, and park a run for a reason +// that never happened. +func TestExhaustedAttemptEntryIsUnaffected(t *testing.T) { + conn := mustDB(t) + activateInterposed(t, conn, dkt589OnFailSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", "the findings", unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: the threshold entry must mint fix@1") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + + // check@1 never completes: its one attempt fails, so `on_fail = "fix-loop"` + // enters the loop with nothing recorded at ordinal 1. + checkID := stepIDByInstance(t, conn, "check@1") + claim, err := ClaimStep(conn, checkID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claiming check@1: %v", err) + testsupport.Must(t, e.FailStep(conn, checkID, claim.Token, "the tool crashed", "", nowMS), + "failing check@1: %v", nil) + + if !stepExists(t, conn, "fix@2") { + t.Error("an exhausted attempt budget was refused as a repeated verdict; " + + "the step recorded nothing at this ordinal, so there is no verdict " + + "to have repeated") + } +} + +// dkt589SiblingSrc puts a SECOND producer of the same artifact kind beside the +// routing step, outside its `after_loop` chain so it survives a loop entry and +// can record after one. +const dkt589SiblingSrc = ` +[pipeline] +name = "dkt589-sibling" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "check" +executor = "check" +emits = "findings" +threshold = { "fix-loop" = "any(status == unmet)" } + +[[step]] +name = "note" +after = [] +executor = "note" +emits = "findings" + +[[step]] +name = "fix" +executor = "fix" +emits = "findings" +loop = true +after_loop = "check" +` + +// TestAnotherProducersArtifactIsNotThisStepsVerdict pins the query's SCOPE: +// (run, issue, STEP NAME, kind), priorRoundHandBack's shape rather than +// latestIssueDiffHead's issue-wide one. +// +// `note` emits `findings` too, and records the newest one at ordinal 0 — so a +// comparison scoped by kind alone would ask whether `check@1`'s verdict equals +// NOTE's artifact, conclude it does not, and let a genuinely repeated verdict +// through. The issue-wide read is right for `issue.diff`, which is one +// cumulative fact about the tree whoever produced it, and wrong for a question +// about whether ONE step reached the same conclusion twice. +func TestAnotherProducersArtifactIsNotThisStepsVerdict(t *testing.T) { + conn := mustDB(t) + activateInterposed(t, conn, dkt589SiblingSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", "the same findings", unmetPayload) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: check@0's threshold must enter the loop") + } + + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + + // `note` records LAST at ordinal 0, so it — not check@0 — owns the issue's + // newest `findings` below ordinal 1. + claimAndComplete(t, conn, e, "note@0", "an unrelated note", "") + + claimAndComplete(t, conn, e, "check@1", "the same findings", unmetPayload) + + if stepExists(t, conn, "fix@2") { + t.Error("a repeated verdict got through because another producer's " + + "artifact of the same kind was newer; the comparison must be scoped " + + "to the ROUTING STEP's own emitted stream") + } + if got := stepStatus(t, conn, "check@1"); got != db.StepWaitingHuman { + t.Errorf("check@1 = %q, want the repeated-verdict park %q", + got, db.StepWaitingHuman) + } +} diff --git a/internal/engine/dkt590_test.go b/internal/engine/dkt590_test.go new file mode 100644 index 00000000..8bba063f --- /dev/null +++ b/internal/engine/dkt590_test.go @@ -0,0 +1,267 @@ +package engine + +import ( + "database/sql" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-590: a registered workflow's `source_path` and `source_sha256` are two +// columns nothing compared against each other, so the file at the recorded path +// could become a different definition entirely and every read verb went on +// reporting the registered hash beside the stale path. +// +// The observed state: `workflow show investigation --json` reported version 4 +// at 4cb066e3 while the file at its own source_path was version 8 at 6ed74d17, +// on six registered workflows at once — and RUN-40 bound investigation@4 and +// ran it. These tests fix the two dispositions apart: a binding this activation +// MAKES over a drifted source refuses, a binding it INHERITS warns. + +// registerFixtureAtTemp registers the committed fixture from a WRITABLE +// absolute path, and returns that path so a test can drift it. +// +// The path matters twice over: the check declines relative paths on purpose +// (they name no particular file from another cwd), and the file has to be one a +// test may rewrite — the committed fixture is neither. +func registerFixtureAtTemp(t *testing.T, conn *sql.DB) (*model.Workflow, string) { + t.Helper() + registerFixtureSchema(t, conn) + + src, err := os.ReadFile(fixturePath) + testsupport.Must(t, err, "reading fixture: %v", err) + + path := filepath.Join(t.TempDir(), "example-workflow.toml") + err = os.WriteFile(path, src, 0o644) + testsupport.Must(t, err, "writing the fixture copy: %v", err) + + return registerSource(t, conn, src, path), path +} + +// driftSource rewrites the file at path so its bytes no longer hash to what was +// registered, WITHOUT changing what the definition means — the point is the +// hash, and a test that also changed the topology would be testing two things. +func driftSource(t *testing.T, path string) { + t.Helper() + src, err := os.ReadFile(path) + testsupport.Must(t, err, "reading %s: %v", path, err) + err = os.WriteFile(path, append(src, []byte("\n# edited after registration\n")...), 0o644) + testsupport.Must(t, err, "drifting %s: %v", path, err) +} + +// TestActivateRefusesADriftedWorkflowSource is the RUN-40 shape: the file at +// the bound workflow's own source_path is no longer the file that was +// registered, and this activation is the one making the binding. +func TestActivateRefusesADriftedWorkflowSource(t *testing.T) { + conn := mustDB(t) + wf, path := registerFixtureAtTemp(t, conn) + driftSource(t, path) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + _, err := activate(conn, run.ID) + if err == nil { + t.Fatal("activation succeeded over a workflow whose source file no " + + "longer holds the registered bytes; the run would bind and pin a " + + "definition nobody can read at that path") + } + code, ok := CodeOf(err) + if !ok || code != CodeConflict { + t.Errorf("code = %q (engine error: %v), want %s", code, ok, CodeConflict) + } + + // The message has to be actionable without a second command: which + // workflow, which file, and BOTH hashes. + for _, want := range []string{wf.Ref(), path, wf.SourceSHA256} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q:\n%v", want, err) + } + } + + // The fat transaction rolled back: a refusal at binding leaves no steps. + if n := countRows(t, conn, "steps"); n != 0 { + t.Errorf("%d step(s) written by a refused activation, want 0", n) + } +} + +// TestDryRunRefusesADriftedWorkflowSource: `--dry-run` is the real activation +// rolled back, so it must report the same refusal rather than printing a +// roster an operator would then fail to activate. +func TestDryRunRefusesADriftedWorkflowSource(t *testing.T) { + conn := mustDB(t) + _, path := registerFixtureAtTemp(t, conn) + driftSource(t, path) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + _, err := Activate(conn, run.ID, ActivateOptions{NowMS: nowMS, DryRun: true}) + if err == nil { + t.Fatal("--dry-run reported an activation the real verb would refuse") + } + if code, _ := CodeOf(err); code != CodeConflict { + t.Errorf("code = %q, want %s", code, CodeConflict) + } +} + +// TestActivateAcceptsAnUndriftedWorkflowSource is the baseline the refusals are +// measured against: the same setup with the file left alone activates, and +// reports nothing. +func TestActivateAcceptsAnUndriftedWorkflowSource(t *testing.T) { + conn := mustDB(t) + registerFixtureAtTemp(t, conn) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + if len(result.SourceWarnings) != 0 { + t.Errorf("source warnings on a source that matches: %+v", result.SourceWarnings) + } + if result.Run.Status != model.RunActive { + t.Errorf("run status = %q, want %q", result.Run.Status, model.RunActive) + } +} + +// TestActivateWarnsWhenTheSourceFileIsGone: an unreadable source is NOT drift +// and must not be refused as if it were. The registered bytes are intact and +// still reproduce — only their provenance is gone — and refusing would wedge +// every activation in a store holding a workflow whose file was moved, a state +// only the install path can resolve. +func TestActivateWarnsWhenTheSourceFileIsGone(t *testing.T) { + conn := mustDB(t) + wf, path := registerFixtureAtTemp(t, conn) + testsupport.Must(t, os.Remove(path), "removing the source file %s", path) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + if len(result.SourceWarnings) != 1 { + t.Fatalf("source warnings = %+v, want exactly one", result.SourceWarnings) + } + got := result.SourceWarnings[0] + if got.State != string(model.WorkflowSourceUnreadable) { + t.Errorf("state = %q, want %q — a missing file and edited bytes are "+ + "different facts and send an operator to different places", + got.State, model.WorkflowSourceUnreadable) + } + if got.Workflow != wf.Ref() { + t.Errorf("workflow = %q, want %q", got.Workflow, wf.Ref()) + } + if got.CurrentSHA256 != "" { + t.Errorf("current_sha256 = %q on a file that was never read", got.CurrentSHA256) + } +} + +// TestReactivationWarnsRatherThanRefusingOnAnInheritedBinding is RA2/F15 held +// intact: a workflow edited after a run bound it must not reach that run, and +// must not stop it either. Refusing here would wedge every in-flight run the +// moment its workflow was superseded — the ordinary retro-loop bump leaves the +// old version's recorded path holding the new version's bytes. +func TestReactivationWarnsRatherThanRefusingOnAnInheritedBinding(t *testing.T) { + conn := mustDB(t) + wf, path := registerFixtureAtTemp(t, conn) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "first activation: %v", err) + + // The edit lands AFTER the run bound the definition. + driftSource(t, path) + + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "re-activation: %v", err) + if !result.Reactivation { + t.Fatal("premise: the second call must be a re-activation") + } + if len(result.SourceWarnings) != 1 { + t.Fatalf("source warnings = %+v, want exactly one", result.SourceWarnings) + } + got := result.SourceWarnings[0] + if got.State != string(model.WorkflowSourceDrifted) { + t.Errorf("state = %q, want %q", got.State, model.WorkflowSourceDrifted) + } + if got.Workflow != wf.Ref() { + t.Errorf("workflow = %q, want %q", got.Workflow, wf.Ref()) + } + if got.CurrentSHA256 == "" || got.CurrentSHA256 == got.RegisteredSHA256 { + t.Errorf("the warning carries no distinct on-disk hash: %+v", got) + } +} + +// TestCheckWorkflowSourceStates fixes the four verdicts apart at the helper +// both call sites share. They are different facts: an operator restores a +// missing file, re-registers an edited one, and does nothing at all about a +// path that was never absolute. +func TestCheckWorkflowSourceStates(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "wf.toml") + body := []byte("[pipeline]\nname = \"unit\"\nversion = 1\n") + err := os.WriteFile(path, body, 0o644) + testsupport.Must(t, err, "writing %s: %v", path, err) + + registered := "9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08" + if s := CheckWorkflowSource(path, workflow.SHA256(body)); s.State != model.WorkflowSourceMatches { + t.Errorf("identical bytes = %q, want matches (%+v)", s.State, s) + } + s := CheckWorkflowSource(path, registered) + if s.State != model.WorkflowSourceDrifted { + t.Errorf("different bytes = %q, want drifted", s.State) + } + if s.CurrentSHA256 != workflow.SHA256(body) || s.RegisteredSHA256 != registered { + t.Errorf("drift reports one hash, not both: %+v", s) + } + + if s := CheckWorkflowSource(filepath.Join(dir, "gone.toml"), registered); s.State != + model.WorkflowSourceUnreadable { + t.Errorf("a missing file = %q, want unreadable", s.State) + } + if s := CheckWorkflowSource("wf.toml", registered); s.State != + model.WorkflowSourceUnchecked { + t.Errorf("a relative path = %q, want unchecked — resolving it against "+ + "an unrelated cwd would manufacture drift out of a namesake", s.State) + } + if s := CheckWorkflowSource("", registered); s.State != model.WorkflowSourceUnchecked { + t.Errorf("an unrecorded path = %q, want unchecked", s.State) + } +} + +// TestAdoptionDeclinedWarnsRatherThanRefusing: `registration.auto = false` +// means "bind what is REGISTERED, not what the corpus now says", so a registry +// that lags its files is the state the operator ASKED FOR. Refusing over it +// would turn the documented off switch into a wedge the moment the corpus +// moved — the run could not activate at all, and the only remedy would be the +// version adoption the operator just declined. +func TestAdoptionDeclinedWarnsRatherThanRefusing(t *testing.T) { + conn := mustDB(t) + wf, path := registerFixtureAtTemp(t, conn) + testsupport.Must(t, db.SetConfig(conn, 1, db.KeyAutoRegister, "false"), + "turning registration.auto off") + driftSource(t, path) + + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate with adoption declined: %v", err) + + if len(result.SourceWarnings) != 1 { + t.Fatalf("source warnings = %+v, want exactly one", result.SourceWarnings) + } + if got := result.SourceWarnings[0]; got.State != string(model.WorkflowSourceDrifted) || + got.Workflow != wf.Ref() { + t.Errorf("warning = %+v, want drifted on %s", got, wf.Ref()) + } +} diff --git a/internal/engine/dkt591_test.go b/internal/engine/dkt591_test.go new file mode 100644 index 00000000..ccce2e6f --- /dev/null +++ b/internal/engine/dkt591_test.go @@ -0,0 +1,269 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-591 — a `.` input naming a SKIPPED step resolved to a +// different step's artifact of the same kind. RUN-39's spec-doc review +// declared six per-lane `.doc` inputs; five lanes were skipped, one +// revise body ran, and loopProducerRedirect's kind-only scan handed that one +// body's doc to EVERY entry — six byte-identical copies, five of them from a +// step none of those inputs named. +// +// The remedy: `.` resolves to the named step's own artifact (or +// its ordinal/loop substitute when that substitute is UNAMBIGUOUS in the +// definition) or to nothing. With several same-kind loop bodies re-entering +// the same consumer, the redirect refuses to guess. + +// multiReviseDocSrc is RUN-39's shape, minimized to the incident's mechanics: +// six when-gated authoring lanes that all emit `doc`, a review step declaring +// one input entry per lane, and ONE `loop = true` revise body PER LANE — all +// emitting `doc`, all re-entering `review`. Only the issue's own lane ever +// runs; the other five authors are `skipped` and their revise bodies idle. +const multiReviseDocSrc = ` +[pipeline] +name = "multi-revise-doc" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "author-a" +after = [] +executor = "author-a" +when = "labels contains lane:a" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "author-b" +after = [] +executor = "author-b" +when = "labels contains lane:b" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "author-c" +after = [] +executor = "author-c" +when = "labels contains lane:c" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "author-d" +after = [] +executor = "author-d" +when = "labels contains lane:d" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "author-e" +after = [] +executor = "author-e" +when = "labels contains lane:e" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "author-f" +after = [] +executor = "author-f" +when = "labels contains lane:f" +emits = "doc" +inputs = ["issue.body"] + +[[step]] +name = "review" +after = ["author-a", "author-b", "author-c", "author-d", "author-e", "author-f"] +fanout = ["judge-one", "judge-two"] +emits = "findings" +inputs = ["author-a.doc", "author-b.doc", "author-c.doc", "author-d.doc", "author-e.doc", "author-f.doc", "issue.body"] + +[[step]] +name = "synthesize" +after = ["review"] +executor = "synthesize-findings" +emits = "findings" +inputs = ["review.*"] + +[[step]] +name = "reconcile" +after = ["synthesize"] +action = "aggregate" +params = { field = "severity", method = "max", hold_spread = 2, output = "findings" } +inputs = ["synthesize.findings"] +payload = "findings@1" +threshold = { "fix-loop" = "any(severity >= blocker)" } +max_fix_loops = 2 + +[[step]] +name = "revise-a" +executor = "author-a" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" + +[[step]] +name = "revise-b" +executor = "author-b" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" + +[[step]] +name = "revise-c" +executor = "author-c" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" + +[[step]] +name = "revise-d" +executor = "author-d" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" + +[[step]] +name = "revise-e" +executor = "author-e" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" + +[[step]] +name = "revise-f" +executor = "author-f" +emits = "doc" +loop = true +inputs = ["reconcile.findings"] +after_loop = "review" +` + +// activateMultiReviseDoc registers the schema the aggregate needs, the +// multi-revise definition, and activates a run over one issue on lane A. +func activateMultiReviseDoc(t *testing.T, conn *sql.DB) int { + t.Helper() + registerFixtureSchema(t, conn) + registerSource(t, conn, []byte(multiReviseDocSrc), "multi-revise-doc.toml") + issue := createIssue(t, conn, "write the doc", "the issue body", "task", + []string{"lane:a"}) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + return run.ID +} + +// TestSkippedProducerInputDoesNotBindAnotherBodysDoc is RUN-39 verbatim: on +// re-entry, one revise body has run and emitted `doc` at this ordinal, and the +// five inputs naming SKIPPED authors must resolve to NOTHING — not to that +// body's artifact, which none of them named. The one input whose author +// actually authored keeps the author's own draft: with six same-kind bodies +// re-entering `review`, no redirect target is identifiable, so ordinalScoped's +// per-input fallback stands. +func TestSkippedProducerInputDoesNotBindAnotherBodysDoc(t *testing.T) { + conn := mustDB(t) + e := testEngine() + activateMultiReviseDoc(t, conn) + + // Round 0: lane A authors, both judges review, the synthesis carries a + // blocker, and the aggregate routes `fix-loop`. + driveFixtureRound(t, 0) + claimAndComplete(t, conn, e, "author-a@0", authoredDoc, "") + for i := range 2 { + claimAndComplete(t, conn, e, fmt.Sprintf("review@0#%d", i), "findings", "") + } + claimAndComplete(t, conn, e, "synthesize@0", "the synthesis", blockerPayload) + driveAction(t, conn, e, "reconcile@0") + + if !stepExists(t, conn, "review@1#0") { + t.Fatalf("premise: the blocker did not enter the fix loop; review@1 " + + "was never instantiated") + } + // The incident's asymmetry: of the six bodies instantiated at ordinal 1, + // exactly ONE runs — and it is NOT lane A's, so no declared input names it. + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "revise-d@1", revisedDoc, "") + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "review@1#0"), nowMS) + testsupport.Must(t, err, "assembling review@1#0's bundle: %v", err) + + var docs []string + for _, in := range bundle.Inputs { + if in.ProducerStep == "revise-d@1" { + t.Errorf("an input bound revise-d@1's %s artifact — no declared "+ + "input names revise-d, and a kind-only redirect must not "+ + "guess between six same-kind bodies", in.Kind) + } + if strings.Contains(in.Body, revisedDoc) { + t.Errorf("input %q from %q carries the unrelated body's revised "+ + "doc", in.Artifact, in.ProducerStep) + } + if in.Kind == "doc" { + docs = append(docs, in.ProducerStep) + } + } + // The five skipped lanes resolve to nothing, so exactly ONE doc input + // survives: lane A's own draft, bound by §7.4's per-input fallback. + if len(docs) != 1 || docs[0] != "author-a@0" { + t.Errorf("the bundle binds doc inputs from %v, want exactly "+ + "[author-a@0] — a skipped author's input resolves to NOTHING", + docs) + } +} + +// TestClusterScopedBodiesStillRedirect pins what DKT-591's refusal must NOT +// take down: two same-kind bodies whose serves-scoped clusters re-enter +// DISJOINT chains (DKT-544's shape) are unambiguous for any one consumer — +// only one body's `after_loop` chain contains it — so the re-entered gate +// still reads its own cluster's revision, never the original draft. +func TestClusterScopedBodiesStillRedirect(t *testing.T) { + conn := mustDB(t) + e := testEngine() + activateInterposed(t, conn, clusterSrc) + + claimAndComplete(t, conn, e, "draft@0", authoredDoc, "") + claimAndComplete(t, conn, e, "prd-gate@0", "findings", blockedPayload) + + if !stepExists(t, conn, "prd-fix@1") { + t.Fatalf("premise: cluster A's entry did not instantiate prd-fix@1") + } + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "prd-fix@1", revisedDoc, "") + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "prd-gate@1"), nowMS) + testsupport.Must(t, err, "assembling prd-gate@1's bundle: %v", err) + + var sawDoc bool + for _, in := range bundle.Inputs { + if in.Kind != "doc" { + continue + } + sawDoc = true + if in.ProducerStep != "prd-fix@1" || in.Body != revisedDoc { + t.Errorf("prd-gate@1's draft.doc bound %q from %q, want the "+ + "revised doc from prd-fix@1 — with disjoint clusters the "+ + "redirect target is unambiguous and must still fire", + in.Body, in.ProducerStep) + } + } + if !sawDoc { + t.Error("prd-gate@1 bound no doc input at all; the unambiguous " + + "cluster redirect must not have been refused") + } +} diff --git a/internal/engine/dkt594_test.go b/internal/engine/dkt594_test.go new file mode 100644 index 00000000..40b21406 --- /dev/null +++ b/internal/engine/dkt594_test.go @@ -0,0 +1,281 @@ +package engine + +import ( + "database/sql" + "os" + "path/filepath" + "strconv" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-594 — the two facts a post-mortem reader had to reconstruct by hand. +// +// The corpus took 41 commits in 4.3 days (9.55/day against 0.86/day the week +// before). Every analyst reading RUN-32 went to git to learn that its +// `ui-change@8` was five registered versions behind before they would trust a +// finding from it, because the run report carried no such field. And RUN-39, +// which repinned mid-flight, could only be read by correlating repin event seqs +// (5375/5376) against step ids (STEP-1350/1353) by hand — the `pins` table +// holds the CURRENT agreement and completed steps' rows are never rewritten, so +// nothing said which bytes a finished step actually consumed. + +// registerVersion registers the fixture's TOML at another `[pipeline].version`, +// through the same parse-validate-lint path `workflow register` uses — which is +// what a corpus commit that edits a workflow produces. +func registerVersion(t *testing.T, conn *sql.DB, version int) *model.Workflow { + t.Helper() + src, err := os.ReadFile(fixturePath) + testsupport.Must(t, err, "reading fixture: %v", err) + bumped := strings.Replace( + string(src), "version = 1", "version = "+strconv.Itoa(version), 1) + if bumped == string(src) { + t.Fatalf("the fixture's `version = 1` line moved; this helper cannot bump it") + } + return registerSource(t, conn, []byte(bumped), fixturePath) +} + +// pinnedWorkflow finds the report's staleness row for one pinned ref. +func pinnedWorkflow( + t *testing.T, r *RunReport, ref string, +) PinnedWorkflowStaleness { + t.Helper() + for _, w := range r.PinnedWorkflows { + if w.Ref == ref { + return w + } + } + t.Fatalf("the report carries no staleness row for %s: %+v", ref, r.PinnedWorkflows) + return PinnedWorkflowStaleness{} +} + +// attemptFor finds one step's attempt row by instance. +func attemptFor(t *testing.T, r *RunReport, instance string) StepAttempt { + t.Helper() + for _, a := range r.Attempts { + if a.Instance == instance { + return a + } + } + t.Fatalf("the report carries no attempt row for %s", instance) + return StepAttempt{} +} + +// TestReportCountsCorpusVersionsSinceActivation is criterion 1: the +// subtraction an analyst was doing in git is in the document. +func TestReportCountsCorpusVersionsSinceActivation(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + // The corpus moves twice under the run, exactly as `just activate` does. + registerVersion(t, conn, 2) + registerVersion(t, conn, 3) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + got := pinnedWorkflow(t, report, "standard-change@1") + if got.PinnedVersion != 1 { + t.Errorf("pinned_version = %d, want 1", got.PinnedVersion) + } + if got.CurrentVersion != 3 { + t.Errorf("current_version = %d, want 3 — the registry's binding head", got.CurrentVersion) + } + if got.Behind != 2 { + t.Errorf("behind = %d, want 2; without it a reader has to derive the "+ + "staleness from git before trusting anything in this report", got.Behind) + } +} + +// TestCurrentPinReportsZeroBehind is the falsifier: a report that always said +// "behind" would pass the test above and be useless. +func TestCurrentPinReportsZeroBehind(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + got := pinnedWorkflow(t, report, "standard-change@1") + if got.Behind != 0 || got.CurrentVersion != 1 { + t.Errorf("a run pinning the registry's head reports behind=%d current=%d, "+ + "want 0 and 1", got.Behind, got.CurrentVersion) + } +} + +// TestRetiredVersionsAreNotCorpusAdvance: a version registered and then RETIRED +// is not somewhere the run could have gone. Binding filters retirement out +// first (bindableDefinitions), so a staleness count that included it would send +// an operator to chase a version nothing can bind. +func TestRetiredVersionsAreNotCorpusAdvance(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + registerVersion(t, conn, 2) + _, err := db.DeprecateWorkflow(conn, run.ProjectID, "standard-change", 2, nowMS) + testsupport.Must(t, err, "DeprecateWorkflow: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + got := pinnedWorkflow(t, report, "standard-change@1") + if got.Behind != 0 { + t.Errorf("behind = %d over a RETIRED version, want 0", got.Behind) + } + if got.CurrentVersion != 1 { + t.Errorf("current_version = %d, want 1 — the highest version that still binds", + got.CurrentVersion) + } +} + +// TestARunThatNeverRepinnedHasNoEpochs. One agreement is not a timeline: the +// `pins` table already states what every step of such a run read, and a +// `pin_epoch: 1` on every step of every report in the store would be a column +// of constants. +func TestARunThatNeverRepinnedHasNoEpochs(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if len(report.PinEpochs) != 0 { + t.Errorf("a run that never repinned carries %d epoch(s): %+v", + len(report.PinEpochs), report.PinEpochs) + } + for _, a := range report.Attempts { + if a.PinEpoch != 0 { + t.Errorf("%s carries pin_epoch %d on a run with one agreement", + a.Instance, a.PinEpoch) + } + } +} + +// repinFixture is the RUN-39 shape: one pinned contract that a corpus install +// is about to replace mid-run. It returns the activated run, the pinned ref, +// the hash the run froze, and the config root the ref resolves under. +func repinFixture(t *testing.T, conn *sql.DB) ( + run *model.Run, ref, before, root string, +) { + t.Helper() + run, _ = activatedRun(t, conn) + root = t.TempDir() + ref = "contracts/implement.md" + before = pinAFile(t, conn, run.ID, root, ref, "BEFORE\n") + return run, ref, before, root +} + +// TestCompletedStepReportsTheAgreementItRanUnder is criterion 2's first half: +// a step that finished BEFORE the repin says so, even though the pins table +// has since moved on. +func TestCompletedStepReportsTheAgreementItRanUnder(t *testing.T) { + conn := mustDB(t) + run, ref, before, root := repinFixture(t, conn) + + e := testEngine() + implID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement@0: %v", err) + testsupport.Must(t, e.CompleteStep(conn, implID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("done"), NowMS: nowMS, + }), "complete implement@0") + + // The corpus install, and the operator's recovery. + testsupport.Must(t, os.WriteFile( + filepath.Join(root, ref), []byte("AFTER\n"), 0o644), "replacing the contract") + outcome, err := repinRunIn(conn, run.ID, "corpus install 2026-08-23", nowMS+1000, + []string{root}) + testsupport.Must(t, err, "repinRunIn: %v", err) + if len(outcome.Repinned) != 1 { + t.Fatalf("repinned %d pin(s), want 1: %+v", len(outcome.Repinned), outcome.Repinned) + } + + report, err := LoadRunReport(conn, run.ID, nowMS+2000) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if len(report.PinEpochs) != 2 { + t.Fatalf("the timeline has %d epoch(s), want activation + one repin: %+v", + len(report.PinEpochs), report.PinEpochs) + } + if report.PinEpochs[0].Origin != PinEpochActivation { + t.Errorf("epoch 1 origin = %q, want %q", + report.PinEpochs[0].Origin, PinEpochActivation) + } + repin := report.PinEpochs[1] + if repin.Epoch != 2 || repin.Origin != PinEpochRepin { + t.Errorf("epoch 2 = %d %q, want 2 %q", repin.Epoch, repin.Origin, PinEpochRepin) + } + if repin.Reason != "corpus install 2026-08-23" { + t.Errorf("epoch 2 reason = %q, want the operator's own", repin.Reason) + } + if len(repin.Changes) != 1 || repin.Changes[0].Ref != ref || + repin.Changes[0].OldSHA256 != before { + t.Errorf("epoch 2 changes = %+v, want %s moving off %s", + repin.Changes, ref, before) + } + + // THE ACCEPTANCE CRITERION. The step completed before the repin's seq, so + // the bytes it consumed are epoch 1's — which is the fact RUN-39's analysts + // recovered by hand from seq numbers. + if got := attemptFor(t, report, "implement@0"); got.PinEpoch != 1 { + t.Errorf("implement@0 reports pin_epoch %d, want 1 — it recorded before "+ + "the repin, so it consumed the ORIGINAL bytes", got.PinEpoch) + } +} + +// TestStepRunAfterARepinReportsTheNewAgreement is the other half, and the +// falsifier for the test above: an epoch that were merely "the run's first" +// would pass there and be wrong here. +func TestStepRunAfterARepinReportsTheNewAgreement(t *testing.T) { + conn := mustDB(t) + run, ref, _, root := repinFixture(t, conn) + + testsupport.Must(t, os.WriteFile( + filepath.Join(root, ref), []byte("AFTER\n"), 0o644), "replacing the contract") + _, err := repinRunIn(conn, run.ID, "corpus install", nowMS+1000, []string{root}) + testsupport.Must(t, err, "repinRunIn: %v", err) + + e := testEngine() + implID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implID, ClaimOptions{Owner: "w", NowMS: nowMS + 2000}) + testsupport.Must(t, err, "claim implement@0: %v", err) + testsupport.Must(t, e.CompleteStep(conn, implID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("done"), NowMS: nowMS + 2000, + }), "complete implement@0") + + report, err := LoadRunReport(conn, run.ID, nowMS+3000) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if got := attemptFor(t, report, "implement@0"); got.PinEpoch != 2 { + t.Errorf("implement@0 reports pin_epoch %d, want 2 — it was claimed AFTER "+ + "the repin, so it consumed the new bytes", got.PinEpoch) + } +} + +// TestAStepThatHasNotRunCarriesNoEpoch. A pending step's agreement is whatever +// the pins table holds when it is finally claimed; stamping it with an epoch +// would answer a question about the future with a fact about the past. +func TestAStepThatHasNotRunCarriesNoEpoch(t *testing.T) { + conn := mustDB(t) + run, ref, _, root := repinFixture(t, conn) + + testsupport.Must(t, os.WriteFile( + filepath.Join(root, ref), []byte("AFTER\n"), 0o644), "replacing the contract") + _, err := repinRunIn(conn, run.ID, "corpus install", nowMS+1000, []string{root}) + testsupport.Must(t, err, "repinRunIn: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS+2000) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + got := attemptFor(t, report, "implement@0") + if got.Status != db.StepReady && got.Status != db.StepPending { + t.Fatalf("implement@0 is %q; this test needs a step that has not run", got.Status) + } + if got.PinEpoch != 0 { + t.Errorf("a %s step reports pin_epoch %d, want none", got.Status, got.PinEpoch) + } +} diff --git a/internal/engine/dkt609_test.go b/internal/engine/dkt609_test.go new file mode 100644 index 00000000..bfcc661b --- /dev/null +++ b/internal/engine/dkt609_test.go @@ -0,0 +1,184 @@ +package engine + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-609: a corpus commit renamed a workflow, the four registered versions of +// the old name stayed live in the store, and the first issue carrying the old +// label failed activation with "matches 2 workflows: security-change@17, +// security-load-bearing@12". The refusal named both candidates and said +// nothing about the fact that one of them had had no file on disk for weeks — +// which is what turned a `docket workflow deprecate` into a git archaeology +// session. +// +// These tests fix the three verdicts apart and pin the annotation onto the +// refusal that needed it. + +// TestWorkflowOriginVerdicts separates present, orphaned, and unchecked at the +// index every caller shares. They are different facts: a name still declared +// somewhere is ordinary, a name declared nowhere is a deprecation candidate, +// and a name nobody looked for is neither. +func TestWorkflowOriginVerdicts(t *testing.T) { + root := t.TempDir() + err := os.MkdirAll(filepath.Join(root, "workflows"), 0o755) + testsupport.Must(t, err, "creating the workflows directory: %v", err) + live := filepath.Join(root, "workflows", "gone-renamed.toml") + err = os.WriteFile(live, []byte(goneRenamedWorkflowSrc), 0o644) + testsupport.Must(t, err, "writing the definition: %v", err) + + scan, err := scanConfigDirs([]string{root}) + testsupport.Must(t, err, "scanning the config root: %v", err) + index := newWorkflowOriginIndex(scan) + + if !index.Scanned() { + t.Fatal("premise: a root that exists must count as scanned") + } + + present := index.Status("gone-renamed") + if present.State != model.WorkflowOriginPresent { + t.Errorf("a declared name = %q, want %q (%+v)", + present.State, model.WorkflowOriginPresent, present) + } + // Compared against the CANONICALIZED path: the scan resolves its roots + // (macOS puts /tmp behind a symlink), so the recorded path is the resolved + // one and comparing against the unresolved literal would fail on the + // platform this repo is developed on. + wantLive, err := filepath.EvalSymlinks(live) + testsupport.Must(t, err, "resolving %s: %v", live, err) + if present.Path != wantLive { + t.Errorf("the present verdict names %q, want the file that declares "+ + "the name, %q", present.Path, wantLive) + } + + // The rename's other side: "gone" is registered nowhere on this disk. + orphan := index.Status("gone") + if orphan.State != model.WorkflowOriginOrphaned { + t.Errorf("a name no file declares = %q, want %q (%+v)", + orphan.State, model.WorkflowOriginOrphaned, orphan) + } + if len(orphan.Roots) == 0 { + t.Error("the orphaned verdict names no roots, so a reader cannot tell " + + "WHERE the name was looked for") + } + if !index.Orphaned("gone") || index.Orphaned("gone-renamed") { + t.Error("Orphaned() disagrees with Status()") + } + + // UNCHECKED is not "clean". A store with no instance-config root has had + // nothing looked at, and reporting orphans from it would call every + // registration on the machine an orphan on the strength of having looked + // nowhere. + var none *WorkflowOriginIndex + if s := none.Status("gone-renamed"); s.State != model.WorkflowOriginUnchecked { + t.Errorf("a nil index = %q, want %q", s.State, model.WorkflowOriginUnchecked) + } + if none.Orphaned("gone-renamed") { + t.Error("a nil index reports an orphan; nothing was checked") + } + empty := newWorkflowOriginIndex(nil) + if empty.Scanned() { + t.Error("an index over no root reports itself scanned") + } + if s := empty.Status("gone"); s.State != model.WorkflowOriginUnchecked { + t.Errorf("no root = %q, want %q", s.State, model.WorkflowOriginUnchecked) + } +} + +// TestWorkflowOriginIsPerNameNotPerVersion: a SUPERSEDED version is not an +// orphan while its name is still declared. Version bumps are the ordinary way +// the corpus evolves, and a check that flagged every superseded row would light +// up the whole registry — noise that would cost exactly the rename case it +// exists to find. +func TestWorkflowOriginIsPerNameNotPerVersion(t *testing.T) { + root := t.TempDir() + err := os.MkdirAll(filepath.Join(root, "workflows"), 0o755) + testsupport.Must(t, err, "creating the workflows directory: %v", err) + // Only version 2 of "gone" is on disk; version 1 was registered from the + // file this one replaced. + bumped := strings.Replace(goneWorkflowSrc, "version = 1", "version = 2", 1) + err = os.WriteFile(filepath.Join(root, "workflows", "gone.toml"), []byte(bumped), 0o644) + testsupport.Must(t, err, "writing the bumped definition: %v", err) + + scan, err := scanConfigDirs([]string{root}) + testsupport.Must(t, err, "scanning the config root: %v", err) + if index := newWorkflowOriginIndex(scan); index.Orphaned("gone") { + t.Error("a name whose file was BUMPED reads as orphaned; the verdict " + + "is per name, and every superseded version of a live name would " + + "otherwise be reported as a deprecation candidate") + } +} + +// TestDryRunRefusalNamesTheOrphan is DKT-609's second acceptance criterion, on +// the verb the RUN-45 operator actually ran: `run activate --dry-run` refuses +// the ambiguity and its candidate list now says WHICH side of it has no +// definition left. +// +// Both directions are asserted. The orphan carries the annotation AND the live +// candidate does not — an annotation on both would be no distinction at all, +// which is the state this issue is closing. +func TestDryRunRefusalNamesTheOrphan(t *testing.T) { + conn, configDir := configRepo(t) + path := writeConfigFile(t, configDir, "workflows/gone.toml", goneWorkflowSrc) + + first := createIssue(t, conn, "first", "body", "task", nil) + firstRun := startRun(t, conn, first) + _, err := activate(conn, firstRun.ID) + testsupport.Must(t, err, "registering gone@1 through the first activation: %v", err) + + // The rename: the old file leaves the corpus, the new name arrives, and + // the OLD REGISTRATION IS UNTOUCHED — a registration is a row, not a file. + testsupport.Must(t, os.Remove(path), "deleting gone.toml: %v", err) + writeConfigFile(t, configDir, "workflows/gone-renamed.toml", goneRenamedWorkflowSrc) + + issue := createIssue(t, conn, "the first issue after the rename", "a body", "task", nil) + run := startRun(t, conn, issue) + + _, err = Activate(conn, run.ID, ActivateOptions{NowMS: nowMS, DryRun: true}) + if err == nil { + t.Fatal("the dry run bound an issue two names match; it must refuse " + + "exactly as the real activation does") + } + assertRenamePlusBumpWedge(t, err, issue, wedgeCandidatesOrphaned) + + msg := err.Error() + if strings.Contains(msg, "gone-renamed@2"+orphanAnnotation) { + t.Errorf("the LIVE candidate is annotated as an orphan too, so the "+ + "refusal distinguishes nothing: %s", msg) + } + // The remedy, in the message, so the next operator does not have to go + // looking for which verb clears a stranded registration. + if !strings.Contains(msg, "docket workflow deprecate") || + !strings.Contains(msg, "docket workflow list --orphans") { + t.Errorf("the refusal names no remedy for the orphan it found: %s", msg) + } +} + +// TestRefusalIsUnannotatedWithoutAConfigRoot is the other half of the honesty +// rule, asserted on the refusal itself rather than only on the index: with no +// root to scan, the message is EXACTLY the pre-DKT-609 one. Silence is what +// "nothing was checked" has to look like — a hint about deprecating a +// registration nobody verified is worse than no hint. +func TestRefusalIsUnannotatedWithoutAConfigRoot(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(goneWorkflowSrc), "gone.toml") + registerSource(t, conn, []byte(goneRenamedWorkflowSrc), "gone-renamed.toml") + + issue := createIssue(t, conn, "wedged with no corpus on disk", "a body", "task", nil) + run := startRun(t, conn, issue) + + _, err := activate(conn, run.ID) + if err == nil { + t.Fatal("premise: two names matching one issue must refuse") + } + assertRenamePlusBumpWedge(t, err, issue, wedgeCandidatesUnchecked) + if strings.Contains(err.Error(), "ORPHANED") { + t.Errorf("an unscanned store reported an orphan: %s", err.Error()) + } +} diff --git a/internal/engine/dkt725_test.go b/internal/engine/dkt725_test.go new file mode 100644 index 00000000..ef48108b --- /dev/null +++ b/internal/engine/dkt725_test.go @@ -0,0 +1,116 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-725: an operator's steering had no channel that reliably reached a NEW +// round's rendered packet. Observed on RUN-51/AGT-643: after a 3/3 vote +// rejection the operator authorized another fix round and recorded the +// judge-converged remedy as an issue COMMENT; the resulting fix@9 packet +// (STEP-2489) carried zero bytes of it. Comments are not a context source +// (§6.6's five-source rule), a mid-run description edit never renders either +// (`body_snapshot` froze at activation, §9 item 5), and the `fix-round` note +// itself died on the SUPERSEDED park's row — DKT-247's == RESOLUTION section +// is scoped to a step's own routing record, and the new round's rows had none. +// +// The fix: an authorized entry stamps its note onto the round's freshly +// instantiated pending rows as their entering routing record +// (stampEntryRouting), so the standing own-row rendering carries it into +// every packet of the round the operator paid for. + +// TestFixRoundNoteReachesTheNewRoundPackets drives the dkt587 threshold +// fixture to its bound, authorizes one more round with a steering note, and +// asserts the note renders in the NEW round's packets — the fix body's and +// the re-instantiated routing step's. +func TestFixRoundNoteReachesTheNewRoundPackets(t *testing.T) { + conn := mustDB(t) + _, _ = activateInterposed(t, conn, dkt587ThresholdSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", roundReport(0), unmetPayload) + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + // Parked at the bound; the operator authorizes one more round WITH the + // converged remedy as the note. + if got := stepStatus(t, conn, "check@2"); got != db.StepWaitingHuman { + t.Fatalf("check@2 = %q, want %q before the grant", got, db.StepWaitingHuman) + } + const remedy = "route the validator through the deny-by-default table; " + + "the panel converged on closing the fail-open path structurally" + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "check@2"), + ResolveFixRound, remedy, nowMS) + testsupport.Must(t, err, "resolving --as fix-round: %v", err) + + // The note is VERIFIED IN THE RENDERED PACKET, not inferred from rows: + // reading the packet directly is how the original gap was confirmed. + for _, instance := range []string{"fix@3", "check@3"} { + result, err := RenderStep( + conn, stepIDByInstance(t, conn, instance), "", nowMS) + testsupport.Must(t, err, "rendering %s: %v", instance, err) + if !strings.Contains(result.Packet, remedy) { + t.Errorf("%s's packet does not carry the fix-round note:\n%s", + instance, result.Packet) + } + if !strings.Contains(result.Packet, "== RESOLUTION fix-loop") { + t.Errorf("%s's packet carries no == RESOLUTION fix-loop section:\n%s", + instance, result.Packet) + } + } + + // The stamp changes routing ONLY — the rows stay pending and claimable, + // exactly as a DKT-247 retry's ruled-on pending row does. + if got := stepStatus(t, conn, "fix@3"); got != db.StepPending { + t.Errorf("fix@3 = %q after the stamp, want %q", got, db.StepPending) + } + + // And the round proceeds normally: the stamped body claims, completes, + // and its OWN verdict overwrites the entering record at routing time. + driveFixtureRound(t, 3) + claimAndComplete(t, conn, e, "fix@3", "the third fix", "") + raw := stepRoutingRaw(t, conn, "fix@3") + if strings.Contains(raw, remedy) { + t.Errorf("fix@3 routing = %q after completion; its own verdict must "+ + "replace the entering stamp", raw) + } +} + +// TestFixRoundWithoutNoteLeavesTheNewRoundUnchanged pins the other half: an +// authorization carrying no note stamps nothing, so a note-less fix-round's +// packets are byte-identical to what they always were. +func TestFixRoundWithoutNoteLeavesTheNewRoundUnchanged(t *testing.T) { + conn := mustDB(t) + _, _ = activateInterposed(t, conn, dkt587ThresholdSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "check@0", roundReport(0), unmetPayload) + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "fix@1", "the fix", "") + claimAndComplete(t, conn, e, "check@1", roundReport(1), unmetPayload) + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix", "") + claimAndComplete(t, conn, e, "check@2", roundReport(2), unmetPayload) + + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "check@2"), + ResolveFixRound, "", nowMS) + testsupport.Must(t, err, "resolving --as fix-round without a note: %v", err) + + if raw := stepRoutingRaw(t, conn, "fix@3"); raw != "" { + t.Errorf("fix@3 routing = %q after a note-less fix-round, want empty", raw) + } + result, err := RenderStep(conn, stepIDByInstance(t, conn, "fix@3"), "", nowMS) + testsupport.Must(t, err, "rendering fix@3: %v", err) + if strings.Contains(result.Packet, "== RESOLUTION") { + t.Errorf("fix@3's packet grew a == RESOLUTION section with no note:\n%s", + result.Packet) + } +} diff --git a/internal/engine/dkt726_test.go b/internal/engine/dkt726_test.go new file mode 100644 index 00000000..c9c4bc4e --- /dev/null +++ b/internal/engine/dkt726_test.go @@ -0,0 +1,183 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-726 — the refusal `retry` already made on a MATERIALIZED vote-minted +// hold, made on a plain workflow-declared `type="vote"` step too. +// +// The guard that refuses a parked vote was scoped to `step.Materialized`, so it +// saw the engine-minted `reconcile-held@N#M` clusters and nothing else. A +// workflow's own `security-vote@8`, carrying its own tribunal proposal, fell +// through to the generic retry path: attempt budget and lease reset, step back +// to `pending`, and the next `next` re-read the SAME proposal — the idempotency +// key is (run, issue, instance), so no second ballot opens — announced the +// identical verdict and routed to the identical place. Observed on RUN-51 +// STEP-2433 with DKT-V256 rejected 3/3, at the cost of a full +// run-pause/run-resume cycle. +// +// TestRetryRefusesOnADecidedVoteStep is the fix, over BOTH terminal statuses: +// an approved tally is exactly as sticky as a rejected one, and re-reading it +// is exactly as much of a no-op. +func TestRetryRefusesOnADecidedVoteStep(t *testing.T) { + for _, status := range []model.ProposalStatus{ + model.ProposalStatusRejected, + model.ProposalStatusApproved, + // §8.4's manual commit is a decision like any other. + model.ProposalStatusCommitted, + } { + t.Run(string(status), func(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.6", "medium") + e := testEngine() + + step, spec := seedVoteStep(t, conn) + if step.Materialized { + t.Fatal("premise: this must be a PLAIN workflow-declared vote " + + "step — the materialized case was already guarded") + } + id, err := OpenVoteProposal(conn, step, spec, nowMS) + testsupport.Must(t, err, "OpenVoteProposal: %v", err) + _, err = conn.Exec( + `UPDATE proposals SET status = ? WHERE id = ?`, status, id) + testsupport.Must(t, err, "deciding the proposal: %v", err) + + err = e.ResolveStep(conn, step.ID, ResolveRetry, "", nowMS) + if err == nil { + t.Fatal("retry was accepted on a vote step whose proposal is " + + "already decided; it re-tallies the same casts and routes " + + "to the same place, which is the behaviour being removed") + } + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("error code = %q, want %q", code, CodeValidation) + } + + // The refusal must NAME the proposal: an operator told only that a + // vote decided this cannot check which vote without guessing. + if want := model.FormatProposalID(id); !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name the proposal %s: %v", want, err) + } + if !strings.Contains(err.Error(), string(status)) { + t.Errorf("the refusal does not name the status %q: %v", status, err) + } + + // And it must name the verbs that DO move the step — `fix-round` + // first, because it is the one RUN-51 actually wanted. + for _, verb := range []string{ + ResolveFixRound, ResolveOverridePass, ResolveSkip, ResolveAbandonIssue, + } { + if !strings.Contains(err.Error(), verb) { + t.Errorf("the refusal does not offer --as %s: %v", verb, err) + } + } + + // Nothing moved. A refusal that had already reset the budget would + // be the same wasted cycle wearing an error message. + after, err := db.GetStep(conn, step.ID) + testsupport.Must(t, err, "re-reading the step: %v", err) + if after.Status != db.StepPending { + t.Errorf("step status = %q after the refusal, want it "+ + "untouched at %q", after.Status, db.StepPending) + } + }) + } +} + +// TestRetryStillWorksOnAnUndecidedVoteStep is the other half, and the one the +// narrow condition exists to protect. +// +// R11 offers a resolution on a vote step WHATEVER its status precisely so a run +// is not hostage to a quorum that never arrives. A refusal keyed to "this is a +// vote step" rather than "its proposal is decided" would take the least +// destructive verb away from exactly that case. Both shapes of undecided are +// covered: a ballot open with nobody having cast, and a step whose ballot was +// never opened at all (an interposed vote skipped before phase 2 — DKT-470's +// recovery path). +func TestRetryStillWorksOnAnUndecidedVoteStep(t *testing.T) { + t.Run("open proposal", func(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.6", "medium") + e := testEngine() + + step, spec := seedVoteStep(t, conn) + id, err := OpenVoteProposal(conn, step, spec, nowMS) + testsupport.Must(t, err, "OpenVoteProposal: %v", err) + + proposal, err := db.GetProposal(conn, id) + testsupport.Must(t, err, "GetProposal: %v", err) + if proposal.Status != model.ProposalStatusOpen { + t.Fatalf("premise: proposal is %q, want %q", + proposal.Status, model.ProposalStatusOpen) + } + + testsupport.Must(t, e.ResolveStep(conn, step.ID, ResolveRetry, "", nowMS), + "retry was refused on a vote whose proposal is still OPEN, where "+ + "there is no decided tally to re-read: %v", nil) + }) + + t.Run("no proposal opened", func(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.6", "medium") + e := testEngine() + + step, _ := seedVoteStep(t, conn) + proposalID, err := findVoteProposal(conn, step) + testsupport.Must(t, err, "findVoteProposal: %v", err) + if proposalID != 0 { + t.Fatalf("premise: a proposal (%d) already exists", proposalID) + } + + testsupport.Must(t, e.ResolveStep(conn, step.ID, ResolveRetry, "", nowMS), + "retry was refused on a vote step that never opened a ballot: %v", nil) + }) +} + +// TestReadStepVoteOutcomeMatchesTheSpecKeyedReader pins the refactor DKT-726 +// needed: the step-keyed entry point is the SAME read as ReadVoteOutcome, not a +// second one that could drift. `step resolve` reaches it without a pinned spec +// — R11 offers a resolution before the definition is loaded, and a materialized +// step's minted name is never declared — so the two must agree everywhere they +// both apply. +func TestReadStepVoteOutcomeMatchesTheSpecKeyedReader(t *testing.T) { + for _, status := range []model.ProposalStatus{ + model.ProposalStatusOpen, + model.ProposalStatusApproved, + model.ProposalStatusRejected, + model.ProposalStatusCommitted, + } { + t.Run(string(status), func(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.6", "medium") + + step, spec := seedVoteStep(t, conn) + id, err := OpenVoteProposal(conn, step, spec, nowMS) + testsupport.Must(t, err, "OpenVoteProposal: %v", err) + _, err = conn.Exec( + `UPDATE proposals SET status = ? WHERE id = ?`, status, id) + testsupport.Must(t, err, "setting the proposal status: %v", err) + + viaSpec, err := ReadVoteOutcome(conn, step, spec) + testsupport.Must(t, err, "ReadVoteOutcome: %v", err) + viaStep, err := ReadStepVoteOutcome(conn, step) + testsupport.Must(t, err, "ReadStepVoteOutcome: %v", err) + + if viaSpec == nil || viaStep == nil { + t.Fatalf("one reader reported nothing: spec=%v step=%v", viaSpec, viaStep) + } + // Field-by-field: Score is a pointer, and two reads of the same + // row hand back two pointers to equal values. + if viaSpec.ProposalID != viaStep.ProposalID || + viaSpec.Status != viaStep.Status || + viaSpec.Verdict != viaStep.Verdict { + t.Errorf("the two readers disagree: spec-keyed %+v, "+ + "step-keyed %+v", *viaSpec, *viaStep) + } + }) + } +} diff --git a/internal/engine/dkt733_test.go b/internal/engine/dkt733_test.go new file mode 100644 index 00000000..6827e7c2 --- /dev/null +++ b/internal/engine/dkt733_test.go @@ -0,0 +1,165 @@ +package engine + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-733 — vote-seat usage under-reporting. RUN-51's report said "45 of 57 +// seat(s) reported spend — 12 reported NOTHING" and nothing anywhere said +// WHICH twelve or via which seating path, so the backfill verb built to close +// exactly this gap (`vote backfill-usage`, DKT-115) could not be aimed and +// budget-cap enforcement read an understated ledger. These tests pin the +// per-seat enumeration and its path labels. + +// TestSilentVoteSeatsNameTheirSeatingPath: a run with one silent vote-step +// seat and one silent conversational-gate seat enumerates BOTH, each labeled +// with the path that minted its proposal, and the seat that reported spend is +// not listed. +func TestSilentVoteSeatsNameTheirSeatingPath(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + configureHoldTally(t, conn, "dkt733-panel", "seat-a,seat-b") + driveToReconcile(t, conn, e, clusteredPayload) + // The first invocation that observes the held vote step ready opens its + // proposal (§8.1 phase 2). + nextRun(t, conn, e) + held := heldInstances(t, conn) + if len(held) == 0 { + t.Fatal("nothing held") + } + voteStepID := heldProposalID(t, conn, e, held[0]) + + // A vote-step seat casts and reports nothing. + _, err := db.CastVote(conn, &model.Vote{ + ProposalID: voteStepID, VoterName: "seat-a", VoterRole: "security", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + }) + testsupport.Must(t, err, "CastVote (vote-step): %v", err) + + // A conversational gate naming the run: one silent seat, one loud one. + convID, err := db.CreateProposal(conn, &model.Proposal{ + ProjectID: 1, + Description: "activation panel for " + model.FormatRunID(run.ID), + Rationale: "conversational gate", + Criticality: model.CriticalityMedium, + Threshold: 0.5, RequiredVoters: 3, + Status: model.ProposalStatusOpen, CreatedBy: "conductor", + }) + testsupport.Must(t, err, "CreateProposal: %v", err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: convID, VoterName: "gate-seat-quiet", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + }) + testsupport.Must(t, err, "CastVote (conversational, silent): %v", err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: convID, VoterName: "gate-seat-loud", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + Usage: map[string]float64{"tokens": 7}, + }) + testsupport.Must(t, err, "CastVote (conversational, loud): %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if c := report.VoteUsageCoverage; c.Silent() != 2 { + t.Fatalf("coverage = %+v, want 2 silent seats", c) + } + if len(report.SilentVoteSeats) != 2 { + t.Fatalf("silent_vote_seats = %+v, want exactly the two quiet casts", + report.SilentVoteSeats) + } + + byVoter := make(map[string]SilentVoteSeat, len(report.SilentVoteSeats)) + for _, s := range report.SilentVoteSeats { + byVoter[s.Voter] = s + } + if s, ok := byVoter["seat-a"]; !ok || s.Path != SeatPathVoteStep || + s.Proposal != model.FormatProposalID(voteStepID) || s.Role != "security" { + t.Errorf("vote-step silent seat = %+v, want seat-a (security) on %s "+ + "via %q", byVoter["seat-a"], + model.FormatProposalID(voteStepID), SeatPathVoteStep) + } + if s, ok := byVoter["gate-seat-quiet"]; !ok || + s.Path != SeatPathConversationalGate || + s.Proposal != model.FormatProposalID(convID) { + t.Errorf("conversational silent seat = %+v, want gate-seat-quiet on "+ + "%s via %q", byVoter["gate-seat-quiet"], + model.FormatProposalID(convID), SeatPathConversationalGate) + } + if _, ok := byVoter["gate-seat-loud"]; ok { + t.Error("the seat that reported spend is listed as silent") + } +} + +// TestSilentReapAckSeatIsAConversationalGate: a reap-ack ballot — keyed to +// the run, but outside the vote-step family — labels its silent seats +// conversational-gate, matching where that panel is actually seated. +func TestSilentReapAckSeatIsAConversationalGate(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + id, err := db.CreateProposalIdempotent(conn, &model.Proposal{ + ProjectID: 1, Description: "accept the reap at seq 42?", + Criticality: model.CriticalityMedium, + Threshold: 0.5, RequiredVoters: 3, + Status: model.ProposalStatusOpen, CreatedBy: "conductor", + }, ReapAckProposalKey(run.ID, 42)) + testsupport.Must(t, err, "CreateProposalIdempotent: %v", err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: id, VoterName: "seat-a", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + }) + testsupport.Must(t, err, "CastVote: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if len(report.SilentVoteSeats) != 1 { + t.Fatalf("silent_vote_seats = %+v, want the one quiet reap-ack cast", + report.SilentVoteSeats) + } + if s := report.SilentVoteSeats[0]; s.Path != SeatPathConversationalGate || + s.Proposal != model.FormatProposalID(id) || s.Voter != "seat-a" { + t.Errorf("silent seat = %+v, want seat-a on %s via %q", + s, model.FormatProposalID(id), SeatPathConversationalGate) + } +} + +// TestSilentVoteSeatsAbsentWhenEverySeatReports: a run whose every seat +// reported carries no list at all — the coverage line already says N/N, and +// an empty list would restate it (`omitempty`). +func TestSilentVoteSeatsAbsentWhenEverySeatReports(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + id, err := db.CreateProposal(conn, &model.Proposal{ + ProjectID: 1, + Description: "activation panel for " + model.FormatRunID(run.ID), + Criticality: model.CriticalityMedium, + Threshold: 0.5, RequiredVoters: 3, + Status: model.ProposalStatusOpen, CreatedBy: "conductor", + }) + testsupport.Must(t, err, "CreateProposal: %v", err) + _, err = db.CastVote(conn, &model.Vote{ + ProposalID: id, VoterName: "seat-a", + Verdict: model.VerdictApprove, Confidence: 0.9, DomainRelevance: 0.8, + Usage: map[string]float64{"tokens": 42}, + }) + testsupport.Must(t, err, "CastVote: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + if c := report.VoteUsageCoverage; c.Casts != 1 || c.Reported != 1 { + t.Fatalf("coverage = %+v, want 1/1", c) + } + if report.SilentVoteSeats != nil { + t.Errorf("silent_vote_seats = %+v on a fully-reported run, want none", + report.SilentVoteSeats) + } +} diff --git a/internal/engine/dkt741_test.go b/internal/engine/dkt741_test.go new file mode 100644 index 00000000..30e11b70 --- /dev/null +++ b/internal/engine/dkt741_test.go @@ -0,0 +1,254 @@ +package engine + +import ( + "database/sql" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-741: `docket issue edit --scope` does not reach a step that already +// exists, and no verb refreshes it. Observed on RUN-52/VPL-434 (2026-08-24): +// the panel rejected the work 3/3 twice on the same out-of-scope migration +// blocker, the operator AUTHORIZED a widen, the conductor ran the edit — and +// `docket step render STEP-2459` still showed the old two-path scope, so the +// authorized widen was unexecutable and the run paid a +// `run abandon --issue` plus a full re-plan to get a fresh snapshot. +// +// The freeze is CORRECT and stays (§5.1.1, §6.6, §9 item 5): both the packet's +// `context.issue.scope` and the recorded `issue.diff` scope read the +// activation snapshot, and re-snapshotting for a live step would let two steps +// of one run render two different scopes and record diffs over two different +// path sets. What was missing is that nothing SAID so at the moment an +// operator spends a widen on a run that cannot receive it. +// +// So: the snapshot semantics are pinned here as behavior, and +// ScopeEditFrozenForActiveRuns is the disclosure. + +// scopedIssueInRun activates a one-issue run over the parks fixture with the +// given declared scope, and returns the run, the issue, and its `flaky@0` step. +func scopedIssueInRun(t *testing.T, conn *sql.DB, scope string) (int, int, int) { + t.Helper() + registerSource(t, conn, []byte(parkingWorkflow), "parks.toml") + issue := createIssue(t, conn, "widen me", "body", "task", nil) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, scope), "declaring scope") + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + var stepID int + err = conn.QueryRow( + `SELECT id FROM steps WHERE run_id = ? AND issue_id = ? AND instance = 'flaky@0'`, + run.ID, issue).Scan(&stepID) + testsupport.Must(t, err, "finding the step: %v", err) + return run.ID, issue, stepID +} + +// TestScopeEditDoesNotReachAnActivatedPacket is the semantics, asserted rather +// than assumed: the widen lands on `issues.scope_globs`, and the already- +// created step's RENDERED PACKET still carries the scope activation froze. +// +// This is the acceptance path DKT-741 offered first — "an engine verb makes a +// widened scope visible to an existing unclaimed step's packet" — recorded as +// a deliberate NON-goal. A test that pins the freeze is what stops a later +// reading of the issue from adding the refresh verb by accident. +func TestScopeEditDoesNotReachAnActivatedPacket(t *testing.T) { + conn := mustDB(t) + _, issue, stepID := scopedIssueInRun(t, conn, `["cli/src/command/start.rs"]`) + + before, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "rendering before the widen: %v", err) + if !strings.Contains(before.Packet, "cli/src/command/start.rs") { + t.Fatalf("premise: the packet does not carry the declared scope:\n%s", + before.Packet) + } + + // The authorized widen, exactly as the conductor ran it. + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, + `["cli/src/command/start.rs","script/install.sh","makefile"]`), + "widening scope") + + after, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "rendering after the widen: %v", err) + if after.Packet != before.Packet { + t.Errorf("the packet changed after a mid-run scope edit; §9 item 5's "+ + "edit immunity requires it not to:\nbefore:\n%s\nafter:\n%s", + before.Packet, after.Packet) + } + for _, added := range []string{"script/install.sh", "makefile"} { + if strings.Contains(after.Packet, added) { + t.Errorf("the packet carries %q, which was added after activation:\n%s", + added, after.Packet) + } + } + + // The diff scope reads the same frozen blob, so the two never disagree — + // a packet that said one thing while the recorded diff covered another + // would be worse than either answer alone. + frozen, err := snapshotScope(conn, runIDOfStep(t, conn, stepID), issue) + testsupport.Must(t, err, "reading the snapshot scope: %v", err) + if len(frozen) != 1 || frozen[0] != "cli/src/command/start.rs" { + t.Errorf("snapshotScope = %v after the widen, want the frozen "+ + "single-path scope", frozen) + } +} + +// TestScopeEditWarnsAndNamesTheAbandonPath is DKT-741's remedy: the operator +// who spends a widen on a live run is TOLD, in the same breath, that it did +// not reach the run and what would. +func TestScopeEditWarnsAndNamesTheAbandonPath(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["cli/src/command/start.rs"]`) + + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Fatalf("premise: warned before any edit: %v", got) + } + + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, + `["cli/src/command/start.rs","script/install.sh"]`), "widening scope") + + warnings := ScopeEditFrozenForActiveRuns(conn, issue) + if len(warnings) != 1 { + t.Fatalf("got %d warnings, want 1: %v", len(warnings), warnings) + } + w := warnings[0] + + // It must name the run, the issue, BOTH scopes, and the sanctioned verb — + // a warning that only says "this did not work" leaves the operator where + // RUN-52's conductor already was. + for _, want := range []string{ + model.FormatRunID(runID), + model.FormatID(issue), + "cli/src/command/start.rs", + "script/install.sh", + "run abandon " + model.FormatRunID(runID) + " --issue " + model.FormatID(issue), + "re-plan", + } { + if !strings.Contains(w, want) { + t.Errorf("the warning does not name %q:\n%s", want, w) + } + } +} + +// TestScopeEditWarningStaysSilentWhenThereIsNothingToDiscover pins the other +// half. An advisory that fires on edits it has nothing to say about is one an +// operator learns to ignore, and this one has to survive being read on the +// day it matters. +func TestScopeEditWarningStaysSilentWhenThereIsNothingToDiscover(t *testing.T) { + t.Run("a re-declaration of the same globs", func(t *testing.T) { + conn := mustDB(t) + _, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/a/**"]`), + "re-declaring") + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Errorf("warned on an edit that changed nothing: %v", got) + } + }) + + t.Run("a run that never activated", func(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(parkingWorkflow), "parks.toml") + issue := createIssue(t, conn, "not yet", "body", "task", nil) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/a/**"]`), + "declaring") + startRun(t, conn, issue) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/b/**"]`), + "widening") + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Errorf("warned on a planning run, which has frozen nothing yet "+ + "and will snapshot the new scope at activation: %v", got) + } + }) + + t.Run("a terminal run", func(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + mustExec(t, conn, `UPDATE runs SET status = ? WHERE id = ?`, + string(model.RunAbandoned), runID) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/b/**"]`), + "widening") + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Errorf("warned about an abandoned run: %v", got) + } + }) + + t.Run("a live run whose steps have all recorded", func(t *testing.T) { + conn := mustDB(t) + _, issue, stepID := scopedIssueInRun(t, conn, `["internal/a/**"]`) + mustExec(t, conn, `UPDATE steps SET status = ? WHERE id = ?`, + db.StepDone, stepID) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/b/**"]`), + "widening") + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Errorf("warned when no step will ever render again: %v", got) + } + }) + + t.Run("an issue in no run at all", func(t *testing.T) { + conn := mustDB(t) + issue := createIssue(t, conn, "loose", "body", "task", nil) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/a/**"]`), + "declaring") + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 0 { + t.Errorf("warned on an unbound issue: %v", got) + } + }) +} + +// TestScopeEditWarningFiresForAParkedStep is the case DKT-741 was actually +// filed from: the step is `waiting-human`, which is NOT terminal — an operator +// will resolve it and its packet will render again, from the stale snapshot. +// StepTerminal is the right membership here and StepOffScheduler is not. +func TestScopeEditWarningFiresForAParkedStep(t *testing.T) { + conn := mustDB(t) + _, issue, stepID := scopedIssueInRun(t, conn, `["internal/a/**"]`) + mustExec(t, conn, `UPDATE steps SET status = ? WHERE id = ?`, + db.StepWaitingHuman, stepID) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issue, `["internal/b/**"]`), + "widening") + + if got := ScopeEditFrozenForActiveRuns(conn, issue); len(got) != 1 { + t.Fatalf("got %d warnings for a parked step, want 1: %v", len(got), got) + } +} + +// TestTerminalStepStatusesMatchesStepTerminal keeps the SQL list and the +// predicate from drifting: a tenth status added to the machine must land in +// both or in neither. +func TestTerminalStepStatusesMatchesStepTerminal(t *testing.T) { + all := []string{ + db.StepPending, db.StepClaimed, db.StepRunning, db.StepGated, + db.StepDone, db.StepWaitingHuman, db.StepSkipped, db.StepSuperseded, + db.StepFailedRouted, + } + listed := make(map[string]bool, len(terminalStepStatuses)) + for _, s := range terminalStepStatuses { + listed[s] = true + } + for _, s := range all { + if listed[s] != db.StepTerminal(s) { + t.Errorf("%q: terminalStepStatuses says %v, db.StepTerminal says %v", + s, listed[s], db.StepTerminal(s)) + } + } + if len(terminalStepStatuses) != len(listed) { + t.Errorf("terminalStepStatuses has a duplicate: %v", terminalStepStatuses) + } +} + +func mustExec(t *testing.T, conn *sql.DB, query string, args ...any) { + t.Helper() + _, err := conn.Exec(query, args...) + testsupport.Must(t, err, "exec %q: %v", query, err) +} + +func runIDOfStep(t *testing.T, conn *sql.DB, stepID int) int { + t.Helper() + var runID int + err := conn.QueryRow(`SELECT run_id FROM steps WHERE id = ?`, stepID).Scan(&runID) + testsupport.Must(t, err, "reading the step's run: %v", err) + return runID +} diff --git a/internal/engine/dkt804_test.go b/internal/engine/dkt804_test.go new file mode 100644 index 00000000..180739a0 --- /dev/null +++ b/internal/engine/dkt804_test.go @@ -0,0 +1,95 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-804 — `step claim --render` must be atomic with respect to the packet +// render. On RUN-56 the claim committed FIRST and the render refused SECOND +// (an unpinned packet file), which stranded eight steps `claimed` with no +// token ever issued: every recovery verb requires the token nobody received, +// and the only exits were the 1800s TTL or an operator-gated `step reap`. +// +// The fixture reproduces RUN-56's exact shape: a step whose `packet` entry +// substitutes `{executor}`, claimed with a resolved hint whose contract the +// run never pinned. + +// TestClaimRenderRefusalLeavesTheStepClaimable is the regression: a +// `claim --render` whose packet render fails validation writes NOTHING — the +// step keeps its pre-claim status, holds no lease, spends no attempt — and +// the very next claim of the same step succeeds. +func TestClaimRenderRefusalLeavesTheStepClaimable(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/{executor}.md\"]\n") + // The DECLARED hint's contract exists and is pinned by activation + // (DKT-581's closure); no other contract is. + writeConfigFile(t, configDir, "contracts/w.md", "the declared contract\n") + + issue := createIssue(t, conn, "atomic claim", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := stepIDByInstance(t, conn, "implement@0") + + // The claim, rendered for a RESOLVED hint whose contract is not pinned — + // the label-resolved-executor path RUN-56 died on. + _, _, err = NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "rogue") + if err == nil { + t.Fatal("a claim whose packet references an unpinned file succeeded") + } + // The caller sees the SAME refusal it always did: VALIDATION_ERROR naming + // the exact unpinned path — executor error handling is unchanged. + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("code = %q, want %q", code, CodeValidation) + } + // DKT-818 rewrote this refusal's wording — it now names the pin set and + // the run rather than "not pinned by this run", and says which of the two + // unpinned causes applies (here: nothing wrote contracts/rogue.md). + if !strings.Contains(err.Error(), "is not in "+run.Ref()+"'s pin set") || + !strings.Contains(err.Error(), "contracts/rogue.md") { + t.Errorf("err = %q, want the unpinned-file refusal naming contracts/rogue.md", + err.Error()) + } + + // The refusal wrote NOTHING: pre-claim status, no lease, no attempt. + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepPending { + t.Errorf("status = %q after a refused claim, want %q — this is the "+ + "zombie claim: a lease held that no token can ever end", + step.Status, db.StepPending) + } + if step.Owner != "" || step.TokenHash != "" || step.ExpiresMS != 0 { + t.Errorf("a refused claim left a lease: owner=%q token_hash set=%v expires=%d", + step.Owner, step.TokenHash != "", step.ExpiresMS) + } + if step.Attempt != 0 { + t.Errorf("attempt = %d after a refused claim, want 0 — the refusal "+ + "consumed an attempt on its way out", step.Attempt) + } + + // And the step is STILL CLAIMABLE, immediately — no TTL wait, no reap. + // The same claim with the declared (pinned) hint succeeds and delivers + // both halves of the atomic response. + result, packet, err := NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "") + testsupport.Must(t, err, "re-claim after the refusal: %v", err) + if result.Token == "" { + t.Error("the successful claim returned no token") + } + if result.Attempt != 1 { + t.Errorf("attempt = %d on the first successful claim, want 1", result.Attempt) + } + if packet == nil || !strings.Contains(packet.Packet, "the declared contract") { + t.Errorf("the packet does not carry the pinned contract: %+v", packet) + } +} diff --git a/internal/engine/dkt805_test.go b/internal/engine/dkt805_test.go new file mode 100644 index 00000000..26d02c3a --- /dev/null +++ b/internal/engine/dkt805_test.go @@ -0,0 +1,212 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-805 — repin adopted drifted refs but could not ADD newly-referenced +// ones. On RUN-56 an operator-approved repin adopted contract bytes whose +// `packet_includes` reached two fragments the run never snapshotted; repin +// reported full success, and the next dispatch's every judge step refused at +// claim (VALIDATION_ERROR, "not pinned by this run … start a new run") — the +// disposition repin exists to avoid. These tests pin the closed gap: adopting +// bytes adopts their packet closure, in the same transaction, or refuses up +// front naming what cannot be pinned. + +// dkt805WorkflowSrc declares one step reading one contract — the pin set at +// activation is that contract plus policy.toml, and nothing else. +const dkt805WorkflowSrc = ` +[pipeline] +name = "dkt805-dev" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" +after = [] +packet = ["contracts/implement.md"] +` + +// dkt805ContractB1 is the corpus edit: the adopted contract bytes now declare +// a fragment the run never pinned. +const dkt805ContractB1 = "---\npacket_includes:\n - fragments/new.md\n---\n" + + "the implement contract, second edition\n" + +// TestRepinPinsTheAdoptedBytesNewlyRequiredRefs is the acceptance criterion +// verbatim: a run pinned against contract bytes B0 is repinned to B1, where B1 +// includes a fragment absent from the pin set. The repin must pin the fragment +// in the same transaction, record its own run-repinned event (null old_sha256, +// added: true), and leave the step claimable with the full packet. +func TestRepinPinsTheAdoptedBytesNewlyRequiredRefs(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt805-dev.toml", dkt805WorkflowSrc) + writeConfigFile(t, configDir, "contracts/implement.md", "the implement contract\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "closure subject", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + oldSHA := pinSHA(t, conn, run.ID, "contracts/implement.md") + if n := pinCount(t, conn, run.ID, "fragments/new.md"); n != 0 { + t.Fatalf("premise: fragments/new.md is already pinned (%d row(s)); B0 "+ + "declares no includes", n) + } + + // The corpus edit: B1 replaces the contract and creates the fragment it + // includes — RUN-56's dee7670, a file that did not exist at activation. + writeConfigFile(t, configDir, "contracts/implement.md", dkt805ContractB1) + fragmentBody := "the new fragment\n" + writeConfigFile(t, configDir, "fragments/new.md", fragmentBody) + + outcome, err := RepinRunWith(conn, run.ID, RepinOptions{ + Reason: "dee7670 introduced the fragment"}, nowMS) + testsupport.Must(t, err, "RepinRunWith: %v", err) + + if len(outcome.Repinned) != 1 || outcome.Repinned[0].Ref != "contracts/implement.md" { + t.Fatalf("repinned %+v, want exactly contracts/implement.md", outcome.Repinned) + } + if len(outcome.Added) != 1 { + t.Fatalf("added %d pin(s), want 1: %+v", len(outcome.Added), outcome.Added) + } + wantSHA := workflow.SHA256([]byte(fragmentBody)) + a := outcome.Added[0] + if a.Ref != "fragments/new.md" || a.Kind != db.PinKindFile || + a.OldSHA256 != "" || a.NewSHA256 != wantSHA || !a.Added { + t.Errorf("addition = %+v, want fragments/new.md (nothing) -> %s, marked added", + a, wantSHA) + } + + // The pin row exists, at the fragment's current disk hash, and the run is + // SOUND — repin never reports success on a run verify-pins still fails. + if got := pinSHA(t, conn, run.ID, "fragments/new.md"); got != wantSHA { + t.Errorf("pinned fragments/new.md at %s, want %s", got, wantSHA) + } + after, err := VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if !after.Sound() { + t.Errorf("the run is unsound after the repin: %s", PinReportReason(after)) + } + + // AC-3: the addition carries its own event — the sha, the reason, a null + // old_sha256, `added: true`, and what requires the ref. + var added map[string]any + for _, ev := range repinEvents(t, conn, run.ID) { + if ev["added"] == true { + added = ev + } + } + if added == nil { + t.Fatal("no run-repinned event carries added: true") + } + if added["ref"] != "fragments/new.md" || added["new_sha256"] != wantSHA || + added["old_sha256"] != nil || + added["reason"] != "dee7670 introduced the fragment" { + t.Errorf("added event = %v, want fragments/new.md at %s with a null "+ + "old_sha256 and the operator's reason", added, wantSHA) + } + required, _ := added["required_by"].([]any) + if len(required) != 1 || required[0] != "implement@0" { + t.Errorf("required_by = %v, want [implement@0]", added["required_by"]) + } + + // AC-5's second half: the step is CLAIMABLE, and the rendered packet + // carries both the adopted contract and the fragment it introduced — + // the exact render RUN-56's dispatch died on. + stepID := stepIDByInstance(t, conn, "implement@0") + result, packet, err := NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "") + testsupport.Must(t, err, "claim after the repin: %v", err) + if result.Token == "" { + t.Error("the claim returned no token") + } + if packet == nil || + !strings.Contains(packet.Packet, "the implement contract, second edition") || + !strings.Contains(packet.Packet, "the new fragment") { + t.Errorf("the packet does not carry the adopted contract and its fragment: %+v", + packet) + } + + // The event trail still reconstructs the drifted ref's move: the ordinary + // repin event carries old -> new for the contract. + if oldSHA == outcome.Repinned[0].NewSHA256 { + t.Errorf("premise: B1 hashed identically to B0 (%s)", oldSHA) + } +} + +// TestRepinRefusesWhenANewlyRequiredRefHasNoBytes is the up-front refusal: the +// adopted bytes require a fragment that resolves NOWHERE, so there is nothing +// to pin and proceeding would trade the CONFLICT for a render-time +// VALIDATION_ERROR. The refusal writes nothing, names the ref and its readers +// and the dispositions, and restoring the file is a working recovery. +func TestRepinRefusesWhenANewlyRequiredRefHasNoBytes(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt805-dev.toml", dkt805WorkflowSrc) + contractB0 := "the implement contract\n" + writeConfigFile(t, configDir, "contracts/implement.md", contractB0) + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "closure subject", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + oldSHA := pinSHA(t, conn, run.ID, "contracts/implement.md") + + // The corpus edit, MINUS the fragment: B1 includes a file nobody wrote. + writeConfigFile(t, configDir, "contracts/implement.md", dkt805ContractB1) + + _, err = RepinRunWith(conn, run.ID, RepinOptions{Reason: "adopt dee7670"}, nowMS) + if err == nil { + t.Fatal("a repin whose adopted bytes require an unresolvable ref succeeded") + } + if code, _ := CodeOf(err); code != CodeNotFound { + t.Errorf("code = %q, want %q", code, CodeNotFound) + } + // AC-4: the refusal names the ref, what reads it, and the dispositions. + for _, want := range []string{ + "fragments/new.md", "implement@0", "restore the file", "abandon the run", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to name %q", err.Error(), want) + } + } + + // The refusal wrote NOTHING: the contract pin still records B0, no pin row + // names the fragment, and no run-repinned event exists. + if got := pinSHA(t, conn, run.ID, "contracts/implement.md"); got != oldSHA { + t.Errorf("the refused repin moved the contract pin: %s, want %s", got, oldSHA) + } + if n := pinCount(t, conn, run.ID, "fragments/new.md"); n != 0 { + t.Errorf("the refused repin left %d pin row(s) for the fragment", n) + } + if evs := repinEvents(t, conn, run.ID); len(evs) != 0 { + t.Errorf("the refused repin recorded %d event(s): %v", len(evs), evs) + } + + // AC-5's second half, under the refusal outcome: the named disposition + // works. Restoring B0 puts the run back exactly where it was, and the + // step claims under the original agreement. + writeConfigFile(t, configDir, "contracts/implement.md", contractB0) + stepID := stepIDByInstance(t, conn, "implement@0") + result, packet, err := NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "") + testsupport.Must(t, err, "claim after restoring B0: %v", err) + if result.Token == "" { + t.Error("the claim returned no token") + } + if packet == nil || !strings.Contains(packet.Packet, "the implement contract") { + t.Errorf("the packet does not carry the restored contract: %+v", packet) + } +} diff --git a/internal/engine/dkt818_test.go b/internal/engine/dkt818_test.go new file mode 100644 index 00000000..d9a89fd6 --- /dev/null +++ b/internal/engine/dkt818_test.go @@ -0,0 +1,110 @@ +package engine + +import ( + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-818 — the "not pinned" refusal named a remedy that was already +// satisfied. On RUN-59 a repin adopted contract bytes reaching two fragments +// the run had never snapshotted, and every judge claim then died with "add it +// under an instance-config root and start a new run" — while BOTH fragments sat +// under `~/.docket/config/fragments/`. The conductor went looking for a missing +// file, found it present, and had to re-derive the real cause: the pin set, not +// the filesystem. These tests hold the refusal to a TRUE sentence about which of +// the two unpinned causes it actually has. +// +// The fixture is DKT-804's shape, which is the one way a live run still reaches +// an unpinned packet file at render time: a `{executor}` entry resolved to a +// hint whose contract activation never pinned. Whether that contract exists on +// disk is the whole variable under test. + +// TestUnpinnedButPresentPacketFileNamesThePinSetNotTheFilesystem is RUN-59's +// half: the file IS on disk, so the old remedy was a dead end. The refusal must +// name the run's pin set, say the file is present and where, and offer the +// remedy that actually applies. +func TestUnpinnedButPresentPacketFileNamesThePinSetNotTheFilesystem(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/{executor}.md\"]\n") + // The DECLARED hint's contract is what activation pins (DKT-581's + // closure). The resolved hint's contract is written too — present under an + // instance-config root, and still outside the run's frozen pin set. + writeConfigFile(t, configDir, "contracts/w.md", "the declared contract\n") + writeConfigFile(t, configDir, "contracts/rogue.md", "the resolved contract\n") + + issue := createIssue(t, conn, "present but unpinned", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := stepIDByInstance(t, conn, "implement@0") + _, _, err = NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "rogue") + if err == nil { + t.Fatal("a claim whose packet references an unpinned file succeeded") + } + // The DISPOSITION is unchanged — an unpinned entry has no snapshot, and + // only the sentence explaining it moves. + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("code = %q, want %q", code, CodeValidation) + } + + full := filepath.Join(configDir, "contracts/rogue.md") + for _, want := range []string{ + "contracts/rogue.md", + "is not in " + run.Ref() + "'s pin set", + "froze at activation", + full, + "start a new run to pin it", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to say %q", err.Error(), want) + } + } + // The RUN-59 defect itself: the refusal must not send anyone to the + // filesystem for a file that is already sitting there. + if strings.Contains(err.Error(), "add it under an instance-config root") { + t.Errorf("err = %q still names the already-satisfied remedy", err.Error()) + } +} + +// TestUnpinnedAndAbsentPacketFileStillSaysToAddIt is the other half: nothing +// wrote the file, so writing it IS the remedy and the message must keep saying +// so. Splitting the sentence must not cost the honest case its answer. +func TestUnpinnedAndAbsentPacketFileStillSaysToAddIt(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/{executor}.md\"]\n") + writeConfigFile(t, configDir, "contracts/w.md", "the declared contract\n") + + issue := createIssue(t, conn, "absent and unpinned", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := stepIDByInstance(t, conn, "implement@0") + _, _, err = NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "rogue") + if err == nil { + t.Fatal("a claim whose packet references an absent file succeeded") + } + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("code = %q, want %q", code, CodeValidation) + } + for _, want := range []string{ + "contracts/rogue.md", + "is not in " + run.Ref() + "'s pin set", + "resolves under no instance-config root", + "add it under one and start a new run to pin it", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to say %q", err.Error(), want) + } + } +} diff --git a/internal/engine/dkt820_test.go b/internal/engine/dkt820_test.go new file mode 100644 index 00000000..a411086c --- /dev/null +++ b/internal/engine/dkt820_test.go @@ -0,0 +1,90 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-820 — the RETRY half of DKT-804's zombie claim. On RUN-59 four judge +// claims died on an unpinned packet file, and STEP-2679's executor saw the +// damage from the other side: its retry was refused CONFLICT "step review@2#0 +// is not ready to claim: the step is not pending" — blocked by its OWN first +// attempt's lease, a lease no token was ever issued for. +// +// DKT-804's fix moved the render's validation ahead of the lease, so the first +// refusal writes nothing. This test holds the CONSEQUENCE that RUN-59's +// executors actually needed: the retry must hear the SAME diagnosis as the +// first attempt — the packet file the run never pinned — and never a conflict +// about a lease that the refusal itself left behind. + +// TestRepeatedRenderRefusalKeepsRefusingForTheSameReason claims the same step +// twice against an UNCHANGED broken config: both refusals are the identical +// VALIDATION_ERROR, and the step is pending, lease-free, and attempt-free after +// each. A CONFLICT on the second claim is the regression. +func TestRepeatedRenderRefusalKeepsRefusingForTheSameReason(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/{executor}.md\"]\n") + // Only the DECLARED hint's contract is pinned by activation (DKT-581's + // closure); the resolved hint's is not, and nothing fixes that between the + // two claims below. + writeConfigFile(t, configDir, "contracts/w.md", "the declared contract\n") + + issue := createIssue(t, conn, "the retry sees the same wall", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := stepIDByInstance(t, conn, "implement@0") + + claim := func() error { + _, _, err := NewEngine().ClaimStepRendered(conn, stepID, ClaimOptions{ + Owner: "wave:STEP-1", NowMS: nowMS, + }, "", "rogue") + return err + } + + // Attempt one and attempt two, with NOTHING changed in between — RUN-59's + // STEP-2679, whose executor retried a claim it had never been handed a + // token for. + for _, attempt := range []string{"the first claim", "the retry"} { + err := claim() + if err == nil { + t.Fatalf("%s: a claim whose packet references an unpinned file succeeded", attempt) + } + // The retry must diagnose the PACKET, not the lease. CONFLICT "the + // step is not pending" here is the zombie claim reported from the + // claimant's side. + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("%s: code = %q, want %q — err = %q", attempt, code, CodeValidation, err.Error()) + } + if strings.Contains(err.Error(), string(CondStatus)) { + t.Errorf("%s: err = %q — the claim was refused by the lease its own "+ + "predecessor left behind, not by the packet", attempt, err.Error()) + } + if !strings.Contains(err.Error(), "is not in "+run.Ref()+"'s pin set") || + !strings.Contains(err.Error(), "contracts/rogue.md") { + t.Errorf("%s: err = %q, want the unpinned-file refusal naming contracts/rogue.md", + attempt, err.Error()) + } + + // And the step is exactly as it was before the claim: no lease for a + // token nobody holds, no attempt burned on a claim that delivered + // nothing. + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepPending { + t.Errorf("%s: status = %q, want %q", attempt, step.Status, db.StepPending) + } + if step.Owner != "" || step.TokenHash != "" || step.ExpiresMS != 0 { + t.Errorf("%s: a refused claim left a lease: owner=%q token_hash set=%v expires=%d", + attempt, step.Owner, step.TokenHash != "", step.ExpiresMS) + } + if step.Attempt != 0 { + t.Errorf("%s: attempt = %d, want 0", attempt, step.Attempt) + } + } +} diff --git a/internal/engine/dkt821_test.go b/internal/engine/dkt821_test.go new file mode 100644 index 00000000..de6ffd3b --- /dev/null +++ b/internal/engine/dkt821_test.go @@ -0,0 +1,293 @@ +package engine + +import ( + "database/sql" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-821 — `verify-pins` reported a structurally wedged run as healthy. +// +// On RUN-59, minutes after all four review@2 claims died on `packet file +// "fragments/laziness-ladder.md" is not pinned by this run` (VALIDATION_ERROR), +// `docket run verify-pins RUN-59 --json` answered exit 0 with all 30 pins +// `"status":"ok"`. Every pinned ref DID match its bytes — and those very bytes +// referenced two fragments the run never pinned, which is what made every +// remaining judge step unclaimable. A conductor used the verb as a pre-dispatch +// health check and launched four executors into claims that could not succeed. +// +// The per-pin check cannot see this: it asks "do these bytes still match", and +// they do. The missing question is what the bytes REFERENCE. These tests hold +// the verb to both halves. + +// dkt821WorkflowSrc declares one step reading one contract — the pin set at +// activation is that contract, whatever its `packet_includes` reach, and +// policy.toml. +const dkt821WorkflowSrc = ` +[pipeline] +name = "dkt821-dev" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "judge" +executor = "judge" +emits = "change-summary" +after = [] +packet = ["contracts/judge.md"] +` + +// dkt821ContractB1 is RUN-59's corpus edit: the contract's second edition +// reaches a fragment the run's pin set knows nothing about. +const dkt821ContractB1 = "---\npacket_includes:\n - fragments/laziness-ladder.md\n---\n" + + "the judge contract, second edition\n" + +// movePinTo is the ONE statement a pre-DKT-805 repin ran: the pin row moves to +// the adopted bytes, and nothing walks what those bytes now reference. +// +// It is how RUN-59 reached the state it is in, and how every run repinned +// before dbd3e7b still sits in the store. A repin TODAY closes the hole in the +// same transaction (dkt805_test.go), so driving one here would build the +// opposite of the fixture — see the companion test below, which drives exactly +// that and asserts the closed result. +func movePinTo(t *testing.T, conn *sql.DB, runID int, ref, sha string) { + t.Helper() + res, err := conn.Exec( + `UPDATE pins SET sha256 = ? WHERE run_id = ? AND ref = ?`, sha, runID, ref) + testsupport.Must(t, err, "moving pin %s: %v", ref, err) + n, err := res.RowsAffected() + testsupport.Must(t, err, "RowsAffected: %v", err) + if n != 1 { + t.Fatalf("moving pin %s touched %d row(s), want 1", ref, n) + } +} + +// writtenAt is writeConfigFile returning the path the config ROOTS resolve to. +// On darwin `t.TempDir()` hands back a /tmp path whose real location is +// /private/tmp, and the roots are evaluated — an assertion on the unresolved +// spelling would fail for a reason that has nothing to do with pins. +func writtenAt(t *testing.T, configDir, rel, body string) string { + t.Helper() + path := writeConfigFile(t, configDir, rel, body) + real, err := filepath.EvalSymlinks(path) + testsupport.Must(t, err, "resolving %s: %v", path, err) + return real +} + +// TestVerifyPinsReportsAReferenceThePinSetDoesNotHold is AC1 and AC2 — RUN-59's +// shape, and the verdict the verb owed it. Every pin matches disk; the pinned +// contract's own bytes reach a fragment the run never pinned; the verb must +// refuse and name BOTH files. +func TestVerifyPinsReportsAReferenceThePinSetDoesNotHold(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt821-dev.toml", dkt821WorkflowSrc) + writeConfigFile(t, configDir, "contracts/judge.md", "the judge contract\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "closure blindness", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // The corpus commit, then the repin that adopted it as repin behaved on the + // day RUN-59 wedged: the contract's pin moves to B1's bytes, the fragment + // B1 now reaches is pinned by nothing. + writeConfigFile(t, configDir, "contracts/judge.md", dkt821ContractB1) + fragment := writtenAt(t, configDir, "fragments/laziness-ladder.md", + "the laziness ladder\n") + movePinTo(t, conn, run.ID, "contracts/judge.md", + workflow.SHA256([]byte(dkt821ContractB1))) + if n := pinCount(t, conn, run.ID, "fragments/laziness-ladder.md"); n != 0 { + t.Fatalf("premise: the fragment is already pinned (%d row(s))", n) + } + + report, err := VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + + // THE BLINDNESS ITSELF: the per-pin half is spotless. That is exactly what + // exit 0 was reporting, and why the closure half had to be a separate + // question rather than a stricter reading of the same one. + if report.Changed != 0 || report.Missing != 0 { + t.Fatalf("changed = %d, missing = %d, want 0 and 0 — the fixture is "+ + "RUN-59's, where every pinned ref matches its bytes", + report.Changed, report.Missing) + } + if report.Sound() { + t.Fatal("the report reads sound on a run whose pinned contract " + + "references a fragment the run does not pin — every step reading " + + "that contract is unclaimable") + } + if report.Unpinned != 1 || len(report.References) != 1 { + t.Fatalf("unpinned = %d, references = %+v, want exactly one", + report.Unpinned, report.References) + } + + got := report.References[0] + if got.Status != PinUnpinnedReference { + t.Errorf("status = %q, want %q", got.Status, PinUnpinnedReference) + } + if got.Ref != "fragments/laziness-ladder.md" { + t.Errorf("ref = %q, want fragments/laziness-ladder.md", got.Ref) + } + // AC1 asks for BOTH names: the ref, and the contract that references it. + if len(got.IncludedBy) != 1 || got.IncludedBy[0] != "contracts/judge.md" { + t.Errorf("included_by = %v, want [contracts/judge.md] — an operator "+ + "needs the file that wrote the reference, not just the ref", got.IncludedBy) + } + // And the claims that will die, which is what a conductor is deciding about. + if len(got.RequiredBy) != 1 || got.RequiredBy[0] != "judge@0" { + t.Errorf("required_by = %v, want [judge@0]", got.RequiredBy) + } + // Present-but-unpinned is RUN-59's case, and DKT-818's distinction: the + // remedy is not on the filesystem, so the report must say the file is there. + if got.Path != fragment { + t.Errorf("path = %q, want %q", got.Path, fragment) + } + + reason := PinReportReason(report) + for _, want := range []string{ + "contracts/judge.md", "fragments/laziness-ladder.md", + run.Ref() + " does not pin", fragment, + } { + if !strings.Contains(reason, want) { + t.Errorf("reason = %q, want it to say %q", reason, want) + } + } +} + +// TestVerifyPinsReportsAnUnpinnedReferenceWithNoBytesOnDisk is the other half of +// DKT-818's distinction, and the path that still opens this hole with NO repin +// in the story: activation pins the closure it can SEE, and a `packet_includes` +// naming a file no config root holds yet is pinned by nothing. The run +// activates clean, and the reference is unpinned from the first second — before +// and after the file lands. +func TestVerifyPinsReportsAnUnpinnedReferenceWithNoBytesOnDisk(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt821-dev.toml", dkt821WorkflowSrc) + writeConfigFile(t, configDir, "contracts/judge.md", dkt821ContractB1) + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "the fragment arrives later", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + if n := pinCount(t, conn, run.ID, "fragments/laziness-ladder.md"); n != 0 { + t.Fatalf("premise: the fragment is pinned (%d row(s)) though no root "+ + "held it at activation", n) + } + + report, err := VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if report.Sound() || report.Unpinned != 1 { + t.Fatalf("sound = %v, unpinned = %d, want an unsound report naming the "+ + "one unpinned reference", report.Sound(), report.Unpinned) + } + if got := report.References[0].Path; got != "" { + t.Errorf("path = %q, want empty — no instance-config root holds the "+ + "file, and the report must not imply one does", got) + } + if reason := PinReportReason(report); !strings.Contains( + reason, "no instance-config root holds it") { + t.Errorf("reason = %q, want it to say the file is absent, not merely "+ + "unpinned", reason) + } + + // The file lands — the pin set is still frozen without it, so the verdict + // stands and only its remedy changes. + fragment := writtenAt(t, configDir, "fragments/laziness-ladder.md", + "the laziness ladder\n") + report, err = VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if report.Sound() { + t.Fatal("writing the file made the report sound; the pin set is what " + + "froze without it, and a file on disk is not a pin") + } + if got := report.References[0].Path; got != fragment { + t.Errorf("path = %q, want %q", got, fragment) + } +} + +// TestARepinTodayLeavesNoUnpinnedReference is the companion the fixture above +// needs: driving a repin on RUN-59's corpus edit no longer produces the wedge, +// because DKT-805 taught repin to pin the closure of the bytes it adopts. The +// two verbs share one walk (unpinnedClosureRefs), and this is the assertion +// that they agree — the disagreement is what DKT-821 was. +func TestARepinTodayLeavesNoUnpinnedReference(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt821-dev.toml", dkt821WorkflowSrc) + writeConfigFile(t, configDir, "contracts/judge.md", "the judge contract\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "repin closes it", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + writeConfigFile(t, configDir, "contracts/judge.md", dkt821ContractB1) + writeConfigFile(t, configDir, "fragments/laziness-ladder.md", "the ladder\n") + + _, err = RepinRunWith(conn, run.ID, RepinOptions{ + Reason: "adopting the second edition"}, nowMS) + testsupport.Must(t, err, "RepinRunWith: %v", err) + + report, err := VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if !report.Sound() { + t.Fatalf("verify-pins refuses a run repin just closed: %s", + PinReportReason(report)) + } + if report.Unpinned != 0 || len(report.References) != 0 { + t.Errorf("references = %+v, want none — repin pinned what the adopted "+ + "bytes require, so there is no hole left to name", report.References) + } +} + +// TestVerifyPinsStaysCleanOnAClosedPinSet is AC3, and it is not vacuous: the +// contract reaches a fragment which reaches a second fragment, and activation +// pinned the whole chain. A closure check that reported those as unpinned would +// fail every healthy run in the fleet. +func TestVerifyPinsStaysCleanOnAClosedPinSet(t *testing.T) { + conn, configDir := configRepo(t) + writeConfigFile(t, configDir, "workflows/dkt821-dev.toml", dkt821WorkflowSrc) + writeConfigFile(t, configDir, "contracts/judge.md", + "---\npacket_includes:\n - fragments/ladder.md\n---\nthe judge contract\n") + writeConfigFile(t, configDir, "fragments/ladder.md", + "---\npacket_includes:\n - fragments/boundaries.md\n---\nthe ladder\n") + writeConfigFile(t, configDir, "fragments/boundaries.md", "the boundaries\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "a closed pin set", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // The premise: activation pinned the transitive chain, which is the thing + // the closure check is judged against. + for _, ref := range []string{ + "contracts/judge.md", "fragments/ladder.md", "fragments/boundaries.md", + } { + if n := pinCount(t, conn, run.ID, ref); n != 1 { + t.Fatalf("premise: %s has %d pin(s), want 1", ref, n) + } + } + + report, err := VerifyPins(conn, run.ID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if !report.Sound() { + t.Fatalf("a fully closed run reads unsound: %s", PinReportReason(report)) + } + if report.Unpinned != 0 || len(report.References) != 0 { + t.Errorf("references = %+v, want none", report.References) + } + // Empty, never nil: `--json` emits an array on a clean run so a consumer + // parses one shape either way. + if report.References == nil { + t.Error("references is nil; the wire shape must be an empty array") + } +} diff --git a/internal/engine/dkt861_test.go b/internal/engine/dkt861_test.go new file mode 100644 index 00000000..b664a29f --- /dev/null +++ b/internal/engine/dkt861_test.go @@ -0,0 +1,161 @@ +package engine + +import ( + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-861 — `step resolve --as override-pass` silently dropped interposed gate +// steps, warning only after the mutation. Observed on RUN-61: the operator +// chose override-pass on verify@2 BECAUSE the offered option promised the run +// would proceed to its verify-tribunal gate; the generic pass skipped the +// tribunal instead, the run rolled to `done`, and the panel the operator +// explicitly bought never ruled. The DKT-470 warning named exactly this, but +// only beside a resolution already committed. +// +// The remedy under test: resolveStep REFUSES override-pass on a step whose +// threshold interposes other step(s) unless the operator passes +// --drop-interposed (ResolveStepDropInterposed), and the refusal carries the +// DKT-470 sentence — the consequence is presented BEFORE anything mutates. A +// step with no interposed targets resolves exactly as it always has, no flag +// required. The fixture is dkt470_test.go's interposeOverridePassSrc: a +// verify step whose threshold interposes a tribunal vote. + +// TestOverridePassRefusedWithoutDropInterposed is acceptance criterion (i): +// with interposed dependents and no acknowledgment, the resolution is refused +// with the warning text, and NOTHING mutates — the step stays parked, the +// interposed vote stays pending, no routing is recorded. +func TestOverridePassRefusedWithoutDropInterposed(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(interposeOverridePassSrc), "interpose-op.toml") + issue := createIssue(t, conn, "op", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := parkVerifyWaitingHuman(t, conn, e) + + err = e.ResolveStep(conn, stepID, ResolveOverridePass, "accepted", nowMS) + if err == nil { + t.Fatal("override-pass with interposed dependents and no " + + "--drop-interposed was accepted") + } + // The refusal IS the warning: the pre-mutation message carries the same + // sentence the post-hoc warning used, plus the acknowledgment it asks for. + for _, want := range []string{ + "tribunal", "verify@0", "will NOT be routed", "--drop-interposed", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("refusal = %q, want it to contain %q", err, want) + } + } + + // Nothing mutated: the refusal precedes the transaction. + if got := stepStatus(t, conn, "verify@0"); got != db.StepWaitingHuman { + t.Errorf("verify@0 = %q after the refusal, want it still %q", + got, db.StepWaitingHuman) + } + if got := stepStatus(t, conn, "tribunal@0"); got != db.StepPending { + t.Errorf("tribunal@0 = %q after the refusal, want it still %q", + got, db.StepPending) + } + // The park's own routing record is untouched — no `pass` was written. + if got := stepRouting(t, conn, "verify@0"); got != string(db.StepWaitingHuman) { + t.Errorf("verify@0 routing = %q after the refusal, want the park's "+ + "own %q record unchanged", got, db.StepWaitingHuman) + } +} + +// TestOverridePassBatchRefusedWithoutDropInterposed: --batch rides the same +// verb, so the acknowledgment gate covers it too — a standing authorization is +// exactly the ruling that must not slip past the interposed-step consequence. +func TestOverridePassBatchRefusedWithoutDropInterposed(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(interposeOverridePassSrc), "interpose-op.toml") + issue := createIssue(t, conn, "op", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := parkVerifyWaitingHuman(t, conn, e) + + err = e.ResolveStepBatch(conn, stepID, ResolveOverridePass, "accepted", nowMS) + if err == nil { + t.Fatal("batch override-pass with interposed dependents and no " + + "--drop-interposed was accepted") + } + if !strings.Contains(err.Error(), "--drop-interposed") { + t.Errorf("refusal = %q, want it to name --drop-interposed", err) + } + if got := stepStatus(t, conn, "tribunal@0"); got != db.StepPending { + t.Errorf("tribunal@0 = %q after the refusal, want it still %q", + got, db.StepPending) + } +} + +// Acceptance criterion (ii) — the acknowledged resolution proceeds and the +// interposed step still ends up skipped exactly as today — is pinned by +// dkt470_test.go's TestOverridePassStillBypassesTheThreshold, which now +// resolves through ResolveStepDropInterposed. The CLI half (the warning +// printed BEFORE the mutation on the acknowledged path) is +// internal/cli/step_resolve_interposed_test.go. + +// TestOverridePassWithoutInterposedNeedsNoFlag is acceptance criterion (iii), +// the regression guard: a step whose threshold interposes nothing resolves +// under the plain, unacknowledged call exactly as before. +func TestOverridePassWithoutInterposedNeedsNoFlag(t *testing.T) { + conn := mustDB(t) + e := testEngine() + // The default corpus fixture: verify's threshold routes only to the + // reserved fix-loop / waiting-human vocabulary — no step-name targets. + activatedRun(t, conn) + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", "report", unverifiablePayload) + if got := stepStatus(t, conn, "verify@0"); got != db.StepWaitingHuman { + t.Fatalf("premise: verify@0 = %q, want %q", got, db.StepWaitingHuman) + } + + stepID := stepIDByInstance(t, conn, "verify@0") + testsupport.Must(t, + e.ResolveStep(conn, stepID, ResolveOverridePass, "accepted", nowMS), + "override-pass on a step with no interposed targets: %v", nil) + if got := stepStatus(t, conn, "verify@0"); got != db.StepDone { + t.Errorf("verify@0 = %q after the override-pass, want %q", got, db.StepDone) + } + if got := stepRouting(t, conn, "verify@0"); got != RoutingPass { + t.Errorf("verify@0 routing = %q, want %q", got, RoutingPass) + } +} + +// TestDropInterposedRequiresOverridePass mirrors the --batch flag-combo guard: +// the acknowledgment waives a refusal only override-pass can trigger, so on +// any other resolution it is refused rather than silently accepted. +func TestDropInterposedRequiresOverridePass(t *testing.T) { + conn := mustDB(t) + e := testEngine() + registerSource(t, conn, []byte(interposeOverridePassSrc), "interpose-op.toml") + issue := createIssue(t, conn, "op", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + stepID := parkVerifyWaitingHuman(t, conn, e) + + err = e.ResolveStepDropInterposed(conn, stepID, ResolveSkip, "nope", false, nowMS) + if err == nil { + t.Fatal("--drop-interposed with --as skip was accepted") + } + if !strings.Contains(err.Error(), ResolveOverridePass) { + t.Errorf("refusal = %q, want it to name %s", err, ResolveOverridePass) + } + if got := stepStatus(t, conn, "verify@0"); got != db.StepWaitingHuman { + t.Errorf("verify@0 = %q after the refusal, want it still %q", + got, db.StepWaitingHuman) + } +} diff --git a/internal/engine/dkt867_test.go b/internal/engine/dkt867_test.go new file mode 100644 index 00000000..ac8a5629 --- /dev/null +++ b/internal/engine/dkt867_test.go @@ -0,0 +1,249 @@ +package engine + +import ( + "math" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-867 — `expected_cost` was variant-blind: the definition declares one +// number per step template, so a claim the dispatcher escalated to a far +// pricier executor variant accrued exactly what the cheap variant did +// (RUN-60 STEP-2752 vs STEP-2769: a 22.51M-token escalated fix attempt and a +// 1.83M ordinary one priced identically at the cap), and every measured run +// showed budget.spend == budget.floor exactly. +// +// The remedy under test: the DISPATCHER — the party that resolves a step to a +// variant, at claim time — declares the scaling (`step claim +// --cost-multiplier`), and the claim checks the SCALED cost at the cap and +// records it in its own `step-claimed` event, which RunFloorTx sums in place +// of the step row's declaration. Core learns no variant vocabulary +// (genericity.md): it learns a number, from the one party positioned to know +// it. The ledger-reading remedy DKT-867 also named is NOT taken here — its +// safe form already exists as DKT-238's measured usage cap, and folding token +// counts into the declared cap's `max()` is the incomparability trap +// budgetSnapshot.usageCap documents. +// +// The fixture's costs are read, never restated: `implement` is 1.50 in the +// activatedRun workflow, and every expectation below is derived from +// expectedCostOf so a fixture change cannot silently hollow these tests. + +// TestEscalatedClaimAccruesScaledCost is the core defect, fixed: the SAME step +// kind claimed as an escalated variant contributes materially more to the cap +// sum than a claim that stayed on the cheap variant. +func TestEscalatedClaimAccruesScaledCost(t *testing.T) { + // The cheap run: no multiplier, the declared cost verbatim. + connCheap := mustDB(t) + runCheap, _ := budgetRun(t, connCheap, 0) + cost := expectedCostOf(t, connCheap, "implement@0") + claimInstance(t, connCheap, "implement@0", nowMS) + cheap := runFloor(t, connCheap, runCheap) + + // The escalated run: the dispatcher resolved the same step kind to a + // variant it prices at 4x. + connEsc := mustDB(t) + runEsc, _ := budgetRun(t, connEsc, 0) + _, err := ClaimStep(connEsc, stepIDByInstance(t, connEsc, "implement@0"), + ClaimOptions{Owner: "worker", CostMultiplier: 4, NowMS: nowMS}) + testsupport.Must(t, err, "escalated claim: %v", err) + escalated := runFloor(t, connEsc, runEsc) + + if cheap != cost { + t.Fatalf("cheap-variant floor = %g, want the declared %g", cheap, cost) + } + if escalated != 4*cost { + t.Errorf("escalated floor = %g, want %g — the scaled cost, not the "+ + "declared one", escalated, 4*cost) + } + if escalated <= cheap { + t.Errorf("escalated floor %g is not above the cheap floor %g — the "+ + "DKT-867 defect (variants pricing identically) is not fixed", + escalated, cheap) + } +} + +// TestUnescalatedClaimIsByteForByteUnchanged is the regression half: a run +// that never escalates — no `--cost-multiplier` anywhere — accrues exactly the +// declared costs, writes the claim event with an EMPTY data payload (no new +// key for consumers of the feed to trip on), and hits the cap at exactly the +// boundary it always did. +func TestUnescalatedClaimIsByteForByteUnchanged(t *testing.T) { + conn := mustDB(t) + runID, _ := budgetRun(t, conn, 0) + cost := expectedCostOf(t, conn, "implement@0") + execSQL(t, conn, `UPDATE runs SET budget = ? WHERE id = ?`, cost, runID) + + // Exactly at the cap: admitted, and the floor is the declared cost. + claimInstance(t, conn, "implement@0", nowMS) + if got := runFloor(t, conn, runID); got != cost { + t.Fatalf("floor = %g after an unescalated claim, want the declared %g", + got, cost) + } + + // The claim event carries no cost override — only the instance key every + // step-claimed event has always carried — so RunFloorTx's fallback to the + // step row is what priced it, and a feed consumer sees the exact + // pre-DKT-867 bytes. + var data string + err := conn.QueryRow( + `SELECT data FROM events WHERE run_id = ? AND kind = ? ORDER BY seq DESC LIMIT 1`, + runID, EventStepClaimed).Scan(&data) + testsupport.Must(t, err, "reading the claim event: %v", err) + if strings.Contains(data, "expected_cost") { + t.Errorf("unescalated claim event data = %q, want no expected_cost "+ + "override — an unscaled claim must serialize exactly as before", data) + } + if data != `{"instance":"implement@0"}` { + t.Errorf("unescalated claim event data = %q, want the pre-DKT-867 "+ + `{"instance":"implement@0"}`, data) + } + + // The NEXT costed claim crosses and is refused — the pre-DKT-867 boundary, + // unchanged. + execSQL(t, conn, `UPDATE steps SET expires_ms = 1 WHERE instance = 'implement@0'`) + _, err = ClaimStep(conn, stepIDByInstance(t, conn, "implement@0"), + ClaimOptions{Owner: "w2", NowMS: nowMS + 1}) + if err == nil { + t.Fatal("the claim that crosses the cap was admitted") + } + if code, _ := CodeOf(err); code != CodeConflict { + t.Errorf("the refusal is %v, want %v", code, CodeConflict) + } + if !strings.Contains(err.Error(), "budget") { + t.Errorf("the refusal does not name the budget: %v", err) + } +} + +// TestEscalatedClaimIsRefusedAtTheCap is consequence (1) of DKT-867, closed: a +// cap with headroom for the DECLARED cost refuses a claim whose dispatcher +// declared an escalation the cap cannot absorb — the hop is visible at the +// cap, before the accrual commits, and the run pauses exactly as any other +// breach does. +func TestEscalatedClaimIsRefusedAtTheCap(t *testing.T) { + conn := mustDB(t) + runID, _ := budgetRun(t, conn, 0) + cost := expectedCostOf(t, conn, "implement@0") + // Room for the declared cost with headroom to spare, but not for 4x it. + execSQL(t, conn, `UPDATE runs SET budget = ? WHERE id = ?`, 2*cost, runID) + + _, err := ClaimStep(conn, stepIDByInstance(t, conn, "implement@0"), + ClaimOptions{Owner: "worker", CostMultiplier: 4, NowMS: nowMS}) + if err == nil { + t.Fatal("the escalated claim that crosses the cap was admitted — " + + "the escalation hop is still invisible to the budget") + } + if code, _ := CodeOf(err); code != CodeConflict { + t.Errorf("the refusal is %v, want %v", code, CodeConflict) + } + if !strings.Contains(err.Error(), "budget") { + t.Errorf("the refusal does not name the budget: %v", err) + } + + // The refusal and the pause are one fact (B20), same as every breach. + run, err := db.GetRun(conn, runID) + testsupport.Must(t, err, "GetRun: %v", err) + if run.Status != model.RunWaitingHuman { + t.Errorf("run is %s after the escalated breach, want %s", + run.Status, model.RunWaitingHuman) + } + + // Nothing accrued: the refused claim wrote no claim event, so the floor + // still reads zero — a refusal must not cost anything. + if got := runFloor(t, conn, runID); got != 0 { + t.Errorf("floor = %g after the refused escalated claim, want 0", got) + } +} + +// TestEscalatedRetryReAccruesItsOwnCost is B9's re-accrual made variant-aware +// — the DOT-846 shape: a cheap first attempt, then an escalated retry of the +// SAME step. Each claim accrues what ITS dispatcher declared, so the retry's +// hop lands in the floor at its real price. +func TestEscalatedRetryReAccruesItsOwnCost(t *testing.T) { + conn := mustDB(t) + runID, _ := budgetRun(t, conn, 0) + cost := expectedCostOf(t, conn, "implement@0") + + claimInstance(t, conn, "implement@0", nowMS) + if got := runFloor(t, conn, runID); got != cost { + t.Fatalf("floor after the cheap first attempt = %g, want %g", got, cost) + } + + // The lease lapses; the dispatcher escalates the retry to a 4x variant. + execSQL(t, conn, `UPDATE steps SET expires_ms = 1 WHERE instance = 'implement@0'`) + _, err := ClaimStep(conn, stepIDByInstance(t, conn, "implement@0"), + ClaimOptions{Owner: "w2", CostMultiplier: 4, NowMS: nowMS + 1}) + testsupport.Must(t, err, "escalated retry: %v", err) + + if got, want := runFloor(t, conn, runID), cost+4*cost; got != want { + t.Errorf("floor after the escalated retry = %g, want %g — each claim "+ + "accrues what its own dispatcher declared", got, want) + } +} + +// TestCheaperVariantAdmitsWhereDeclaredWouldRefuse is the multiplier's other +// direction: a cap the DECLARED cost would cross admits a claim the +// dispatcher routed to a cheaper variant, and the floor records the honest +// smaller number. Budget is §6.3's last clause, so nothing else can hide +// behind the fall-through. +func TestCheaperVariantAdmitsWhereDeclaredWouldRefuse(t *testing.T) { + conn := mustDB(t) + runID, _ := budgetRun(t, conn, 0) + cost := expectedCostOf(t, conn, "implement@0") + // Below the declared cost: an unscaled claim would breach here. + execSQL(t, conn, `UPDATE runs SET budget = ? WHERE id = ?`, cost/2, runID) + + _, err := ClaimStep(conn, stepIDByInstance(t, conn, "implement@0"), + ClaimOptions{Owner: "worker", CostMultiplier: 0.25, NowMS: nowMS}) + testsupport.Must(t, err, "cheap-variant claim under a tight cap: %v", err) + + if got, want := runFloor(t, conn, runID), cost*0.25; got != want { + t.Errorf("floor = %g, want %g — the accrual is the scaled cost", got, want) + } + + // The run is still active: no breach was recorded on the admitted claim. + run, err := db.GetRun(conn, runID) + testsupport.Must(t, err, "GetRun: %v", err) + if run.Status != model.RunActive { + t.Errorf("run is %s after an admitted cheap-variant claim, want %s", + run.Status, model.RunActive) + } +} + +// TestCostMultiplierValidation: a multiplier that is not a positive finite +// number is refused BEFORE the transaction opens — nothing mutates, no +// attempt is consumed, and the floor is untouched. +func TestCostMultiplierValidation(t *testing.T) { + conn := mustDB(t) + runID, _ := budgetRun(t, conn, 0) + stepID := stepIDByInstance(t, conn, "implement@0") + + for _, bad := range []float64{-1, math.NaN(), math.Inf(1), math.Inf(-1)} { + _, err := ClaimStep(conn, stepID, + ClaimOptions{Owner: "worker", CostMultiplier: bad, NowMS: nowMS}) + if err == nil { + t.Fatalf("cost multiplier %g was accepted", bad) + } + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("multiplier %g refused with %v, want %v", bad, code, CodeValidation) + } + } + + // Nothing mutated across all the refusals: the step is still unclaimed at + // attempt zero and the floor never moved. + var status string + var attempt int + err := conn.QueryRow( + `SELECT status, attempt FROM steps WHERE id = ?`, stepID).Scan(&status, &attempt) + testsupport.Must(t, err, "reading the step after refusals: %v", err) + if status != string(db.StepPending) || attempt != 0 { + t.Errorf("step is %s at attempt %d after refused claims, want %s at 0", + status, attempt, db.StepPending) + } + if got := runFloor(t, conn, runID); got != 0 { + t.Errorf("floor = %g after refused claims, want 0", got) + } +} diff --git a/internal/engine/dkt868_test.go b/internal/engine/dkt868_test.go new file mode 100644 index 00000000..468fea7a --- /dev/null +++ b/internal/engine/dkt868_test.go @@ -0,0 +1,305 @@ +package engine + +import ( + "database/sql" + "reflect" + "sort" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-868 — a run's tier routing was unaggregatable. +// +// `run report` rolled step metadata up as key -> distinct-values-with-counts, +// which answers a run-level question and destroys a per-step one: grouping BY +// KEY is exactly what discards which values two keys took TOGETHER on one step. +// A bag whose keys are a REQUEST and its RESOLUTION therefore had no reader. +// RUN-51's rollup published one key with no `low` value and its partner with +// one — a real mismatch, on exactly one step, that the document could not name +// — and recovering it meant `docket step show` per step: the audit that found +// this ran ~90 of them across 19 runs. +// +// THE DISPOSITION, and why not the other one. The issue offers an alternative: +// an `escalated` / `variant-resolved` event kind emitted when resolution +// differs from the request. That remedy is not available to core and must not +// be made available. Deciding that two values "differ" in a way worth an event +// requires core to know WHICH keys are the request and the resolution, which is +// R7's line (db.MetadataRollup's comment, TestMetadataRollupReadsNoKey) and +// docs/design/genericity.md's whole subject — and the vocabulary it would carry +// is the vocabulary scripts/qa/genericity.sh bans from core surface by name. +// The event set is closed besides (event.go), and admits kinds for TRANSITIONS +// CORE PERFORMS; a dispatcher resolving a step to a variant is not one core +// observes at all — DKT-867 had to have the dispatcher DECLARE its cost scaling +// for precisely that reason. +// +// So the remedy is the read half: publish the bag the rollup collapsed, per +// step, joined to that step's status. Core still reads no key; the consumer +// makes the comparison, in one `run report` instead of an N-step-show sweep. +// +// The keys below are the corpus's own (`*_requested` / `*_resolved`) in the +// escalation cases and a deliberately unrelated vocabulary in the genericity +// case, because core must not be able to tell the two apart. + +// tierAudit is the retro sweep this issue exists to make possible, written the +// way a consumer would write it against one `run report`: find every step whose +// resolution disagrees with its request. +// +// It is a TEST-SIDE function on purpose. Core publishes the pairs and never +// makes this comparison — that is the genericity line, and a helper living here +// rather than in the engine is what keeps the test honest about which side of +// it the knowledge lives on. +func tierAudit(report *RunReport, requested, resolved string) []string { + var drifted []string + for _, a := range report.Attempts { + want, hasWant := a.Metadata[requested] + got, hasGot := a.Metadata[resolved] + if hasWant && hasGot && want != got { + drifted = append(drifted, a.Instance) + } + } + sort.Strings(drifted) + return drifted +} + +// setStepBags writes one bag per instance directly to the column, which is +// where every writer — definition, claim, completion, fail, annotate — lands +// after its own merge. The subject here is the READ, so the fixture states the +// stored shape rather than driving four verbs to produce it. +func setStepBags(t *testing.T, conn *sql.DB, bags map[string]string) { + t.Helper() + for instance, bag := range bags { + execSQL(t, conn, `UPDATE steps SET metadata = ? WHERE instance = ?`, + bag, instance) + } +} + +// TestReportPairsRequestedWithResolvedPerStep is the defect, fixed: one read of +// one run names the drifted step, and names only it. +func TestReportPairsRequestedWithResolvedPerStep(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + steps := runSteps(t, conn, run.ID) + if len(steps) < 3 { + t.Fatalf("the fixture expanded %d steps; this case needs three", len(steps)) + } + // Two steps served the tier they were dispatched at, one did not — RUN-51's + // shape, where `effort_resolved` held a `low` that `effort_requested` never + // showed. + setStepBags(t, conn, map[string]string{ + steps[0].Instance: `{"effort_requested":"high","effort_resolved":"high"}`, + steps[1].Instance: `{"effort_requested":"high","effort_resolved":"low"}`, + steps[2].Instance: `{"effort_requested":"high","effort_resolved":"high"}`, + }) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + // THE ROLLUP ALONE CANNOT DO IT — the half of the document that existed + // before. It publishes the anomaly (a value under `effort_resolved` that + // `effort_requested` never took) and carries no step identity anywhere, so + // the reader who sees it has nowhere to go but `step show`. + for _, key := range report.Metadata { + for _, v := range key.Values { + for _, s := range steps { + if v.Value == s.Instance { + t.Fatalf("the rollup names a step (%q); this test's premise "+ + "is that it cannot", v.Value) + } + } + } + } + + drifted := tierAudit(report, "effort_requested", "effort_resolved") + if len(drifted) != 1 || drifted[0] != steps[1].Instance { + t.Fatalf("one read of the report found drift on %v, want exactly [%s] — "+ + "the pairing the rollup collapses is what makes a retro an "+ + "N-step-show sweep (DKT-868)", drifted, steps[1].Instance) + } + + // And the row that names it carries what a reader needs to act: the step + // id, the issue, and the status the step ended in. + var row StepAttempt + for _, a := range report.Attempts { + if a.Instance == steps[1].Instance { + row = a + } + } + if row.Step == "" || row.Status == "" { + t.Errorf("the drifted row is %+v; a per-step fact with no step id or "+ + "status is not aimable", row) + } +} + +// TestFailedStepCarriesItsDispatchBag is the consequence the issue calls out as +// invisible entirely: drift concentrated in FAILURES. +// +// The write half already exists — `step claim --metadata` (DKT-592) lands the +// dispatcher's bag before the work runs, and no failure path touches the +// column. What was missing was a reader that kept the half-bag attached to the +// status explaining why it is a half: in the rollup, a step that recorded a +// request and never a resolution is indistinguishable from a completed step +// whose two keys happened to agree. +func TestFailedStepCarriesItsDispatchBag(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + id := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, id, ClaimOptions{ + Owner: "worker", NowMS: nowMS, + Metadata: `{"effort_requested":"high"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + // The executor dies before it can report what it resolved to. + testsupport.Must(t, e.FailStep(conn, id, claim.Token, "crashed", "", nowMS+1), + "fail: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS+2) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + var row StepAttempt + for _, a := range report.Attempts { + if a.Instance == "implement@0" { + row = a + } + } + if row.Metadata["effort_requested"] != "high" { + t.Fatalf("the failed step's row carries metadata %v; the dispatcher's "+ + "claim-time bag must survive a failure (DKT-592) and reach the "+ + "report (DKT-868)", row.Metadata) + } + if _, ok := row.Metadata["effort_resolved"]; ok { + t.Fatalf("the fixture's failed step somehow reported a resolution: %v", + row.Metadata) + } + // The status is what makes the missing half readable as an absence rather + // than as agreement. + if row.Status == db.StepDone { + t.Errorf("the row reports %q; a half-bag on a step that did not "+ + "complete must be attributable to the failure", row.Status) + } +} + +// TestUnescalatedRunReportsAsBefore is the regression half: a run where every +// step served the tier it was asked for is unchanged in every way that was ever +// published. +func TestUnescalatedRunReportsAsBefore(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + steps := runSteps(t, conn, run.ID) + if len(steps) < 2 { + t.Fatalf("the fixture expanded %d steps; this case needs two", len(steps)) + } + for _, s := range steps[:2] { + execSQL(t, conn, `UPDATE steps SET metadata = ? WHERE id = ?`, + `{"effort_requested":"high","effort_resolved":"high"}`, s.ID) + } + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + // The rollup is byte-for-byte what it always was. + want := []db.MetadataKeyRollup{ + {Key: "effort_requested", Values: []db.MetadataValueCount{ + {Value: "high", Count: 2}}}, + {Key: "effort_resolved", Values: []db.MetadataValueCount{ + {Value: "high", Count: 2}}}, + } + if !reflect.DeepEqual(report.Metadata, want) { + t.Errorf("rollup = %+v, want %+v", report.Metadata, want) + } + if drifted := tierAudit(report, "effort_requested", "effort_resolved"); len(drifted) > 0 { + t.Errorf("an unescalated run reports drift on %v", drifted) + } + // A step that carried no bag carries no key on the wire either — nil, so + // `omitempty` elides it exactly as before. A `{}` on every step of every + // report in the store would be this change leaking into runs it has nothing + // to say about. + for _, a := range report.Attempts { + if a.Instance == steps[0].Instance || a.Instance == steps[1].Instance { + continue + } + if a.Metadata != nil || a.MetadataUnreadable { + t.Errorf("%s reports metadata %v (unreadable=%v) and never had a bag", + a.Instance, a.Metadata, a.MetadataUnreadable) + } + } +} + +// TestStepMetadataIsVerbatimAndUninterpreted is R7 held through the new +// section: two unrelated vocabularies arrive identically, because core does not +// know either of them, and a non-string value is neither coerced nor dropped. +func TestStepMetadataIsVerbatimAndUninterpreted(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + steps := runSteps(t, conn, run.ID) + if len(steps) < 2 { + t.Fatalf("the fixture expanded %d steps; this case needs two", len(steps)) + } + setStepBags(t, conn, map[string]string{ + steps[0].Instance: `{"desk_requested":"front","desk_resolved":"back"}`, + steps[1].Instance: `{"sirens":3,"nested":{"a":1}}`, + }) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + // A vocabulary nobody would use pairs exactly as the corpus's own does. + if drifted := tierAudit(report, "desk_requested", "desk_resolved"); len(drifted) != 1 { + t.Errorf("the audit found %v on a `desk` bag; the report must not be "+ + "able to tell one opaque vocabulary from another", drifted) + } + + var bag map[string]any + for _, a := range report.Attempts { + if a.Instance == steps[1].Instance { + bag = a.Metadata + } + } + if bag["sirens"] != float64(3) { + t.Errorf("a numeric value came through as %#v, want the number verbatim", + bag["sirens"]) + } + if _, ok := bag["nested"].(map[string]any); !ok { + t.Errorf("a nested object came through as %#v, want it verbatim", + bag["nested"]) + } +} + +// TestUnreadableStepBagIsNotSilence is R10's tolerance, made legible. A stored +// bag that does not decode must not fail the report — a read verb that refused +// over one odd cell is useless during exactly the run an operator wants to +// inspect — and must not vanish either, because on a PER-STEP row an absent bag +// reads as "the dispatcher recorded nothing", which is the comfortable claim +// and the wrong one. +func TestUnreadableStepBagIsNotSilence(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + steps := runSteps(t, conn, run.ID) + execSQL(t, conn, `UPDATE steps SET metadata = ? WHERE id = ?`, + `["front"]`, steps[0].ID) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + for _, a := range report.Attempts { + if a.Instance != steps[0].Instance { + continue + } + if !a.MetadataUnreadable { + t.Fatalf("%s stores a non-object bag and its row says nothing about "+ + "it (metadata=%v)", a.Instance, a.Metadata) + } + if a.Metadata != nil { + t.Errorf("%s reports a decoded bag %v from bytes that do not decode", + a.Instance, a.Metadata) + } + } +} diff --git a/internal/engine/dkt869_test.go b/internal/engine/dkt869_test.go new file mode 100644 index 00000000..5585a595 --- /dev/null +++ b/internal/engine/dkt869_test.go @@ -0,0 +1,456 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-869: the authorized mid-run widen, made executable. +// +// RUN-52 (VPL-434), 2026-08-26 retro: "scope snapshots per-step at creation and +// does not reach the already-created fix@2 step — no engine verb refreshes an +// existing step's scope". The panel had rejected the work as out of scope, the +// operator had AGREED and widened it, and the honest remedy still could not be +// rendered into the step that was to perform it, so the issue was abandoned +// mid-loop. DKT-741 pinned the freeze (dkt741_test.go) and named abandon + +// re-plan as the only disposition; this is the narrower one, for the case where +// the run's premise is intact and one declaration was corrected. +// +// What is asserted here is the whole bargain: the refresh reaches the pending +// step, it copies ONLY what the gated writer declared, it refuses every state +// where a step could straddle the change, it rewrites nothing else in the +// snapshot, and it leaves the discontinuity in the ledger. The regression guard +// for "a step never refreshed behaves exactly as before" is +// TestScopeEditDoesNotReachAnActivatedPacket in dkt741_test.go, which still +// runs unchanged — the freeze is still the default, and this verb is the +// exception to it. + +// refresh drives the engine entry point with the fixture's reason. +func refresh(conn *sql.DB, runID, issueID int) (*RefreshedScope, error) { + return RefreshIssueScopeInRun(conn, runID, issueID, "scope widened", nowMS) +} + +// widen is the authorized act the refresh copies: the ONLY writer of +// `issues.scope_globs`, exactly as `issue edit --scope` calls it. +func widen(t *testing.T, conn *sql.DB, issueID int, globsJSON string) { + t.Helper() + testsupport.Must(t, db.SetIssueScopeGlobs(conn, issueID, globsJSON), + "widening scope") +} + +// TestRefreshReachesTheAlreadyCreatedStep is the acceptance criterion, in the +// shape RUN-52 met it: a step that already exists renders the widened scope +// after the refresh, and its diff will be recorded over the same paths. +func TestRefreshReachesTheAlreadyCreatedStep(t *testing.T) { + conn := mustDB(t) + runID, issue, stepID := scopedIssueInRun(t, conn, `["cli/src/command/start.rs"]`) + + before, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "rendering before the widen: %v", err) + if !strings.Contains(before.Packet, "cli/src/command/start.rs") { + t.Fatalf("premise: the packet does not carry the declared scope:\n%s", + before.Packet) + } + + widen(t, conn, issue, `["cli/src/command/start.rs","script/install.sh","makefile"]`) + + // The edit ALONE still does not reach it — DKT-741's freeze is the default + // and this test must not pass because the freeze quietly stopped holding. + stillFrozen, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "rendering after the widen: %v", err) + if stillFrozen.Packet != before.Packet { + t.Fatalf("the widen alone changed the packet; §9 item 5's edit "+ + "immunity requires the refresh to be a separate act:\n%s", + stillFrozen.Packet) + } + + outcome, err := refresh(conn, runID, issue) + testsupport.Must(t, err, "refreshing: %v", err) + + after, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "rendering after the refresh: %v", err) + for _, added := range []string{"script/install.sh", "makefile"} { + if !strings.Contains(after.Packet, added) { + t.Errorf("the packet does not carry %q after the refresh:\n%s", + added, after.Packet) + } + } + + // The diff scope reads the same blob, so the packet and the recorded diff + // still cannot disagree — the refresh moved BOTH or it moved neither. + frozen, err := snapshotScope(conn, runID, issue) + testsupport.Must(t, err, "reading the snapshot scope: %v", err) + if len(frozen) != 3 || frozen[0] != "cli/src/command/start.rs" || + frozen[1] != "script/install.sh" || frozen[2] != "makefile" { + t.Errorf("snapshotScope = %v, want the widened declaration in the "+ + "author's order", frozen) + } + + if len(outcome.From) != 1 || len(outcome.To) != 3 { + t.Errorf("outcome = %+v, want the one-path scope replaced by the three-path one", + outcome) + } + if len(outcome.Steps) != 1 || outcome.Steps[0] != "flaky@0" { + t.Errorf("outcome.Steps = %v, want the one live instance", outcome.Steps) + } +} + +// TestRefreshRefusesWithoutTheWiden is the GATE, and the reason this verb takes +// no `--scope` of its own: it can only copy what `issue create|edit --scope` +// declared, so a refresh nobody authorized a widen for has nothing to make +// real. Without this refusal the verb would be a second, ungated writer of what +// a live run renders. +func TestRefreshRefusesWithoutTheWiden(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + + _, err := refresh(conn, runID, issue) + if err == nil { + t.Fatal("a refresh with no declared widen behind it was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + // It must name the gated verb: an operator here has typed the second act + // without the first, and the fix is the first. + if !strings.Contains(err.Error(), "issue edit") { + t.Errorf("the refusal does not name the gated writer:\n%s", err) + } + assertNoRefreshEvent(t, conn, runID) + + // A RE-DECLARATION of the same globs is the same non-authorization: the + // operator wrote the column, but nothing about the run changed, and an + // event for it would be a ruling that ruled nothing. + widen(t, conn, issue, `["internal/a/**"]`) + if _, err := refresh(conn, runID, issue); err == nil { + t.Error("a refresh over an unchanged declaration was accepted") + } + assertNoRefreshEvent(t, conn, runID) +} + +// TestRefreshRefusesAStraddlingStep is the quiescence half — repin's rule +// (repin.go) applied to the other frozen premise. A step holding a packet +// rendered under the old scope must not record its diff under the new one. +func TestRefreshRefusesAStraddlingStep(t *testing.T) { + for _, status := range []string{db.StepClaimed, db.StepRunning, db.StepGated} { + t.Run(status, func(t *testing.T) { + conn := mustDB(t) + runID, issue, stepID := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + mustExec(t, conn, `UPDATE steps SET status = ? WHERE id = ?`, + status, stepID) + + _, err := refresh(conn, runID, issue) + if err == nil { + t.Fatalf("a refresh under a %s step was accepted", status) + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + for _, want := range []string{"flaky@0", status} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q:\n%s", want, err) + } + } + assertFrozenScope(t, conn, runID, issue, "internal/a/**") + assertNoRefreshEvent(t, conn, runID) + }) + } +} + +// TestRefreshRefusesAnOpenDispatch: a manifest is a frozen offer, and its relay +// must see the world before or after the refresh, never both. +func TestRefreshRefusesAnOpenDispatch(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + mustExec(t, conn, + `INSERT INTO dispatches (run_id, status, opened_seq, expires_ms, created_at_ms) + VALUES (?, 'open', 0, ?, ?)`, runID, nowMS+60_000, nowMS) + + _, err := refresh(conn, runID, issue) + if err == nil { + t.Fatal("a refresh under an open dispatch was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + if !strings.Contains(err.Error(), "DISPATCH-") { + t.Errorf("the refusal does not name the open dispatch:\n%s", err) + } + assertFrozenScope(t, conn, runID, issue, "internal/a/**") +} + +// TestRefreshRefusesWhereItCouldOnlyRewriteHistory covers the states in which a +// refresh means nothing: nothing left to render, a run that froze nothing yet, +// a terminal run, an issue this run does not hold. +func TestRefreshRefusesWhereItCouldOnlyRewriteHistory(t *testing.T) { + t.Run("every step of the issue is terminal", func(t *testing.T) { + conn := mustDB(t) + runID, issue, stepID := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + mustExec(t, conn, `UPDATE steps SET status = ? WHERE id = ?`, + db.StepDone, stepID) + + _, err := refresh(conn, runID, issue) + if err == nil { + t.Fatal("a refresh over a fully-recorded issue was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + assertFrozenScope(t, conn, runID, issue, "internal/a/**") + }) + + t.Run("a run that never activated", func(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(parkingWorkflow), "parks.toml") + issue := createIssue(t, conn, "not yet", "body", "task", nil) + widen(t, conn, issue, `["internal/a/**"]`) + run := startRun(t, conn, issue) + + _, err := refresh(conn, run.ID, issue) + if err == nil { + t.Fatal("a refresh on a planning run was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + }) + + t.Run("a terminal run", func(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + mustExec(t, conn, `UPDATE runs SET status = ? WHERE id = ?`, + string(model.RunAbandoned), runID) + + _, err := refresh(conn, runID, issue) + if err == nil { + t.Fatal("a refresh on an abandoned run was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + }) + + t.Run("an issue the run does not hold", func(t *testing.T) { + conn := mustDB(t) + runID, _, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + other := createIssue(t, conn, "elsewhere", "body", "task", nil) + + _, err := refresh(conn, runID, other) + if err == nil { + t.Fatal("a refresh named an issue the run does not hold") + } + if code, ok := CodeOf(err); !ok || code != CodeNotFound { + t.Errorf("error code = %v, want NOT_FOUND: %v", code, err) + } + }) + + t.Run("no reason", func(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + + _, err := RefreshIssueScopeInRun(conn, runID, issue, " ", nowMS) + if err == nil { + t.Fatal("a refresh with no reason was accepted") + } + if code, ok := CodeOf(err); !ok || code != CodeValidation { + t.Errorf("error code = %v, want VALIDATION_ERROR: %v", code, err) + } + assertFrozenScope(t, conn, runID, issue, "internal/a/**") + }) +} + +// TestRefreshRewritesTheScopeAndNothingElse is the immunity that SURVIVES the +// exception. The snapshot also carries the title, kind and labels a mid-run +// edit must never reach — labels decide how a step routes — so the refresh has +// to be surgical, not a re-snapshot. +func TestRefreshRewritesTheScopeAndNothingElse(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(parkingWorkflow), "parks.toml") + issue := createIssue(t, conn, "widen me", "body", "task", []string{"alpha", "beta"}) + widen(t, conn, issue, `["internal/a/**"]`) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + before := snapshotBlob(t, conn, run.ID, issue) + for _, want := range []string{`"title":"widen me"`, `"kind":"task"`, + `"labels":["alpha","beta"]`} { + if !strings.Contains(before, want) { + t.Fatalf("premise: the snapshot does not carry %s — this test "+ + "would pass vacuously:\n%s", want, before) + } + } + + // Everything a mid-run edit could touch, edited mid-run. + mustExec(t, conn, + `UPDATE issues SET title = ?, kind = ? WHERE id = ?`, + "renamed after activation", "bug", issue) + mustExec(t, conn, `DELETE FROM issue_labels WHERE issue_id = ?`, issue) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + + if _, err := refresh(conn, run.ID, issue); err != nil { + t.Fatalf("refreshing: %v", err) + } + + after := snapshotBlob(t, conn, run.ID, issue) + var was, now map[string]any + testsupport.Must(t, json.Unmarshal([]byte(before), &was), "decoding before") + testsupport.Must(t, json.Unmarshal([]byte(after), &now), "decoding after") + + for _, key := range []string{"title", "kind", "labels"} { + if fmtJSON(was[key]) != fmtJSON(now[key]) { + t.Errorf("%s moved from %s to %s; only scope may move", + key, fmtJSON(was[key]), fmtJSON(now[key])) + } + } + if fmtJSON(now["scope"]) != `["internal/a/**","internal/b/**"]` { + t.Errorf("scope = %s, want the widened declaration", fmtJSON(now["scope"])) + } +} + +// TestRefreshedSnapshotIsByteIdenticalApartFromScope pins the round trip the +// rewrite depends on. reScopedSnapshot decodes and re-encodes through +// `issueSnapshotFields`, so a snapshot whose scope is replaced with the SAME +// value must come back byte for byte — anything else means a key was dropped +// or the canonical order moved, and either would corrupt a live run's snapshot +// while reporting success. +func TestRefreshedSnapshotIsByteIdenticalApartFromScope(t *testing.T) { + for _, blob := range []string{ + `{"title":"t","kind":"task","labels":[],"scope":[]}`, + `{"title":"t","kind":"bug","labels":["a","b"],"scope":["x/**","y/**"]}`, + `{"title":"t","kind":"task","labels":["a"],"scope":["x"],"linked":{"blocks.diff":[7,9]}}`, + } { + scope, err := decodeSnapshotScope(blob) + testsupport.Must(t, err, "decoding %s: %v", blob, err) + out, err := reScopedSnapshot(blob, scope) + testsupport.Must(t, err, "re-encoding %s: %v", blob, err) + if out != blob { + t.Errorf("re-encoding with the same scope changed the blob:\n"+ + "before: %s\nafter: %s", blob, out) + } + } +} + +// TestRefreshRecordsTheDiscontinuity is what keeps this an exception rather +// than a hole: two steps of one run rendering two different scopes is legible +// only because one dated, attributable event says when the premise moved and +// why. +func TestRefreshRecordsTheDiscontinuity(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + + if _, err := refresh(conn, runID, issue); err != nil { + t.Fatalf("refreshing: %v", err) + } + + var data string + err := conn.QueryRow( + `SELECT data FROM events WHERE kind = ? AND run_id = ? AND issue_id = ?`, + EventIssueScopeRefreshed, runID, issue).Scan(&data) + testsupport.Must(t, err, "reading the refresh event: %v", err) + + var payload struct { + Issue string `json:"issue"` + Reason string `json:"reason"` + From []string `json:"from"` + To []string `json:"to"` + Steps []string `json:"steps"` + } + testsupport.Must(t, json.Unmarshal([]byte(data), &payload), "decoding the event") + + if payload.Issue != model.FormatID(issue) || payload.Reason != "scope widened" { + t.Errorf("event payload = %+v, want the issue and the operator's reason", payload) + } + if len(payload.From) != 1 || len(payload.To) != 2 { + t.Errorf("event payload = %+v, want BOTH scopes — the old one is what "+ + "the recorded diffs were computed over", payload) + } + if len(payload.Steps) != 1 || payload.Steps[0] != "flaky@0" { + t.Errorf("event payload steps = %v, want the reached instance", payload.Steps) + } + + // §9 item 2: the kind is attributable, and it is a person's act. + actor, ok := ActorFor(EventIssueScopeRefreshed) + if !ok || actor != ActorHuman { + t.Errorf("ActorFor(%s) = %v/%v, want ActorHuman", + EventIssueScopeRefreshed, actor, ok) + } +} + +// TestScopeEditAdvisoryNamesTheRefresh: DKT-741's disclosure fires at the +// moment an operator spends a widen on a live run, and it is the only place +// this verb is discoverable from. An advisory still naming only abandon + +// re-plan would leave RUN-52's conductor exactly where it was. +func TestScopeEditAdvisoryNamesTheRefresh(t *testing.T) { + conn := mustDB(t) + runID, issue, _ := scopedIssueInRun(t, conn, `["internal/a/**"]`) + widen(t, conn, issue, `["internal/a/**","internal/b/**"]`) + + warnings := ScopeEditFrozenForActiveRuns(conn, issue) + if len(warnings) != 1 { + t.Fatalf("got %d warnings, want 1: %v", len(warnings), warnings) + } + want := "run refresh-scope " + model.FormatRunID(runID) + + " --issue " + model.FormatID(issue) + if !strings.Contains(warnings[0], want) { + t.Errorf("the advisory does not name %q:\n%s", want, warnings[0]) + } +} + +func snapshotBlob(t *testing.T, conn *sql.DB, runID, issueID int) string { + t.Helper() + var blob string + err := conn.QueryRow( + `SELECT issue_snapshot FROM run_issues WHERE run_id = ? AND issue_id = ?`, + runID, issueID).Scan(&blob) + testsupport.Must(t, err, "reading the snapshot: %v", err) + return blob +} + +func assertFrozenScope(t *testing.T, conn *sql.DB, runID, issueID int, want ...string) { + t.Helper() + got, err := snapshotScope(conn, runID, issueID) + testsupport.Must(t, err, "reading the snapshot scope: %v", err) + if len(got) != len(want) { + t.Fatalf("snapshotScope = %v, want %v — a refused refresh must write "+ + "nothing", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("snapshotScope = %v, want %v", got, want) + } + } +} + +func assertNoRefreshEvent(t *testing.T, conn *sql.DB, runID int) { + t.Helper() + var n int + err := conn.QueryRow( + `SELECT COUNT(*) FROM events WHERE kind = ? AND run_id = ?`, + EventIssueScopeRefreshed, runID).Scan(&n) + testsupport.Must(t, err, "counting refresh events: %v", err) + if n != 0 { + t.Errorf("%d %s event(s) recorded by a refused refresh", + n, EventIssueScopeRefreshed) + } +} + +func fmtJSON(v any) string { + out, err := json.Marshal(v) + if err != nil { + return "" + } + return string(out) +} diff --git a/internal/engine/dkt870_test.go b/internal/engine/dkt870_test.go new file mode 100644 index 00000000..6bdb1858 --- /dev/null +++ b/internal/engine/dkt870_test.go @@ -0,0 +1,459 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-870: fix-loop non-convergence was invisible to the engine. The 2026-08-26 +// retro across 19 fix-loop runs found three shapes, all ending only by operator +// action; two are closed here and one was already closed: +// +// 1. FLAT VOLUME NEVER TRIPPED ANYTHING. RUN-51 held 8-12 clusters across TEN +// rounds (~271k + ~251k output tokens on the last two alone, after the +// run's own ruling that the defect was structural); RUN-50 held 7-10 +// across six, ended by a hand-broken deadlock. DKT-340 saw moving trees +// and DKT-589 saw byte-distinct verdicts, so both stayed silent. The +// author now declares `max_stalled_rounds` on the routing step (V38), and +// a loop entry after that many consecutive measured rounds without a new +// minimum routed volume parks in the non-convergence refusal's exact +// shape — `--as fix-round` stays the way out. +// +// 2. EMPTY WORK LISTS RAN FULL ROUNDS. Already closed by DKT-588 +// (dkt588_test.go drives the verbatim RUN-34 shape): a loop body handing +// back its previous round's commit parks at its source before the fanout. +// +// 3. LOOP EXIT WAS NOT GATED ON THE RECORDED PAYLOAD. RUN-58's reconcile@1 +// routed `pass` and the loop exited with all 16 clusters open, SIX at the +// order's high position, none held and none operator-resolved — +// "converged" in the ledger meaning "dispositioned". The author now +// declares `pass_floor = { field, at }` (V37/V37a), and a `pass` whose +// recorded payload still holds unexempt elements at or above the floor's +// position parks `waiting-human` instead, naming `--as override-pass` +// and `--as fix-round` as the ways out. +// +// Both knobs are OPT-IN (absent means the engine behaves exactly as before) +// and both are positional: field names and floor values are opaque tokens +// compared only by position in the pinned schema's order, so core learns +// nothing about severities (genericity.md). + +// floorWorkflowSrc is the RUN-58 shape as a workflow: the routing step's +// threshold reads `status`, so a payload whose elements are open at `high` +// severity but carry no unmet status routes `pass` — the exact self-reported +// "converged" that contradicted the recorded evidence. `pass_floor` is the +// declared exit bar; `hold_spread` is kept so the held-exemption path is the +// fixture's real one. +const floorWorkflowSrc = ` +[pipeline] +name = "floor-exit" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +class = "write" +emits = "change-summary" + +[[step]] +name = "scan" +after = ["implement"] +executor = "scan" +emits = "findings" + +[[step]] +name = "reconcile" +after = ["scan"] +action = "aggregate" +inputs = ["scan.findings"] +payload = "findings@1" +params = { field = "severity", method = "median", hold_spread = 2, output = "findings" } +threshold = { "fix-loop" = "any(status == unmet)" } +pass_floor = { field = "severity", at = "high" } +max_fix_loops = 4 + +[[step]] +name = "fix" +executor = "fix" +class = "write" +emits = "change-summary" +loop = true +inputs = ["reconcile.findings"] +after_loop = "scan" + +[[step]] +name = "verify" +after = ["reconcile"] +executor = "verify" +emits = "ac-report" +` + +// volumeWorkflowSrc is the RUN-51/RUN-50 shape: a loop whose routing step +// re-routes `fix-loop` every round over a standing set that never shrinks. +// `max_stalled_rounds = 2` is the declared tolerance; `max_fix_loops` is high +// enough that the plateau, not the budget, must be what stops the loop — +// which is the retro's own finding ("max_fix_loops never fired as the +// terminator in the runs where it existed"). +const volumeWorkflowSrc = ` +[pipeline] +name = "flat-volume" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +class = "write" +emits = "change-summary" + +[[step]] +name = "scan" +after = ["implement"] +executor = "scan" +emits = "findings" + +[[step]] +name = "reconcile" +after = ["scan"] +action = "aggregate" +inputs = ["scan.findings"] +payload = "findings@1" +params = { field = "severity", method = "median", output = "findings" } +threshold = { "fix-loop" = "any(severity >= high)" } +max_fix_loops = 8 +max_stalled_rounds = 2 + +[[step]] +name = "fix" +executor = "fix" +class = "write" +emits = "change-summary" +loop = true +inputs = ["reconcile.findings"] +after_loop = "scan" +` + +// activatedCustomRun is activatedRun over one of this file's workflows: the +// fixture's schema (the aggregate needs `findings@1`'s order), the custom +// TOML through the real parse-validate-lint path, one task issue, activation. +func activatedCustomRun(t *testing.T, conn *sql.DB, src string) (*model.Run, int) { + t.Helper() + registerFixtureSchema(t, conn) + registerSource(t, conn, []byte(src), "dkt870.toml") + issue := createIssue(t, conn, "converge the loop", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + return run, issue +} + +// driveFloorRound completes floor-exit's round 0 up to and including +// `reconcile@0`, with the scan payload under the test's control. +func driveFloorRound(t *testing.T, conn *sql.DB, e *Engine, payload string) { + t.Helper() + claimAndComplete(t, conn, e, "implement@0", "the change summary", "") + claimAndComplete(t, conn, e, "scan@0", "the scan", payload) + driveAction(t, conn, e, "reconcile@0") +} + +// TestPassWithStandingFloorPayloadParks is the verbatim RUN-58 shape: the +// threshold reads a field that says nothing is unmet, the payload's own +// severity evidence says otherwise, and the exit is refused. +func TestPassWithStandingFloorPayloadParks(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + driveFloorRound(t, conn, e, + `[{"id":"C-1","severity":"high"},{"id":"C-2","severity":"medium"}]`) + + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepWaitingHuman { + t.Errorf("reconcile@0 = %q, want the pass_floor park %q", + got, db.StepWaitingHuman) + } + // The park says WHY and NAMES THE WAYS OUT, like every refusal in the + // loop family. + routing := stepRoutingRaw(t, conn, "reconcile@0") + for _, want := range []string{"pass_floor", "severity >= high", + "override-pass", "fix-round"} { + if !strings.Contains(routing, want) { + t.Errorf("the park does not mention %q: %q", want, routing) + } + } + // And the chain did NOT advance: a parked pass hands nothing downstream. + if ready := readyInstances(t, conn); contains(ready, "verify@0") { + t.Error("verify@0 became ready past a parked pass; the floor park " + + "must stop the lineage at its source") + } +} + +// TestPassBelowTheFloorExitsClean is the regression half that matters most: a +// payload with nothing at the floor exits exactly as it always has. +func TestPassBelowTheFloorExitsClean(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + driveFloorRound(t, conn, e, `[{"id":"C-1","severity":"medium"}]`) + + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepDone { + t.Errorf("reconcile@0 = %q, want %q — a pass below the floor is a "+ + "genuine exit", got, db.StepDone) + } + if got := stepRouting(t, conn, "reconcile@0"); got != RoutingPass { + t.Errorf("routing = %q, want %q", got, RoutingPass) + } +} + +// TestFloorLeavesFixLoopRoutingAlone: the floor gates only the `pass` exit. A +// threshold that already decided to loop is not second-guessed — the loop IS +// the remedy the floor would otherwise ask an operator for. +func TestFloorLeavesFixLoopRoutingAlone(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + driveFloorRound(t, conn, e, + `[{"id":"C-1","severity":"high","status":"unmet"}]`) + + if !stepExists(t, conn, "fix@1") { + t.Error("the threshold's own fix-loop routing did not enter the loop; " + + "the floor must never override a routing that already decided") + } + if got := stepStatus(t, conn, "reconcile@0"); got == db.StepWaitingHuman { + t.Error("reconcile@0 parked on a fix-loop routing; the floor gates " + + "only the pass exit") + } +} + +// TestOperatorResolvedElementsDoNotBlockThePass drives the real held path: a +// cluster the spread held, an operator approving it AT a value above the +// floor, and the resumed pass standing — the decision channel already ran, and +// re-parking on it would ask the answered question again. +func TestOperatorResolvedElementsDoNotBlockThePass(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + // Spread 4 across the five-value order: held. + driveFloorRound(t, conn, e, `[{"id":"C-1","severity":["low","blocker"]}]`) + held := heldInstances(t, conn) + if len(held) != 1 { + t.Fatalf("premise: expected 1 held step, got %v", held) + } + + // The operator accepts the cluster at `blocker` — ABOVE the floor. The + // resumed routing must still pass: `operator_resolved` is the exemption. + err := e.DecideStepValue(conn, stepIDByInstance(t, conn, held[0]), + true, "ship it, tracked separately", "blocker", nowMS) + testsupport.Must(t, err, "approving %s: %v", held[0], err) + + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepDone { + t.Errorf("reconcile@0 = %q after an operator resolved the only "+ + "cluster, want %q — the floor must not re-ask an answered question", + got, db.StepDone) + } +} + +// TestOverridePassExitsTheFloorPark is the recorded operator exit the park +// names: the pass the floor refused, taken anyway, on an operator's authority. +func TestOverridePassExitsTheFloorPark(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + driveFloorRound(t, conn, e, `[{"id":"C-1","severity":"high"}]`) + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepWaitingHuman { + t.Fatalf("premise: reconcile@0 = %q, want the park", got) + } + + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "reconcile@0"), + ResolveOverridePass, "accepted; the opens are tracked in follow-ups", nowMS) + testsupport.Must(t, err, "resolve --as override-pass: %v", err) + + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepDone { + t.Errorf("reconcile@0 = %q after override-pass, want %q", got, db.StepDone) + } + if ready := readyInstances(t, conn); !contains(ready, "verify@0") { + t.Errorf("verify@0 is not ready after the override; the resolved pass "+ + "must hand the chain onward (ready: %v)", ready) + } +} + +// TestFixRoundBuysARoundFromTheFloorPark is the other way out the park names: +// instead of exiting over standing work, the operator mints the round that +// addresses it. +func TestFixRoundBuysARoundFromTheFloorPark(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, floorWorkflowSrc) + e := testEngine() + + driveFloorRound(t, conn, e, `[{"id":"C-1","severity":"high"}]`) + if got := stepStatus(t, conn, "reconcile@0"); got != db.StepWaitingHuman { + t.Fatalf("premise: reconcile@0 = %q, want the park", got) + } + + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "reconcile@0"), + ResolveFixRound, "fix the standing highs", nowMS) + testsupport.Must(t, err, "resolve --as fix-round: %v", err) + + if !stepExists(t, conn, "fix@1") { + t.Error("the operator authorized a round from the floor park and none " + + "was minted") + } +} + +// --------------------------------------------------------------------------- +// Flat volume (`max_stalled_rounds`) +// --------------------------------------------------------------------------- + +// volumePayload is one round's scan output: `volume` one-member clusters, ids +// unique to the round so no verdict is ever byte-identical across rounds — +// DKT-589's guard must stay silent for the plateau to be what parks. +func volumePayload(round, volume int) string { + elements := make([]string, 0, volume) + for i := range volume { + elements = append(elements, + fmt.Sprintf(`{"id":"C-%d-%d","severity":"high"}`, round, i)) + } + return "[" + strings.Join(elements, ",") + "]" +} + +// driveVolumeRound completes one flat-volume ordinal — the tree MOVES each +// round (driveFixtureRound), so DKT-340's guard stays silent too and the +// plateau is the only signal left standing, exactly as in RUN-51. +func driveVolumeRound(t *testing.T, conn *sql.DB, e *Engine, ordinal, volume int) { + t.Helper() + driveFixtureRound(t, ordinal) + if ordinal == 0 { + claimAndComplete(t, conn, e, "implement@0", "the change summary", "") + } else { + claimAndComplete(t, conn, e, fmt.Sprintf("fix@%d", ordinal), "the fix summary", "") + } + claimAndComplete(t, conn, e, fmt.Sprintf("scan@%d", ordinal), + "the scan", volumePayload(ordinal, volume)) + driveAction(t, conn, e, fmt.Sprintf("reconcile@%d", ordinal)) +} + +// TestFlatVolumeAcrossDeclaredRoundsParksTheLoop is the RUN-51 shape at test +// scale: every round moves the tree and reaches a byte-distinct verdict, and +// the standing volume never improves. After `max_stalled_rounds = 2` +// consecutive non-improving rounds the next entry is refused. +func TestFlatVolumeAcrossDeclaredRoundsParksTheLoop(t *testing.T) { + conn := mustDB(t) + run, issue := activatedCustomRun(t, conn, volumeWorkflowSrc) + e := testEngine() + + driveVolumeRound(t, conn, e, 0, 2) + if !stepExists(t, conn, "fix@1") { + t.Fatal("premise: round 0 must have entered the loop") + } + driveVolumeRound(t, conn, e, 1, 2) + if !stepExists(t, conn, "fix@2") { + t.Fatal("premise: one non-improving round is within tolerance") + } + driveVolumeRound(t, conn, e, 2, 2) + + if stepExists(t, conn, "fix@3") { + t.Error("a third round was minted after two consecutive rounds of " + + "flat volume; the declared tolerance must park the loop") + } + if got := stepStatus(t, conn, "reconcile@2"); got != db.StepWaitingHuman { + t.Errorf("reconcile@2 = %q, want the flat-volume park %q", + got, db.StepWaitingHuman) + } + routing := stepRoutingRaw(t, conn, "reconcile@2") + for _, want := range []string{"max_stalled_rounds", "fix-round"} { + if !strings.Contains(routing, want) { + t.Errorf("the park does not mention %q: %q", want, routing) + } + } + // The counter is put back, the bound refusal's own discipline (DKT-78): a + // refusal minted no ordinal. + if got := loopCount(t, conn, run.ID, issue); got != 2 { + t.Errorf("loop_count = %d after the refused entry, want 2", got) + } +} + +// TestShrinkingVolumeKeepsTheLoopRunning is the lower bound: a loop that IS +// converging sets a new minimum every round and must never park on this +// signal — a guard that fired on convergence would break every workflow that +// declares the tolerance. +func TestShrinkingVolumeKeepsTheLoopRunning(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, volumeWorkflowSrc) + e := testEngine() + + driveVolumeRound(t, conn, e, 0, 3) + driveVolumeRound(t, conn, e, 1, 2) + driveVolumeRound(t, conn, e, 2, 1) + + if !stepExists(t, conn, "fix@3") { + t.Error("a converging loop was refused a round; shrinking volume is " + + "progress and must keep the loop running") + } +} + +// TestNoiseAroundABestVolumeStillParks pins the measurement's semantics: "no +// improvement" means no NEW MINIMUM, not endpoint-to-endpoint decrease. +// RUN-51's volumes (8, 12, 10, 10, 7, 10, 7, 11, 10, ~8) fell between plenty +// of adjacent rounds while never trending anywhere; a consecutive-pairs +// reading would have stayed silent for exactly the run this rule exists for, +// and a tie with the best round is two rounds at the same wall, not progress. +func TestNoiseAroundABestVolumeStillParks(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, volumeWorkflowSrc) + e := testEngine() + + driveVolumeRound(t, conn, e, 0, 3) + driveVolumeRound(t, conn, e, 1, 2) // A new minimum: the streak resets. + driveVolumeRound(t, conn, e, 2, 3) // Worse than best: one stalled round. + if !stepExists(t, conn, "fix@3") { + t.Fatal("premise: one stalled round is within tolerance") + } + driveVolumeRound(t, conn, e, 3, 2) // TIES the best: still not progress. + + if stepExists(t, conn, "fix@4") { + t.Error("a round was minted after two rounds that never beat the best " + + "volume; oscillating around a floor is the RUN-51 shape and must park") + } + if got := stepStatus(t, conn, "reconcile@3"); got != db.StepWaitingHuman { + t.Errorf("reconcile@3 = %q, want the flat-volume park %q", + got, db.StepWaitingHuman) + } +} + +// TestFixRoundOverridesTheFlatVolumePark: the plateau park is the +// non-convergence refusal's shape, so its escape hatch is the same one — the +// operator who has read the park keeps the decision (DKT-237). +func TestFixRoundOverridesTheFlatVolumePark(t *testing.T) { + conn := mustDB(t) + activatedCustomRun(t, conn, volumeWorkflowSrc) + e := testEngine() + + driveVolumeRound(t, conn, e, 0, 2) + driveVolumeRound(t, conn, e, 1, 2) + driveVolumeRound(t, conn, e, 2, 2) + if got := stepStatus(t, conn, "reconcile@2"); got != db.StepWaitingHuman { + t.Fatalf("premise: reconcile@2 = %q, want the park", got) + } + + err := e.ResolveStep(conn, stepIDByInstance(t, conn, "reconcile@2"), + ResolveFixRound, "the plateau is expected here; one more round", nowMS) + testsupport.Must(t, err, "resolve --as fix-round: %v", err) + + if !stepExists(t, conn, "fix@3") { + t.Error("the operator authorized a round past the plateau and none " + + "was minted") + } +} diff --git a/internal/engine/dkt895_test.go b/internal/engine/dkt895_test.go new file mode 100644 index 00000000..1a2cdb47 --- /dev/null +++ b/internal/engine/dkt895_test.go @@ -0,0 +1,150 @@ +package engine + +import ( + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-895 — the `vote-tallied` event's detail rendered the WEIGHTED SCORE as a +// bare parenthesised number, which reads as a ballot count. +// +// RUN-62 is the measured case: a three-seat panel all cast +// approve-with-concerns, the tally scored 1.00 against a 67% threshold, and the +// feed said `DKT-V289 approved (1)`. A reader following the run saw what looked +// like a one-ballot approval on a three-seat panel — a panel failure — and +// could only disprove it by leaving the feed for `docket vote show`. + +// dkt895VoteSrc is that shape reduced: one declared vote gate, three voters, +// and a rejection that skips rather than looping (the loop is DKT-168's test, +// not this one). +const dkt895VoteSrc = ` +[pipeline] +name = "dkt895-vote" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "seed" +after = [] +executor = "x" +emits = "findings" + +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["seat-a", "seat-b", "seat-c"] +vote_rule = "majority" +on_fail = "skip" +` + +// TestTalliedVoteEventNamesScoreAndBallotsSeparately pins the remedy: the +// detail labels both numbers, so a 3-ballot unanimous approval scoring 1.00 +// cannot be read as one ballot. +func TestTalliedVoteEventNamesScoreAndBallotsSeparately(t *testing.T) { + conn := mustDB(t) + // RUN-62's threshold, verbatim. + registerVoteRule(t, conn, "majority", "0.67", "") + registerSource(t, conn, []byte(dkt895VoteSrc), "dkt895-vote.toml") + issue := createIssue(t, conn, "voted", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + + claimAndComplete(t, conn, e, "seed@0", "the findings", "") + err = e.DriveRunLifecycles(conn, run.ID, nowMS) + testsupport.Must(t, err, "driving after the record: %v", err) + + gate, err := db.GetStep(conn, stepIDByInstance(t, conn, "gate@0")) + testsupport.Must(t, err, "reading gate@0: %v", err) + proposalID, err := findVoteProposal(conn, gate) + testsupport.Must(t, err, "finding gate@0's proposal: %v", err) + if proposalID == 0 { + t.Fatal("no proposal opened for gate@0") + } + + // Three seats, all approve-with-concerns: db.CastVote's own arithmetic + // puts every cast's weight in the numerator, so the score is exactly 1. + for _, seat := range []string{"seat-a", "seat-b", "seat-c"} { + _, err := db.CastVote(conn, &model.Vote{ + ProposalID: proposalID, VoterName: seat, + Verdict: model.VerdictApproveWithConcerns, + Confidence: 0.9, DomainRelevance: 0.8, + }) + testsupport.Must(t, err, "CastVote(%s): %v", seat, err) + } + err = e.DriveVoteProposal(conn, proposalID, nowMS) + testsupport.Must(t, err, "driving the approved proposal: %v", err) + + var raw string + err = conn.QueryRow( + `SELECT data FROM events WHERE run_id = ? AND kind = ? ORDER BY seq DESC LIMIT 1`, + run.ID, EventVoteTallied).Scan(&raw) + testsupport.Must(t, err, "reading the vote-tallied event: %v", err) + + // `data` is the recorded envelope; `detail` is the string an operator reads + // in `docket events list`. + var envelope struct { + Detail string `json:"detail"` + } + testsupport.Must(t, json.Unmarshal([]byte(raw), &envelope), + "decoding the event data %q: %v", raw, err) + + want := model.FormatProposalID(proposalID) + " approved score=1.00 ballots=3/3" + if envelope.Detail != want { + t.Errorf("vote-tallied detail = %q, want %q", envelope.Detail, want) + } + + // The defect stated as its own assertion: the count-like rendering is gone. + if strings.Contains(envelope.Detail, "(1)") { + t.Errorf("vote-tallied detail %q still renders the score as a bare "+ + "count-like number", envelope.Detail) + } + + // And the detail agrees with `vote show`, digit for digit — the surface a + // reader had to leave the feed for. + proposal, err := db.GetProposal(conn, proposalID) + testsupport.Must(t, err, "GetProposal: %v", err) + votes, err := db.GetProposalVotes(conn, proposalID) + testsupport.Must(t, err, "GetProposalVotes: %v", err) + if proposal.WeightedScore == nil || *proposal.WeightedScore != 1 { + t.Fatalf("weighted_score = %v, want 1", proposal.WeightedScore) + } + if len(votes) != 3 || proposal.RequiredVoters != 3 { + t.Fatalf("ballots = %d/%d, want 3/3", len(votes), proposal.RequiredVoters) + } +} + +// TestTalliedVoteEventWithoutAScoreSaysSo: a proposal finalized with no +// weighted score (an operator's manual commit, §8.4) still labels both fields +// rather than printing a bare parenthesis. +func TestTalliedVoteEventWithoutAScoreSaysSo(t *testing.T) { + conn := mustDB(t) + registerVoteRule(t, conn, "majority", "0.6", "medium") + step, spec := seedVoteStep(t, conn) + + id, err := OpenVoteProposal(conn, step, spec, nowMS) + testsupport.Must(t, err, "OpenVoteProposal: %v", err) + _, err = conn.Exec(`UPDATE proposals SET status = ? WHERE id = ?`, + model.ProposalStatusCommitted, id) + testsupport.Must(t, err, "committing the proposal: %v", err) + + outcome, err := ReadVoteOutcome(conn, step, spec) + testsupport.Must(t, err, "ReadVoteOutcome: %v", err) + + detail, err := voteTallyDetail(conn, outcome) + testsupport.Must(t, err, "voteTallyDetail: %v", err) + + want := model.FormatProposalID(id) + " committed score=none ballots=0/3" + if detail != want { + t.Errorf("vote-tallied detail = %q, want %q", detail, want) + } +} diff --git a/internal/engine/dkt982_test.go b/internal/engine/dkt982_test.go new file mode 100644 index 00000000..4d1b4753 --- /dev/null +++ b/internal/engine/dkt982_test.go @@ -0,0 +1,150 @@ +package engine + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-982 — the failed gate a completion parks on has a NAME, and the engine +// has always known it at the moment the recording verb prints. +// +// `step record` printed `✔ Completed STEP-N (waiting-human)`: no gate, no +// verdict, a success glyph on a park. The executor read it as a pass and +// reported the wave green; the conductor reconstructed the real outcome from +// later engine reads. FailedGates is the accessor that was missing — these +// tests pin that it reports the SAME rows the routing decided on. + +// TestFailedGatesNamesTheGateThatParkedTheStep drives a REAL saga: a gate that +// exits 2 parks the step, and the accessor the CLI prints from names that gate +// and that exit code. +// +// It goes through CompleteStep rather than inserting rows because the claim in +// the issue is about what is knowable AT PRINT TIME — a test over hand-written +// rows would prove the reduction and leave the timing unproven. +func TestFailedGatesNamesTheGateThatParkedTheStep(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "dkt982.toml") + + e := testEngine() + e.Gates = &exitGates{fail: true, exit: 2} + + stepID := stepIDInRun(t, conn, runID, "implement@0") + parkThroughFailingGate(t, conn, e, stepID) + + failed, err := FailedGates(conn, stepID) + testsupport.Must(t, err, "FailedGates: %v", err) + + if len(failed) != 1 { + t.Fatalf("FailedGates = %d rows, want 1 — the park has exactly one cause "+ + "and the verb that prints it must be able to name it", len(failed)) + } + if failed[0].Gate != "build" { + t.Errorf("gate = %q, want %q", failed[0].Gate, "build") + } + if failed[0].Verdict != db.GateVerdictFail { + t.Errorf("verdict = %q, want %q", failed[0].Verdict, db.GateVerdictFail) + } + if failed[0].Exit == nil || *failed[0].Exit != 2 { + t.Errorf("exit = %v, want 2 — the exit code is the half of the report an "+ + "executor can act on", failed[0].Exit) + } +} + +// TestFailedGatesIsSilentWhenEveryGatePasses is the other half of the contract, +// and the one that protects the unchanged success line: a record whose gates +// all passed must give the printer nothing to say. +func TestFailedGatesIsSilentWhenEveryGatePasses(t *testing.T) { + conn := mustDB(t) + runID := activatedBatchRun(t, conn, batchOverrideSrc, "dkt982-pass.toml") + + e := testEngine() + e.Gates = &exitGates{} + + stepID := stepIDInRun(t, conn, runID, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the change summary"), NowMS: nowMS, + }) + testsupport.Must(t, err, "complete: %v", err) + + failed, err := FailedGates(conn, stepID) + testsupport.Must(t, err, "FailedGates: %v", err) + if len(failed) != 0 { + t.Errorf("FailedGates = %v over an all-passing record, want none", failed) + } +} + +// TestFailedGatesReportsWhatTheRoutingDecidedOn walks the reduction's edges over +// hand-built rows: the last attempt of a flaky re-run wins, a pre-gate is not a +// judgment of the step, `unmatched` and `skipped` are failures with no exit +// code, and the order is stable. +// +// The invariant at the end is the point of the whole exercise. A REPORT built +// from a second reduction over the same table is how a printed line comes to +// contradict the routing beside it — which is the defect class DKT-982 is — so +// the test asserts the two agree rather than asserting each in isolation. +func TestFailedGatesReportsWhatTheRoutingDecidedOn(t *testing.T) { + exit := func(n int) *int { return &n } + + rows := []db.GateResultRow{ + // A gate declared flaky: failed, then passed. F4 makes the last attempt + // the one that routes, so this must NOT be reported. + {Gate: "flaky", Ordinal: 0, Verdict: db.GateVerdictFail, Exit: exit(1)}, + {Gate: "flaky", Ordinal: 1, Verdict: db.GateVerdictPass, Exit: exit(0)}, + // A pre-gate that failed. PG4: an input to the step, not a judgment of + // it, and it routed nothing. + {Gate: "pre-scan", Ordinal: 0, Verdict: db.GateVerdictFail, Exit: exit(9), Pre: true}, + {Gate: "build", Ordinal: 0, Verdict: db.GateVerdictPass, Exit: exit(0)}, + {Gate: "self-hygiene", Ordinal: 0, Verdict: db.GateVerdictFail, Exit: exit(2)}, + // Never ran: no exit code exists, and 0 would read as a pass (T11). + {Gate: "audit", Ordinal: 0, Verdict: db.GateVerdictUnmatched, + Reason: "no trust entry matched"}, + {Gate: "coverage", Ordinal: 0, Verdict: db.GateVerdictSkipped, + Reason: "the tree was gone at spawn time"}, + } + + failed := failedGatesOverRows(rows) + + var names []string + for _, g := range failed { + names = append(names, g.Gate) + } + want := []string{"audit", "coverage", "self-hygiene"} + if len(names) != len(want) { + t.Fatalf("failed gates = %v, want %v", names, want) + } + for i := range want { + if names[i] != want[i] { + t.Fatalf("failed gates = %v, want %v (sorted, so one run's sentence is "+ + "every run's)", names, want) + } + } + + for _, g := range failed { + switch g.Gate { + case "self-hygiene": + if g.Exit == nil || *g.Exit != 2 { + t.Errorf("self-hygiene exit = %v, want 2", g.Exit) + } + case "audit", "coverage": + if g.Exit != nil { + t.Errorf("%s exit = %d, want none — nothing ran, and a zero exit "+ + "reads as a pass", g.Gate, *g.Exit) + } + if g.Reason == "" { + t.Errorf("%s carries no reason; the recorded row had one", g.Gate) + } + } + } + + // The invariant: the printer's rows and the router's verdict come from one + // reduction, so they cannot disagree. + verdict, _ := verdictOverRows(rows) + if (verdict == VerdictFail) != (len(failed) > 0) { + t.Errorf("verdict = %q with %d failed gates — the report and the routing "+ + "disagree about the same rows", verdict, len(failed)) + } +} diff --git a/internal/engine/dkt992_test.go b/internal/engine/dkt992_test.go new file mode 100644 index 00000000..d7b5a217 --- /dev/null +++ b/internal/engine/dkt992_test.go @@ -0,0 +1,209 @@ +package engine + +import ( + "context" + "os/exec" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/trust" +) + +// DKT-992 — gate processes learn the step's base commit, so a gate can scan +// exactly the step's committed range. +// +// Executors commit before `step record`, so at gate time a worktree-recorded +// step's tree is CLEAN: RUN-66's secret-scan passed 8/8 write steps having +// scanned zero lines, and range-shaped gates fell back to guesses +// (`git diff HEAD~1`, wrong for multi-commit steps). The engine knows the +// step's base at record time — the same fork point the diff stage resolves — +// and DOCKET_GATE_BASE is how a gate child finally learns it. The contract: +// set to the worktree's creation commit for a worktree-recorded step, UNSET +// for a non-worktree step (never an invented value), so a gate that finds it +// absent over a clean tree can fail closed. + +// baseCapture is a GateRunner that records the full StepContext each gate was +// dispatched with, keyed by gate name. +type baseCapture struct { + mu sync.Mutex + seen map[string]StepContext +} + +func (c *baseCapture) Run(_ context.Context, g GateSpec, sc StepContext) (GateResult, error) { + c.mu.Lock() + defer c.mu.Unlock() + c.seen[g.Name] = sc + return GateResult{Gate: g.Name, Verdict: VerdictPass}, nil +} + +// TestGateChildSeesDocketGateBase is the runner half, with a REAL spawn: the +// witness is printenv itself, so the assertion is about the environment the +// child actually received, not about a struct field. A context carrying a +// base exports it; one carrying none leaves the variable UNSET — printenv +// exits 1 having printed nothing, which is the child's own proof of absence. +func TestGateChildSeesDocketGateBase(t *testing.T) { + execRoot, worktree, execHead, _ := gitFixture(t) + printenvPath, err := exec.LookPath("printenv") + if err != nil { + t.Skip("printenv is not installed") + } + + argv := []string{printenvPath, "DOCKET_GATE_BASE"} + runner := NewExecRunner(testRepoPaths(execRoot)) + runner.LoadStore = sandboxTrust(t, trust.Entry{ + Name: "secret-scan", Argv: argv, ArgvSHA256: trust.ArgvSHA256(argv), + Repo: mustResolve(execRoot), + }) + + // The worktree-recorded shape: the fixture's worktree was created at the + // shared checkout's HEAD, so that commit is the base the saga resolves. + ex, err := runner.Execute(context.Background(), + GateSpec{Name: "secret-scan"}, + StepContext{WorkRoot: worktree, Base: execHead}) + testsupport.Must(t, err, "running the gate with a base: %v", err) + if ex.Verdict != VerdictPass { + t.Fatalf("gate verdict = %q (reason %q), want %q — printenv found no "+ + "DOCKET_GATE_BASE in the child environment", + ex.Verdict, ex.Results[0].Reason, VerdictPass) + } + if got := strings.TrimSpace(ex.Results[0].Output); got != execHead { + t.Errorf("the child saw DOCKET_GATE_BASE=%q, want the worktree's "+ + "creation commit %s", got, execHead) + } + + // The non-worktree shape: no base rides in the context, and the variable + // must be ABSENT — not empty — in the child. printenv exiting 1 with no + // output is the observable. + ex, err = runner.Execute(context.Background(), + GateSpec{Name: "secret-scan"}, StepContext{}) + testsupport.Must(t, err, "running the gate without a base: %v", err) + if got := strings.TrimSpace(ex.Results[0].Output); got != "" { + t.Errorf("the child saw DOCKET_GATE_BASE=%q for a non-worktree step, "+ + "want the variable unset", got) + } + if ex.Verdict != VerdictFail { + t.Errorf("gate verdict = %q, want %q — printenv must exit 1 when the "+ + "variable is genuinely unset", ex.Verdict, VerdictFail) + } +} + +// TestCompletionGateBaseIsTheWorktreeCreationCommit is the saga half of the +// acceptance: every completion gate of a `--worktree`-recorded step is +// dispatched with Base naming the commit the worktree was created from — the +// fork point, the SAME base the recorded issue.diff compares against — never +// the run's pinned commit (which predates inherited sibling work, DKT-42's +// over-attribution) and never the worktree's own HEAD (an empty range). +func TestCompletionGateBaseIsTheWorktreeCreationCommit(t *testing.T) { + shared := gitRepo(t) + pinned := strings.TrimSpace(gitOutput(t, shared, "rev-parse", "HEAD")) + + conn := mustDB(t) + registerFixture(t, conn) + issue := createIssue(t, conn, "gate base", "body", "task", nil) + run, err := db.InsertRunWithContext(conn, 1, "wt gate base", 0, nowMS, + db.RunContext{ExecRoot: shared, CommitSHA: pinned}) + testsupport.Must(t, err, "InsertRunWithContext: %v", err) + testsupport.Must(t, db.AddRunIssue(conn, run.ID, issue), "AddRunIssue: %v", err) + _, err = activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // A sibling issue lands on the shared checkout AFTER run start... + writeFile(t, shared, "sibling/left.txt", "a sibling's inherited work\n") + runGit(t, shared, "add", "-A") + runGit(t, shared, "commit", "-qm", "sibling issue's commit") + fork := strings.TrimSpace(gitOutput(t, shared, "rev-parse", "HEAD")) + + // ...and THEN the worktree is created from it, and the step commits its + // own work — the exact shape `step record --worktree` leaves behind. + worktree := filepath.Join(t.TempDir(), "wt") + runGit(t, shared, "worktree", "add", "-q", worktree) + writeFile(t, worktree, "mine/change.txt", "the step's own work\n") + runGit(t, worktree, "add", "-A") + runGit(t, worktree, "commit", "-qm", "the step's commit") + workHead := strings.TrimSpace(gitOutput(t, worktree, "rev-parse", "HEAD")) + + e := testEngine() + captured := &baseCapture{seen: map[string]StepContext{}} + e.Gates = captured + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("summary"), + WorkDir: worktree, NowMS: nowMS, + }) + testsupport.Must(t, err, "complete implement: %v", err) + + if len(captured.seen) == 0 { + t.Fatal("no completion gate was dispatched") + } + for gate, sc := range captured.seen { + if sc.Base != fork { + t.Errorf("gate %s dispatched with Base %q, want the worktree's "+ + "creation commit %q (pinned run commit %q, worktree HEAD %q — "+ + "neither is the step's base)", gate, sc.Base, fork, pinned, workHead) + } + if sc.WorkRoot != worktree { + t.Errorf("gate %s dispatched with WorkRoot %q, want %q", + gate, sc.WorkRoot, worktree) + } + } +} + +// TestCompletionGateBaseUnsetForNonWorktreeStep pins the other acceptance +// direction, and the CHOICE it encodes: a step recorded without `--worktree` +// dispatches every gate with Base == "" — the variable unset — rather than +// some live HEAD read docket cannot vouch for as a range endpoint. Unset is +// the documented pick; absence, not an invented value, is what lets a +// range-shaped gate fail closed instead of scanning the wrong range. +func TestCompletionGateBaseUnsetForNonWorktreeStep(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + e := testEngine() + captured := &baseCapture{seen: map[string]StepContext{}} + e.Gates = captured + + claimAndComplete(t, conn, e, "implement@0", "the change summary", "") + + if len(captured.seen) == 0 { + t.Fatal("no completion gate was dispatched") + } + for gate, sc := range captured.seen { + if sc.Base != "" { + t.Errorf("gate %s dispatched with Base %q for a non-worktree step, "+ + "want \"\" (the variable unset)", gate, sc.Base) + } + } +} + +// TestGateBaseSHADegenerateTreesAnswerUnset pins the resolver's fail-closed +// edges directly: no worktree, a "worktree" that IS the run's exec root, and +// a tree whose fork point cannot be resolved all answer "" — the variable +// stays unset — never a guessed sha and never runDiffBase's pinned fallback, +// which is not the commit any worktree was created from. +func TestGateBaseSHADegenerateTreesAnswerUnset(t *testing.T) { + shared := gitRepo(t) + conn := mustDB(t) + run, err := db.InsertRunWithContext(conn, 1, "degenerate", 0, nowMS, + db.RunContext{ExecRoot: shared, CommitSHA: "feedfacefeedface"}) + testsupport.Must(t, err, "InsertRunWithContext: %v", err) + + if got := gateBaseSHA(conn, run.ID, ""); got != "" { + t.Errorf("gateBaseSHA(no worktree) = %q, want \"\"", got) + } + if got := gateBaseSHA(conn, run.ID, shared); got != "" { + t.Errorf("gateBaseSHA(worktree == exec root) = %q, want \"\" — the "+ + "shared checkout has no fork point and no step-owned range", got) + } + notARepo := t.TempDir() + if got := gateBaseSHA(conn, run.ID, notARepo); got != "" { + t.Errorf("gateBaseSHA(unresolvable fork) = %q, want \"\" — an "+ + "unresolvable base exports nothing, never the pinned run commit", got) + } +} diff --git a/internal/engine/event.go b/internal/engine/event.go index 6b3af02c..35c9baf5 100644 --- a/internal/engine/event.go +++ b/internal/engine/event.go @@ -246,6 +246,47 @@ const ( // it was admitted over), which are exactly the three facts an auditor // asking "why was this allowed?" needs. EventSpawnAdmitted = "spawn-admitted" + + // The batch gate-override kinds (DKT-546). + // + // Both earn their place on §9 item 2's argument in the `spawn-admitted` + // form: each records a park being STEPPED PAST on standing authority. + // `gate-override-granted` is the authority being minted — one operator + // ruling that a gate's failure signature is environmental for the rest of + // the run — and it carries `gate#grantid` so the grant row the feed names + // is findable. `step-batch-overridden` is the authority being SPENT: a + // step whose failed gates were auto-passed under that ruling, carrying the + // covering grant id(s). Without the second kind an auto-applied override + // would be indistinguishable in the feed from an engine-computed pass — + // the one case where "no event" and "the gates passed" say the same thing + // while meaning opposite things. + EventGateOverrideGranted = "gate-override-granted" + EventStepBatchOverridden = "step-batch-overridden" + + // The stale-target waiver being minted (DKT-742) — the same §9 item 2 + // argument as `gate-override-granted`: a warning that stops appearing is + // otherwise indistinguishable in the record from a warning that stopped + // being true, and this kind is what says an operator ruled rather than + // git relented. It carries `targetsha#waiverid` so the waiver row the + // feed names is findable. The waiver being SPENT records no event, + // deliberately: the advisory is recomputed by `dispatch verify`, which + // writes nothing by contract, and an advisory suppressed is not a + // transition — no step changes status because a warning stayed quiet. + EventStaleTargetWaived = "stale-target-waived" + + // The activation snapshot's scope being REFRESHED mid-run (DKT-869). + // + // It earns its place on the `run-repinned` argument, one column over: this + // is the only other transition that moves a frozen premise while a run is + // live. Every packet an issue's remaining steps render, and every diff they + // record, read `run_issues.issue_snapshot`; without this kind a reader of + // the ledger comparing two steps of one run would see two different + // declared scopes with nothing between them explaining the difference — + // which is exactly the drift the freeze exists to prevent, reintroduced + // silently. It carries `from`, `to`, the step instances the refresh + // reaches, and the operator's reason, so the discontinuity is dated and + // attributable rather than inferred. + EventIssueScopeRefreshed = "issue-scope-refreshed" ) // eventKinds is the closed set, as a set. The writer checks membership here, so @@ -274,10 +315,13 @@ var eventKinds = map[string]bool{ EventDispatchOpened: true, EventDispatchClosed: true, EventDispatchAbandoned: true, EventReapAcknowledged: true, EventEventsPruned: true, EventRunBudgetSet: true, - EventStepAnnotated: true, - EventProjectRegistered: true, - EventRunRepinned: true, - EventSpawnAdmitted: true, + EventStepAnnotated: true, + EventProjectRegistered: true, + EventRunRepinned: true, + EventSpawnAdmitted: true, + EventGateOverrideGranted: true, EventStepBatchOverridden: true, + EventStaleTargetWaived: true, + EventIssueScopeRefreshed: true, } // recordEvent writes one event in the caller's transaction. @@ -576,6 +620,12 @@ type TrustGrant struct { // entry being added, without saying it points at `/usr/bin/true`, records // the name of an assurance rather than the assurance. Stub bool + // StubReason is the operator's recorded why-and-what-tracks-it for a stub + // (DKT-607). It rides in the event so the grant's trail carries the + // documented decision, not only the fact of hollowness; empty when the + // grant recorded none (every pre-DKT-607 stub) and always empty on a + // non-stub grant. + StubReason string // Actor is WHO ran the verb — the git identity, falling back to the OS // username (DKT-263). It is A CLAIM, NOT A VERIFIED FACT: `git config // user.name` is whatever the invoking environment says it is, and nothing @@ -583,6 +633,14 @@ type TrustGrant struct { // and it is worth having anyway — before this field, recovering by-whom // meant bracketing runs against wall-clock, which the retro had to do // twice. + // + // IT IS REQUIRED (DKT-595): RecordTrustEvent refuses an empty one. The + // literal "unknown" — the resolver's last fallback when neither a git + // identity nor an OS user exists — DOES count as supplied: it is the + // resolver's honest report of an anonymous environment, the same value + // every other authored row carries there. What is refused is the empty + // string, which no resolver produces and which can only mean the writer + // never filled the field. Actor string // Cwd is where the verb ran from. // @@ -592,6 +650,10 @@ type TrustGrant struct { // separates them. Same reasoning as ProjectRegistration's Cwd/Identity // pair — the case worth attributing is the one where the person is not the // distinguishing fact. + // + // IT IS REQUIRED (DKT-595), same as Actor: RecordTrustEvent refuses an + // empty one, and a caller whose working directory cannot be read must + // refuse the whole trust change rather than degrade the field. Cwd string } @@ -617,7 +679,26 @@ type TrustGrant struct { // is a classification; these two are the person's own account of themselves, // which is an identity. A grant is the one act in the system that widens what // may execute, so its trail should not require a join against a clock. +// +// BOTH ARE MANDATORY, AND THE REFUSAL LIVES HERE (DKT-595). The 2026-08-19 +// batch — 43 unattributed trust events, twelve of which widened network +// egress — came from a writer that supplied neither field, and the ledger +// could not say whether that writer was an older binary or a non-interactive +// path. That ambiguity is unrepairable after the fact, so it is closed at the +// door: this function, the single entry point through which a trust event +// reaches the table, refuses a grant whose Actor or Cwd is empty, BEFORE +// anything is written. Guarding here rather than only at the CLI call site is +// what makes the property hold for a future non-CLI writer too. func RecordTrustEvent(conn *sql.DB, kind string, grant TrustGrant, atMS int64) error { + // THE UNATTRIBUTED-GRANT REFUSAL (DKT-595), ahead of the marshal so a + // refused event leaves no differently-shaped payload behind — the write + // below still emits every key unconditionally or does not happen at all. + if grant.Actor == "" || grant.Cwd == "" { + return fmt.Errorf( + "refusing to record an unattributed %s event (actor=%q cwd=%q): a trust change must say who made it and from where, and a writer that cannot supply both must not write one", + kind, grant.Actor, grant.Cwd) + } + // EVERY KEY IS WRITTEN, including the false ones and the empty ones. A // payload that omitted `tree` when false and `network` when empty would be // byte-identical to one from a writer that never had the field, so a reader @@ -640,11 +721,15 @@ func RecordTrustEvent(conn *sql.DB, kind string, grant TrustGrant, atMS int64) e "network": network, "timeout": grant.Timeout, "stub": grant.Stub, - // Written unconditionally, empty string and all, for the same reason - // the false flags above are: an omitted key is byte-identical to one - // from a writer that never had the field, so a reader could not tell an - // UNRESOLVABLE actor from a build that did not record actors. The first - // is a fact about that grant; the second is a fact about the binary. + "stub_reason": grant.StubReason, + // Written unconditionally like every other key — the key set stays + // exact in both directions — and, since the DKT-595 guard above, + // guaranteed NON-EMPTY. The empty string was once accepted here as + // "the field was unresolvable", but the 2026-08-19 batch showed what + // that buys in practice: an empty value cannot be told apart from a + // writer that never tried, so it distinguished nothing. Emptiness now + // refuses before the write; a key that is present is a value that + // attributes. "actor": grant.Actor, "cwd": grant.Cwd, }) diff --git a/internal/engine/events_project_test.go b/internal/engine/events_project_test.go index 54d32092..abbd6911 100644 --- a/internal/engine/events_project_test.go +++ b/internal/engine/events_project_test.go @@ -57,6 +57,76 @@ func TestEventsScopeToTheProject(t *testing.T) { } } +// TestRunFilterAnswersAcrossProjects is DKT-583. +// +// `events list --run RUN-N` from a neighbouring project's cwd returned +// `{"ok":true,"events":null,"total":0}` for runs with hundreds of events: the +// invoking project's scope was anded onto the run filter, and a run owned by +// another project matched neither arm. Four analysts read that empty page as +// "this run recorded nothing". +// +// A SUCCESSFUL EMPTY FEED IS THE FAILURE. The run clause is already a project +// scope — narrower than any project's — so the second one can only subtract +// rows it has no business subtracting. +func TestRunFilterAnswersAcrossProjects(t *testing.T) { + conn := mustDB(t) + + exec := func(q string, args ...any) { + t.Helper() + if _, err := conn.Exec(q, args...); err != nil { + t.Fatalf("exec %s: %v", q, err) + } + } + + exec(`INSERT INTO projects (id, identity, name) VALUES (2, '/repo/two', 'two')`) + exec(`INSERT INTO runs (id, project_id, request, status, budget, created_at_ms, updated_at_ms, row_version) + VALUES (10, 1, 'r', 'active', 0, 0, 0, 1)`) + exec(`INSERT INTO runs (id, project_id, request, status, budget, created_at_ms, updated_at_ms, row_version) + VALUES (20, 2, 'r', 'active', 0, 0, 0, 1)`) + exec(`INSERT INTO events (at_ms, kind, run_id, data) VALUES (1, 'run-activated', 10, '{}')`) + exec(`INSERT INTO events (at_ms, kind, run_id, data) VALUES (2, 'run-activated', 20, '{}')`) + exec(`INSERT INTO events (at_ms, kind, run_id, data) VALUES (3, 'step-claimed', 20, '{}')`) + + // The verbatim bug: project 1's cwd, project 2's run. + foreign, err := ListEvents(conn, EventQuery{RunID: 20, ProjectID: 1}) + testsupport.Must(t, err, "ListEvents(run 20 from project 1): %v", err) + if len(foreign.Events) != 2 || foreign.Total != 2 { + t.Fatalf("a run in another project answered %d events (total %d), want its 2 — "+ + "a successful empty page is DKT-583's exact defect\n%+v", + len(foreign.Events), foreign.Total, foreign.Events) + } + for _, e := range foreign.Events { + if e.Run != "RUN-20" { + t.Errorf("the run filter leaked a neighbour's event: %+v", e) + } + } + + // A run INSIDE the invoking project is unchanged. + local, err := ListEvents(conn, EventQuery{RunID: 10, ProjectID: 1}) + testsupport.Must(t, err, "ListEvents(run 10 from project 1): %v", err) + if len(local.Events) != 1 || local.Total != 1 { + t.Errorf("the invoking project's own run answered %d events (total %d), want 1", + len(local.Events), local.Total) + } + + // The unscoped (--all-projects) form still answers the same run. + all, err := ListEvents(conn, EventQuery{RunID: 20}) + testsupport.Must(t, err, "ListEvents(run 20, all projects): %v", err) + if len(all.Events) != 2 { + t.Errorf("--all-projects --run answered %d events, want 2", len(all.Events)) + } + + // And a project-scoped feed with NO run filter is still scoped: dropping the + // predicate under --run must not drop it everywhere. + scoped, err := ListEvents(conn, EventQuery{ProjectID: 1}) + testsupport.Must(t, err, "ListEvents(project 1): %v", err) + if len(scoped.Events) != 1 { + t.Errorf("the run-less project feed answered %d events, want only project 1's 1 — "+ + "the scope predicate must survive for every query that names no run\n%+v", + len(scoped.Events), scoped.Events) + } +} + // TestScopedFeedSurvivesACorruptPayload guards DKT-68's filter against the one // way it could take the whole verb down. // diff --git a/internal/engine/events_read.go b/internal/engine/events_read.go index cdc8b798..99efe980 100644 --- a/internal/engine/events_read.go +++ b/internal/engine/events_read.go @@ -110,6 +110,17 @@ type EventQuery struct { // view, because a scoped feed that hid it would be an audit trail with a // blind spot; one that names another repository belongs to that // repository's trail, not to this one's. See eventFilter. + // + // IT IS IGNORED WHEN RunID IS SET (DKT-583). A run belongs to exactly ONE + // project, so `--run` is ALREADY a project scope — a narrower one. Anding + // the invoking project's scope onto it cannot select a meaningful subset: + // it either changes nothing (the run is this project's) or empties the page + // entirely (the run is a neighbor's), and the second case answered + // `{"ok":true,"events":null,"total":0}` for a run with hundreds of events. + // A successful empty feed is the WORST available answer — a consumer + // polling it waits forever on a run it can see in `run report`, which + // resolves a run's project independent of the cwd. `--run` now does the + // same, which is the disposition DKT-583 prefers. ProjectID int } @@ -251,7 +262,13 @@ func eventFilter(q EventQuery) (string, []any) { where += ` AND e.run_id = ?` args = append(args, q.RunID) } - if q.ProjectID != 0 { + // THE PROJECT SCOPE IS DROPPED UNDER `--run` (DKT-583). A run has exactly + // one project, so the run clause above is already that project's scope and + // anding a second one on top can only turn a neighbouring project's feed + // into an empty page that reports success. `run report`, `step show`, and + // `step artifact` all answer for a run outside the invoking project; the + // event feed now agrees with them instead of contradicting them silently. + if q.ProjectID != 0 && q.RunID == 0 { // Three-way attribution, per the field comment: the run's project, // else the issue's, else store-level. The cursor contract survives // scoping — seq stays monotonic, and a filtered-out neighbor is an @@ -449,6 +466,14 @@ var eventActors = map[string]Actor{ EventStepSuperseded: ActorThreshold, EventStepSkipped: ActorThreshold, EventStepHeld: ActorThreshold, + // The batch override being SPENT (DKT-546) is `threshold` on + // `dispatch-abandoned`'s argument: the kind is attributed to the automatic + // path, because that is the one an auditor cannot otherwise see. The + // operator's decision is already in the feed as its own `human` kind + // (`gate-override-granted`, below); this one records the routing stage + // COMPUTING that a standing grant covers this step's failure, and + // `data` carries the grant id(s) so the audit walks back to the person. + EventStepBatchOverridden: ActorThreshold, // Operator verbs, and the harness relaying them. EventRunStarted: ActorHuman, @@ -499,6 +524,28 @@ var eventActors = map[string]Actor{ // it. The recorded agreement moves only because a person ran `run repin` // with a reason, which is precisely what `human` means in this table. EventRunRepinned: ActorHuman, + + // The batch override being MINTED (DKT-546): nothing in the engine grants + // itself standing authority over a gate — a grant exists only because a + // person ran `step resolve --as override-pass --batch`, and the event's + // purpose is to anchor every later `step-batch-overridden` to that verb. + EventGateOverrideGranted: ActorHuman, + + // The stale-target waiver being minted (DKT-742): nothing in the engine + // waives its own advisory — a waiver exists only because a person ran + // `dispatch waive-target`, and unlike the batch override there is no + // "spent" counterpart kind, because applying a waiver changes no step's + // state (the advisory is a recomputed read, served by a verb that writes + // nothing). + EventStaleTargetWaived: ActorHuman, + + // The scope refresh (DKT-869): nothing in the engine re-snapshots a bound + // issue — activation writes the blob once and every automatic path, + // re-activation included, leaves it alone. It moves only because a person + // widened the scope through `issue edit --scope` and then ran `run + // refresh-scope` naming the run it should reach, which is precisely what + // `human` means in this table. + EventIssueScopeRefreshed: ActorHuman, } // ActorFor reports which of the four causes an event kind is attributable to, diff --git a/internal/engine/events_read_test.go b/internal/engine/events_read_test.go index 9b84f188..f6b66d6c 100644 --- a/internal/engine/events_read_test.go +++ b/internal/engine/events_read_test.go @@ -64,7 +64,7 @@ func TestEventShapeOmitsNullRunAndStep(t *testing.T) { conn := mustDB(t) err := RecordTrustEvent( conn, EventTrustAdded, - TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo"}, nowMS, + TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo", Actor: "tester", Cwd: "/repo"}, nowMS, ) testsupport.Must(t, err, "RecordTrustEvent: %v", err) @@ -94,6 +94,41 @@ func TestEventShapeOmitsNullRunAndStep(t *testing.T) { } } +// TestUnattributedTrustEventIsRefused is DKT-595, at the one entry point a +// trust event has into the table. +// +// The 2026-08-19 batch was written by a writer that supplied neither actor nor +// cwd, and the ledger could not afterwards say whether that writer was an old +// binary or a non-interactive path. The guard lives in RecordTrustEvent itself — +// not only in the CLI that resolves the fields today — so a future writer that +// cannot supply them is refused rather than trusted to degrade honestly. The +// refusal is BEFORE the write: no event of any shape lands. +func TestUnattributedTrustEventIsRefused(t *testing.T) { + conn := mustDB(t) + + for _, tc := range []struct { + name string + grant TrustGrant + }{ + {"no actor", TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo", Cwd: "/repo"}}, + {"no cwd", TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo", Actor: "tester"}}, + {"neither", TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo"}}, + } { + t.Run(tc.name, func(t *testing.T) { + err := RecordTrustEvent(conn, EventTrustAdded, tc.grant, nowMS) + if err == nil { + t.Fatal("an unattributed trust event was recorded; a writer that cannot say who and from where must be refused") + } + }) + } + + page, err := ListEvents(conn, EventQuery{}) + testsupport.Must(t, err, "ListEvents: %v", err) + if len(page.Events) != 0 { + t.Errorf("a refused trust event left %d event(s) behind; the refusal must precede the write", len(page.Events)) + } +} + // TestCursorIsStrictlyGreater is E5 and E6: `--since SEQ` returns `seq > SEQ`, so // a consumer stores the last seq it saw and passes it back WITHOUT re-reading it. func TestCursorIsStrictlyGreater(t *testing.T) { @@ -260,7 +295,7 @@ func TestRunFilterScopesTheFeed(t *testing.T) { runID, _ := budgetRun(t, conn, 0) err := RecordTrustEvent( conn, EventTrustAdded, - TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo"}, nowMS, + TrustGrant{Name: "checks", ArgvSHA256: "abc123", Repo: "/repo", Actor: "tester", Cwd: "/repo"}, nowMS, ) testsupport.Must(t, err, "RecordTrustEvent: %v", err) diff --git a/internal/engine/events_test.go b/internal/engine/events_test.go index a0a1cbeb..712a5657 100644 --- a/internal/engine/events_test.go +++ b/internal/engine/events_test.go @@ -167,17 +167,47 @@ func TestEventKindsAreAClosedSet(t *testing.T) { // the one transition that rewrites the evidence other records are // checked against while leaving no trace of the value it replaced. "run-repinned", + + // The batch gate-override kinds (DKT-546) — TWO, each the + // spawn-admitted argument again: a park stepped past on standing + // authority. `gate-override-granted` is the authority being minted + // (one operator ruling per failed gate), `step-batch-overridden` is + // it being spent — without the second, an auto-applied override would + // be indistinguishable in the feed from an engine-computed pass. + "gate-override-granted", + "step-batch-overridden", + + // The stale-target waiver kind (DKT-742) — ONE, the granted half of + // the DKT-546 argument alone: an operator minting standing authority + // over a repeating advisory. No "spent" counterpart, because applying + // a waiver changes no step's state — the advisory is recomputed by + // verbs that write nothing. + "stale-target-waived", + + // The scope-refresh kind (DKT-869) — ONE, on the `run-repinned` + // argument in its other column: this is the second and last transition + // that moves a frozen premise while a run is live. Without it, two + // steps of one run rendering two different declared scopes would be + // indistinguishable in the record from the snapshot drift the freeze + // exists to prevent — the discontinuity has to be dated and + // attributable for `run refresh-scope` to be an exception rather than + // a hole. No "spent" counterpart: the refreshed snapshot IS the new + // premise, and every later render reads it the way every render always + // read the frozen one. + "issue-scope-refreshed", } { if !eventKinds[kind] { t.Errorf("the spec names %q but eventKinds does not contain it", kind) } } - if len(eventKinds) != 43 { + if len(eventKinds) != 47 { t.Errorf("eventKinds has %d entries; §7.6 plus gates-trust §6.4/§8.1, "+ "payloads-thresholds §7.7, runs-dispatch §5/§6, events-follow "+ "§6/§7.3, DKT-35's annotation kind, DKT-61's tenancy kind, "+ - "DKT-236's spawn carve-out, DKT-294's live-status mirror, and "+ - "DKT-408's repin kind enumerate 43 (see DKT-21)", + "DKT-236's spawn carve-out, DKT-294's live-status mirror, "+ + "DKT-408's repin kind, DKT-546's batch-override pair, "+ + "DKT-742's stale-target waiver kind, and DKT-869's scope-refresh "+ + "kind enumerate 47 (see DKT-21)", len(eventKinds)) } } diff --git a/internal/engine/forced_reap_budget_test.go b/internal/engine/forced_reap_budget_test.go new file mode 100644 index 00000000..4cce6777 --- /dev/null +++ b/internal/engine/forced_reap_budget_test.go @@ -0,0 +1,201 @@ +package engine + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-585: a reap carrying `data.forced` does not consume the step's attempt +// budget. RUN-30 STEP-755: implement@0's attempt 1 was force-reaped after an +// accidental hotkey interrupt killed the wave — a relay declaring a dead spawn, +// not an executor failure — yet the step showed attempts 2 of max_attempts 2 +// after the successor succeeded, one interrupt from `waiting-human` on a +// healthy charter. The fix is a classification at the forced call site: a +1 +// nudge of `attempt_base` (never a touch of `attempt`, the usage ledger's key), +// so the exhaustion math `attempt - attempt_base >= max_attempts` skips exactly +// the dead attempt. An ordinary TTL expiry still charges as before. + +const forcedBudgetSrc = ` +[pipeline] +name = "forced-budget" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "work" +after = [] +executor = "w" +emits = "out" +max_attempts = 2 +on_fail = "waiting-human" +` + +// TestForcedReapDoesNotConsumeAttemptBudget is the acceptance criterion +// verbatim, in RUN-30's own shape: attempt 1 force-reaped, attempt 2 is a real +// try — and with `max_attempts = 2` the step must NOT be exhausted by it. The +// budget still bounds: the next genuine failure after that does exhaust. +func TestForcedReapDoesNotConsumeAttemptBudget(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(forcedBudgetSrc), "forced-budget.toml") + issue := createIssue(t, conn, "forced budget", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + id := stepIDByInstance(t, conn, "work@0") + + // Attempt 1: the claim spends the count, then the relay establishes the + // spawn is dead and force-reaps it. + _, err = ClaimStep(conn, id, ClaimOptions{Owner: "doomed", NowMS: nowMS}) + testsupport.Must(t, err, "claim 1: %v", err) + err = ForceReapStep(conn, id, "wave killed by an accidental interrupt", nowMS+1) + testsupport.Must(t, err, "ForceReapStep: %v", err) + + // The exemption is a base nudge, nothing else: attempt stands at 1 (the + // ledger's key), the base moved to 1, so the budget reads zero spent. + step, err := db.GetStep(conn, id) + testsupport.Must(t, err, "GetStep after the forced reap: %v", err) + if step.Attempt != 1 { + t.Fatalf("attempt = %d after a forced reap, want 1 — the counter is the "+ + "usage ledger's key and is never reset or decremented", step.Attempt) + } + if step.AttemptBase != 1 { + t.Fatalf("attempt_base = %d after a forced reap, want 1 — the dead "+ + "attempt is exempted from the budget by the nudge", step.AttemptBase) + } + + // Attempt 2 is the step's FIRST legitimate try. If it fails, the step must + // return to the pool, not park: 2 claims spent, but only 1 counts. + claim2, err := ClaimStep(conn, id, ClaimOptions{Owner: "successor", NowMS: nowMS + 2}) + testsupport.Must(t, err, "claim 2: %v", err) + err = e.FailStep(conn, id, claim2.Token, "genuine failure", "", nowMS+2) + testsupport.Must(t, err, "fail 2: %v", err) + if got := stepStatus(t, conn, "work@0"); got != db.StepPending { + t.Fatalf("status = %q after the first legitimate failure, want %q — "+ + "the force-reaped attempt must not count against max_attempts", + got, db.StepPending) + } + + // The budget is exempted, not abolished: the SECOND legitimate failure is + // 2 of 2 and exhausts as declared. + claim3, err := ClaimStep(conn, id, ClaimOptions{Owner: "w3", NowMS: nowMS + 3}) + testsupport.Must(t, err, "claim 3: %v", err) + err = e.FailStep(conn, id, claim3.Token, "failed again", "", nowMS+3) + testsupport.Must(t, err, "fail 3: %v", err) + if got := stepStatus(t, conn, "work@0"); got != db.StepWaitingHuman { + t.Errorf("status = %q after two legitimate failures, want %q — the "+ + "exemption covers only the forced-reaped attempt", got, db.StepWaitingHuman) + } +} + +// TestExpiryReapStillConsumesAttemptBudget pins the other path: an ordinary +// TTL expiry reap — no `data.forced`, no relay assertion — charges the budget +// exactly as before. An expiry may still be an executor wedged under its own +// load, and that judgment is unchanged. +func TestExpiryReapStillConsumesAttemptBudget(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(forcedBudgetSrc), "forced-budget.toml") + issue := createIssue(t, conn, "expiry budget", "body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + e := testEngine() + id := stepIDByInstance(t, conn, "work@0") + + // Attempt 1's lease lapses in silence and `next`'s lazy reap frees it. + claim1, err := ClaimStep(conn, id, ClaimOptions{Owner: "w1", NowMS: nowMS}) + testsupport.Must(t, err, "claim 1: %v", err) + late := claim1.LeaseExpiresMS + 1 + next, err := e.NextSteps(conn, run.ID, 0, late) + testsupport.Must(t, err, "NextSteps reaping: %v", err) + if len(next.Reaped) != 1 { + t.Fatalf("reaped %v, want the expired claim reaped", next.Reaped) + } + + // No exemption: the base stands at 0, the expired attempt counts. + step, err := db.GetStep(conn, id) + testsupport.Must(t, err, "GetStep after the expiry reap: %v", err) + if step.Attempt != 1 || step.AttemptBase != 0 { + t.Fatalf("attempt/attempt_base = %d/%d after an expiry reap, want 1/0 — "+ + "an ordinary expiry still charges the budget", + step.Attempt, step.AttemptBase) + } + + // Attempt 2's failure is therefore 2 of 2: exhausted, parked. + claim2, err := ClaimStep(conn, id, ClaimOptions{Owner: "w2", NowMS: late}) + testsupport.Must(t, err, "claim 2: %v", err) + err = e.FailStep(conn, id, claim2.Token, "gave up", "", late) + testsupport.Must(t, err, "fail 2: %v", err) + if got := stepStatus(t, conn, "work@0"); got != db.StepWaitingHuman { + t.Errorf("status = %q after an expiry reap plus one failure with "+ + "max_attempts = 2, want %q — expiry behavior must be unchanged", + got, db.StepWaitingHuman) + } +} + +// TestForcedReapKeepsTheDeadAttemptsLedgerSlot is the usage-ledger half of +// RUN-30: the dead attempt's measured usage was back-filled AGAINST ITS OWN +// ATTEMPT NUMBER after the forced reap, and the successor's usage recorded +// beside it. The exemption must leave `attempt` alone so both slots exist. +func TestForcedReapKeepsTheDeadAttemptsLedgerSlot(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + id := stepIDByInstance(t, conn, "implement@0") + + // Attempt 1 dies; the relay force-reaps and back-fills what the dead + // spawn's journal recorded — against attempt 1, the step's recorded + // attempt, exactly as `dispatch backfill-usage` promises. + _, err := ClaimStep(conn, id, ClaimOptions{Owner: "doomed", NowMS: nowMS}) + testsupport.Must(t, err, "claim 1: %v", err) + err = ForceReapStep(conn, id, "spawn died; journal has usage", nowMS+1) + testsupport.Must(t, err, "ForceReapStep: %v", err) + _, err = e.BackfillUsage(conn, run.ID, []BackfillRow{ + {Step: id, Unit: "output_tokens", Quantity: 48344}, + }, "wave-journal:dead", "", nowMS+2) + testsupport.Must(t, err, "back-filling the dead attempt: %v", err) + + // Attempt 2 runs for real, records, and back-fills its own usage. + claim2, err := ClaimStep(conn, id, ClaimOptions{Owner: "successor", NowMS: nowMS + 3}) + testsupport.Must(t, err, "claim 2: %v", err) + step, err := db.GetStep(conn, id) + testsupport.Must(t, err, "GetStep after re-claim: %v", err) + if step.Attempt != 2 { + t.Fatalf("attempt = %d after the successor's claim, want 2 — the "+ + "successor needs its own ledger attempt", step.Attempt) + } + err = e.CompleteStep(conn, id, CompleteOptions{ + Token: claim2.Token, + Artifact: []byte("the change summary"), + NowMS: nowMS + 4, + }) + testsupport.Must(t, err, "complete: %v", err) + _, err = e.BackfillUsage(conn, run.ID, []BackfillRow{ + {Step: id, Unit: "output_tokens", Quantity: 512}, + }, "wave-journal:successor", "", nowMS+5) + testsupport.Must(t, err, "back-filling the successor: %v", err) + + // Both executions hold distinct slots: the dead attempt's row under its + // own number, the successor's beside it. + rows, err := conn.Query( + `SELECT attempt, quantity FROM usage_ledger + WHERE step_id = ? AND unit = 'output_tokens' ORDER BY attempt`, id) + testsupport.Must(t, err, "reading the ledger: %v", err) + defer rows.Close() + var got [][2]float64 + for rows.Next() { + var attempt, quantity float64 + testsupport.Must(t, rows.Scan(&attempt, &quantity), "scanning") + got = append(got, [2]float64{attempt, quantity}) + } + testsupport.Must(t, rows.Err(), "reading the ledger") + want := [][2]float64{{1, 48344}, {2, 512}} + if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] { + t.Errorf("ledger rows = %v, want %v — the exemption must never touch "+ + "`attempt`, the ledger's key half", got, want) + } +} diff --git a/internal/engine/gate.go b/internal/engine/gate.go index 7d607575..6672e314 100644 --- a/internal/engine/gate.go +++ b/internal/engine/gate.go @@ -60,6 +60,19 @@ type StepContext struct { // every gate spawned in the shared checkout unconditionally, so a gate's // evidence could describe a HEAD the step under review never touched. WorkRoot string + // Base is the sha of the step's base commit — the commit WorkRoot was + // created from — exported to the child as DOCKET_GATE_BASE (DKT-992) so a + // gate can scan exactly the step's committed range (base..HEAD of the + // tree it runs in). The saga fills it, for worktree-recorded steps only, + // with the same fork-point resolution the diff stage uses (runDiffBase / + // worktreeForkPoint); it stays "" — the var unset — for a shared-checkout + // step and on the pre-claim path, where no committed range belongs to the + // step being gated. Empty means unset, never an invented value: executors + // commit before `step record`, so a gate that falls back to scanning the + // working tree scans nothing, and before this field it had no way to + // learn the range and guessed (`git diff HEAD~1`, wrong for multi-commit + // steps) or measured the clean tree (always empty). + Base string } // GateResult is one gate's outcome, in §11.4's `gate result` shape. diff --git a/internal/engine/gate_exec.go b/internal/engine/gate_exec.go index 43358092..c80112aa 100644 --- a/internal/engine/gate_exec.go +++ b/internal/engine/gate_exec.go @@ -286,6 +286,7 @@ func (r *ExecRunner) spawnMatched( env, err := exec.BuildEnv(exec.EnvPolicy{ Gate: g.Name, Repo: r.RepoRoot, Network: entry.Network, Issue: model.FormatID(sc.IssueID), Scope: sc.Scope, + Base: sc.Base, }) if err != nil { return GateExecution{}, err diff --git a/internal/engine/gate_preflight.go b/internal/engine/gate_preflight.go index 1131dea4..4fc93a70 100644 --- a/internal/engine/gate_preflight.go +++ b/internal/engine/gate_preflight.go @@ -56,6 +56,12 @@ type GatePreflight struct { // check — an operator reading a green preflight should not have to open the // trust store to learn which they have. Stub bool `json:"stub,omitempty"` + // StubReason is the resolving entry's recorded reason for being a stub — + // why the decision was made and which issue tracks replacing it (DKT-607). + // It rides here so the documented decision is discoverable from the + // preflight itself, not only from a tribunal transcript or the trust file. + // Empty when the entry recorded none, which the renderer calls out. + StubReason string `json:"stub_reason,omitempty"` // Reason explains an unmatched gate, verbatim from the matcher, so the // preflight and the mid-run diagnostic say the same thing. Reason string `json:"reason,omitempty"` @@ -128,7 +134,8 @@ func BuildGatePreflight( if m.Matched { row.Matched = true if m.Entry != nil { - row.Entry, row.Stub = m.Entry.Name, m.Entry.Stub + row.Entry, row.Stub, row.StubReason = + m.Entry.Name, m.Entry.Stub, m.Entry.StubReason } } else { row.Reason = m.Reason @@ -191,9 +198,27 @@ func RenderGatePreflight(w interface{ Write([]byte) (int, error) }, rows []GateP fmt.Fprintf(w, "\nnote: %d declared gate(s) resolve to a STUB entry and will measure nothing:\n", len(stubs)) + // A stub WITH a recorded reason is a documented decision, printed where + // the stub itself is printed (DKT-607) — the decision must be + // discoverable here, not only in a tribunal transcript. One WITHOUT a + // reason gets the remedy line below, because a stub nobody can explain + // from this output is indistinguishable from a placeholder somebody + // forgot. + unexplained := false for _, r := range stubs { fmt.Fprintf(w, " %s (entry %s)\n", exec.Render(r.Gate), exec.Render(r.Entry)) + if r.StubReason != "" { + fmt.Fprintf(w, " stub reason: %s\n", exec.Render(r.StubReason)) + } else { + unexplained = true + } + } + if unexplained { + fmt.Fprintf(w, + " A stub without a recorded reason: say why and what tracks fixing it with "+ + "`docket trust rm ` then `docket trust add -- --stub "+ + "--stub-reason \"\"`.\n") } } } diff --git a/internal/engine/gate_preflight_test.go b/internal/engine/gate_preflight_test.go index 0753df71..80cce3d3 100644 --- a/internal/engine/gate_preflight_test.go +++ b/internal/engine/gate_preflight_test.go @@ -148,9 +148,10 @@ func TestGatePreflightDeduplicatesAcrossDeclarationSites(t *testing.T) { func TestGatePreflightReportsAStubbedEntry(t *testing.T) { repoRoot := t.TempDir() argv := []string{"/usr/bin/true"} + reason := "no scanner selected yet; removal tracked by DKT-607" load := sandboxTrust(t, trust.Entry{ Name: "secret-scan", Argv: argv, ArgvSHA256: trust.ArgvSHA256(argv), - Global: true, Stub: true, + Global: true, Stub: true, StubReason: reason, }) rows, err := BuildGatePreflight( @@ -166,6 +167,48 @@ func TestGatePreflightReportsAStubbedEntry(t *testing.T) { t.Error("a stubbed entry is not reported as a stub; the preflight would " + "read green for a gate that measures nothing") } + // DKT-607: the recorded decision rides the row, so an operator (or a + // tribunal seat) reading the preflight sees why the stub exists and which + // issue tracks removing it, without opening the trust store or a transcript. + if got.StubReason != reason { + t.Errorf("StubReason = %q, want %q", got.StubReason, reason) + } +} + +// TestRenderGatePreflightShowsTheStubReason (DKT-607): a stub WITH a recorded +// reason prints it — the documented decision is discoverable from the +// preflight itself — and a stub WITHOUT one gets the remedy line, because an +// unexplained stub is indistinguishable from a placeholder somebody forgot. +func TestRenderGatePreflightShowsTheStubReason(t *testing.T) { + t.Run("a recorded reason is printed", func(t *testing.T) { + var buf strings.Builder + RenderGatePreflight(&buf, []GatePreflight{ + {Gate: "secret-scan", Matched: true, Entry: "secret-scan", Stub: true, + StubReason: "no scanner selected yet; removal tracked by DKT-607"}, + }) + out := buf.String() + for _, needle := range []string{"STUB", "secret-scan", "DKT-607"} { + if !strings.Contains(out, needle) { + t.Errorf("the stub note does not mention %q:\n%s", needle, out) + } + } + if strings.Contains(out, "--stub-reason") { + t.Errorf("a fully-explained stub still prints the record-a-reason remedy:\n%s", out) + } + }) + + t.Run("an unexplained stub names the remedy", func(t *testing.T) { + var buf strings.Builder + RenderGatePreflight(&buf, []GatePreflight{ + {Gate: "sdet-abuse", Matched: true, Entry: "sdet-abuse", Stub: true}, + }) + out := buf.String() + for _, needle := range []string{"sdet-abuse", "--stub-reason"} { + if !strings.Contains(out, needle) { + t.Errorf("the unexplained-stub note does not mention %q:\n%s", needle, out) + } + } + }) } // TestRenderGatePreflightIsSilentWhenEveryGateResolves. diff --git a/internal/engine/held.go b/internal/engine/held.go index 2475589d..ac920e0e 100644 --- a/internal/engine/held.go +++ b/internal/engine/held.go @@ -45,6 +45,11 @@ import ( type holdTally struct { rule string voters []string + // cost is `vote.hold.cost` (DKT-584): the declared expected_cost a hold + // minted as `vote` carries, so the panel the engine convenes is visible to + // the budget floor the way a declared vote step's cost is. 0 — the default + // — mints exactly the prior row. + cost float64 } // configured reports whether held steps are minted as vote steps. @@ -66,7 +71,14 @@ func loadHoldTally(conn *sql.DB, runID int) (holdTally, error) { if err != nil { return holdTally{}, err } - return holdTally{rule: rule.Value, voters: db.SplitNameList(voters.Value)}, nil + cost, err := db.GetConfig(conn, projectID, db.KeyVoteHoldCost) + if err != nil { + return holdTally{}, err + } + return holdTally{ + rule: rule.Value, voters: db.SplitNameList(voters.Value), + cost: parseHoldCost(cost.Value), + }, nil } // loadHoldTallyTx is loadHoldTally inside a caller's transaction, for the @@ -81,7 +93,39 @@ func loadHoldTallyTx(tx *sql.Tx, projectID int) (holdTally, error) { if err != nil { return holdTally{}, err } - return holdTally{rule: rule.Value, voters: db.SplitNameList(voters.Value)}, nil + cost, err := db.GetConfigTx(tx, projectID, db.KeyVoteHoldCost) + if err != nil { + return holdTally{}, err + } + return holdTally{ + rule: rule.Value, voters: db.SplitNameList(voters.Value), + cost: parseHoldCost(cost.Value), + }, nil +} + +// parseHoldCost reads `vote.hold.cost`'s stored value. `config set` validated +// it as a non-negative number on the way in, so a malformed value can only be +// a hand-edited store — tolerated as 0 rather than failing the saga that is +// materializing a hold, the same tolerance configuredBudgetDefault keeps for +// `budget.default`. +func parseHoldCost(value string) float64 { + var cost float64 + if _, err := fmt.Sscanf(value, "%g", &cost); err != nil || cost < 0 { + return 0 + } + return cost +} + +// heldStepCost is the expected_cost a materialized held step is minted with: +// the configured `vote.hold.cost` when the hold convenes a PANEL (kind +// `vote`), and 0 when one operator decides (kind `human`) — an operator's +// decision is not a panel's spend, and charging the floor for it would make +// the configured number mean two different things. +func (t holdTally) heldStepCost() float64 { + if t.configured() { + return t.cost + } + return 0 } // heldStepKind is the kind a materialized held step is MINTED as. @@ -208,6 +252,10 @@ func materializeHeldCluster( // approve/reject still apply once a failed tally parks it. Kind: tally.heldStepKind(), Status: db.StepPending, + // DKT-584: a hold minted as a VOTE step carries the configured + // `vote.hold.cost` so the panel is visible to the budget floor at + // materialization; a `human` hold stays at 0, the prior row exactly. + ExpectedCost: tally.heldStepCost(), // H4: the flag that tells a reader a declared question from a // computed one. Materialized: true, @@ -725,7 +773,11 @@ func resolveHeldPayload( stillHeld++ } } - body = aggregateBody(routingStep.Instance, len(elements), stillHeld) + // The recorded count is 0 on purpose: this body is regenerated from the + // RESOLVED PAYLOAD, and `route_at`'s below-floor clusters were never in + // it — they live in the aggregate's own `action_results` row, which this + // supersession does not touch. + body = aggregateBody(routingStep.Instance, len(elements), stillHeld, 0) if operatorResolved > 0 { body += fmt.Sprintf(", %d operator-resolved", operatorResolved) } diff --git a/internal/engine/held_test.go b/internal/engine/held_test.go index 08b4ee0f..24e24637 100644 --- a/internal/engine/held_test.go +++ b/internal/engine/held_test.go @@ -718,7 +718,8 @@ func TestUnresolvedHoldIsSweptByALoopEntry(t *testing.T) { testsupport.Must(t, err, "StepDefinitionsTx: %v", err) routing, err := db.GetStepTx(tx, routingID) testsupport.Must(t, err, "GetStepTx: %v", err) - swept, err := supersedeSweep(tx, routing, defs[routing.WorkflowID], 1, nowMS) + swept, err := supersedeSweep(tx, routing, + afterLoopDownstreamFor(defs[routing.WorkflowID], routing.StepName), 1, nowMS) testsupport.Must(t, err, "supersedeSweep: %v", err) if !contains(swept, "reconcile-held@0#0") { t.Errorf("the sweep did not reach the materialized step: %v.\n\n"+ diff --git a/internal/engine/human.go b/internal/engine/human.go index 64f6624f..ef2727bd 100644 --- a/internal/engine/human.go +++ b/internal/engine/human.go @@ -284,6 +284,38 @@ func (e *Engine) DecideStepValue( // §6.10's `waiting-human` resolutions. func (e *Engine) ResolveStep( conn *sql.DB, stepID int, as, note string, nowMS int64, +) error { + return e.resolveStep(conn, stepID, as, note, false, false, nowMS) +} + +// ResolveStepBatch is `step resolve --as override-pass --batch` (DKT-546): the +// resolution plus one run-scoped grant per failed completion gate, so later +// steps in the SAME run failing the same gate with the same failure signature +// (gate name + exit + reason) auto-pass at routing instead of re-asking the +// operator. The grant dies with the run — a new run re-asks. +func (e *Engine) ResolveStepBatch( + conn *sql.DB, stepID int, as, note string, nowMS int64, +) error { + return e.resolveStep(conn, stepID, as, note, true, false, nowMS) +} + +// ResolveStepDropInterposed is `step resolve --as override-pass +// --drop-interposed [--batch]` (DKT-861): the same resolution, under the +// operator's EXPLICIT acknowledgment that the generic pass skips the step(s) +// the threshold interposes. Without the acknowledgment, resolveStep refuses +// override-pass on such a step BEFORE anything commits — the DKT-470 warning +// used to arrive beside a mutation already decided, which an operator promised +// the interposed gate would still run could only regret, not act on (RUN-61's +// verify-tribunal, skipped under the operator who had chosen override-pass +// precisely to reach it). +func (e *Engine) ResolveStepDropInterposed( + conn *sql.DB, stepID int, as, note string, batch bool, nowMS int64, +) error { + return e.resolveStep(conn, stepID, as, note, batch, true, nowMS) +} + +func (e *Engine) resolveStep( + conn *sql.DB, stepID int, as, note string, batch, dropInterposed bool, nowMS int64, ) error { step, err := db.GetStep(conn, stepID) if errors.Is(err, db.ErrStepNotFound) { @@ -297,6 +329,28 @@ func (e *Engine) ResolveStep( return validationErr("--as must be one of %v, got %q", resolveValues, as) } + // --batch WIDENS what one authorization covers — one grant auto-passing N + // future failures is a trust-boundary change — so it rides only the verb + // whose ruling it extends. A batch `skip` or `abandon-issue` has no + // coherent meaning: those decide THIS step, not the failure's signature. + if batch && as != ResolveOverridePass { + return validationErr( + "--batch extends an override-pass ruling to later identical gate "+ + "failures in this run, so it requires --as %s, got %q", + ResolveOverridePass, as) + } + + // --drop-interposed WAIVES a refusal only override-pass can trigger + // (DKT-861): on any other resolution it acknowledges a consequence that + // cannot occur, so it is refused the way --batch is rather than accepted + // as though it had covered something. + if dropInterposed && as != ResolveOverridePass { + return validationErr( + "--drop-interposed acknowledges that an override-pass skips the "+ + "step(s) its threshold interposes, so it requires --as %s, got %q", + ResolveOverridePass, as) + } + // R11: `resolve` on a step that is not `waiting-human` is // VALIDATION_ERROR — with ONE exception, stated rather than special-cased: // a `vote` step waits on ITS VOTERS, and nobody but them can advance it. @@ -350,6 +404,53 @@ func (e *Engine) ResolveStep( step.Instance, step.Instance, step.Instance) } + // The SAME refusal, one shape over — and the one the guard above MISSED + // (DKT-726). That guard is scoped to `step.Materialized`: an engine-minted + // `reconcile-held@N#M` cluster whose tally failed. A plain + // workflow-declared `type="vote"` step — `security-vote@8`, not minted by + // anything — carries its OWN tribunal proposal, and when that proposal was + // already tallied the guard did not see it. Retry fell through to the + // generic path below, reset the attempt budget and the lease, returned the + // step to `pending`, and the next `next` re-read the SAME proposal: the + // idempotency key is (run, issue, instance), so no second ballot is opened + // and no cast changes. `routeVoteStep` re-announced the identical verdict + // and routed to the identical place. Observed on RUN-51 STEP-2433, + // security-vote@8 / DKT-V256 rejected 3/3, which cost a full + // run-pause/run-resume cycle to land exactly where it started. + // + // The condition is the PROPOSAL being decided, not the step being parked: + // an APPROVED tally is just as sticky as a rejected one, and retrying over + // it re-reads the same pass. Only an `open` proposal — nothing decided yet, + // nothing to re-read — leaves retry meaning something, and R11's exception + // exists precisely so a resolution stays offered there. + // + // The remedy list differs from the held cluster's because the question is + // not one an operator can simply answer: a workflow vote step's verdict is + // the panel's. `fix-round` is the verb that was actually wanted on RUN-51 — + // it authorizes another round of WORK on the reported problem and mints a + // fresh vote at a new ordinal, which opens a NEW proposal because the + // instance changed. + if as == ResolveRetry && step.Kind == workflow.TypeVote { + outcome, err := ReadStepVoteOutcome(conn, step) + if err != nil { + return err + } + if outcome != nil && outcome.Verdict != "" { + return validationErr( + "step %s cannot be retried: its proposal %s is already %s, and "+ + "retry resets the retry budget, which is not what is "+ + "blocking it — the decision is sticky, so the same tally "+ + "would be read again and the step would route to the same "+ + "place. Use --as %s to authorize another round of work on "+ + "the problem (a fresh vote, on a new proposal), --as %s to "+ + "accept the step as passing, --as %s to route it skipped, "+ + "or --as %s to drop the issue from this run", + step.Instance, model.FormatProposalID(outcome.ProposalID), + outcome.Status, ResolveFixRound, ResolveOverridePass, + ResolveSkip, ResolveAbandonIssue) + } + } + if as == ResolveRetry { rejected, heldInstance, err := parkedByRejectedHold(conn, step) if err != nil { @@ -401,6 +502,28 @@ func (e *Engine) ResolveStep( } spec := workflow.StepByName(defs[step.WorkflowID], step.StepName) + // DKT-861: the DKT-470 warning arrived beside a mutation already decided — + // an operator promised the interposed gate would still run had no move + // left but regret (RUN-61's verify-tribunal went `skipped` and the run + // rolled to `done` under the operator who chose override-pass precisely to + // reach that gate). The consequence is now a REFUSAL ahead of the + // transaction: override-pass on a step whose threshold interposes other + // step(s) proceeds only under the explicit --drop-interposed + // acknowledgment, and nothing commits until the operator has read the + // exact sentence the warning used to print after the fact. The detection + // is the same spec + ThresholdTargets read skipUnroutedTargets makes when + // this resolution commits, so the refusal and the skip cannot disagree. A + // step with no interposed targets resolves exactly as before, no flag + // required. + if as == ResolveOverridePass && !dropInterposed { + if warning := overridePassInterposedWarning(step.Instance, spec); warning != "" { + return validationErr( + "%s. Refusing without --drop-interposed, the explicit "+ + "acknowledgment that skipping them is intended (DKT-861)", + warning) + } + } + // `rerun-gates` needs gates to re-run (DKT-259). A step that declares none // would rewind to `recorded`, find nothing to measure, and route again on // the same evidence — an expensive no-op that looks like it did something. @@ -416,6 +539,27 @@ func (e *Engine) ResolveStep( } } + // The grant's source rows, read BEFORE the transaction for the same + // pooled-connection reason as routingStepOf above. A park with no failed + // completion gate — a rejected hold, a quorum that never arrived, a + // gap-only completion — has no signature to grant from, and refusing is + // honest where recording a grant that can never match would look like it + // did something. + var grantRows []db.GateResultRow + if batch { + rows, err := db.GateResultsForStep(conn, step.ID) + if err != nil { + return err + } + grantRows = failingCompletionRows(rows) + if len(grantRows) == 0 { + return validationErr( + "step %s has no failed completion gate to grant from; --batch "+ + "records the parked step's failing gate signature(s), and "+ + "this park was not caused by one", step.Instance) + } + } + tx, err := conn.Begin() if err != nil { return fmt.Errorf("beginning the resolution: %w", err) @@ -474,7 +618,11 @@ func (e *Engine) ResolveStep( // AUTHORIZED (DKT-340). The operator has read whatever park stands // and asked for the round; the non-convergence refusal must not fire // against the very verb that park names as its way out. - outcome, err := EnterLoopAuthorized(tx, step, defs[step.WorkflowID], nowMS) + // The note rides into the round it authorizes (DKT-725): stamped onto + // the new instances' routing records, it renders in their packets as + // `== RESOLUTION` — the only channel that reaches a NEW round, since + // comments never render and `body_snapshot` froze at activation. + outcome, err := EnterLoopAuthorized(tx, step, defs[step.WorkflowID], note, nowMS) if err != nil { return err } @@ -527,6 +675,30 @@ func (e *Engine) ResolveStep( }); err != nil { return err } + // The grants and the resolution are ONE transaction (the DKT-237 loop-grant + // discipline): a ruling recorded without its resolution would cover + // failures the operator never overrode, and a resolution without its + // grants would silently drop the ruling's reach. One grant per failed + // gate, each event-logged, so the feed shows exactly what authority was + // minted and the grant rows carry the shared justification (`--note`). + if batch { + for _, r := range grantRows { + grantID, err := db.InsertGateOverrideGrantTx(tx, db.GateOverrideGrant{ + RunID: step.RunID, OriginStepID: step.ID, Gate: r.Gate, + Exit: r.Exit, Reason: r.Reason, Note: note, CreatedAtMS: nowMS, + }) + if err != nil { + return err + } + if err := recordEvent(tx, eventRecord{ + Kind: EventGateOverrideGranted, RunID: step.RunID, + Instance: step.Instance, IssueID: step.IssueID, + Data: fmt.Sprintf("%s#%d", r.Gate, grantID), + }); err != nil { + return err + } + } + } // An `override-pass` on a held cluster IS the approve-computed answer — it // records `done` with a `pass` routing, which is exactly what heldDecision // reads back as an approval — so it must leave the same record `approve` @@ -602,20 +774,36 @@ func OverridePassSkipsInterposedTargets(conn *sql.DB, stepID int) []string { return nil } spec := workflow.StepByName(defs[step.WorkflowID], step.StepName) + if warning := overridePassInterposedWarning(step.Instance, spec); warning != "" { + return []string{warning} + } + return nil +} + +// overridePassInterposedWarning is the DKT-470 sentence, computed ONCE for +// both surfaces that present it: the advisory warning +// (OverridePassSkipsInterposedTargets, printed beside an acknowledged +// resolution) and resolveStep's pre-transaction refusal (DKT-861). The +// detection is the same spec + workflow.ThresholdTargets read the reconcile's +// skipUnroutedTargets makes when the resolution commits — one logic path, so +// what is warned about and what is skipped cannot drift. Empty when there is +// nothing to warn about: a nil spec (a materialized step, whose minted name +// the definition never declares) or a threshold with no step-name routing. +func overridePassInterposedWarning(instance string, spec *workflow.Step) string { if spec == nil { - return nil + return "" } targets := workflow.ThresholdTargets(spec.Threshold) if len(targets) == 0 { - return nil + return "" } - return []string{fmt.Sprintf( + return fmt.Sprintf( "override-pass on %s records a generic %q routing and does not "+ "evaluate its threshold — interposed step(s) %s will NOT be "+ "routed to as a result, whatever the (unevaluated) payload would "+ "have decided; resolve them directly if their condition should "+ "still apply", - step.Instance, RoutingPass, strings.Join(targets, ", "))} + instance, RoutingPass, strings.Join(targets, ", ")) } // FailStep is `step fail` — the explicit-failure counterpart to `complete`. diff --git a/internal/engine/linked.go b/internal/engine/linked.go new file mode 100644 index 00000000..8d6f688f --- /dev/null +++ b/internal/engine/linked.go @@ -0,0 +1,308 @@ +package engine + +import ( + "database/sql" + "fmt" + "sort" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// The `issue.linked..` input form (DKT-547): a step consuming +// an artifact RECORDED UNDER ANOTHER ISSUE, reached through the consuming +// issue's declared relations rather than through an unenforced issue-body +// citation. +// +// The incident that forced it: ui-change@12 claimed its changes were "bound to +// an accepted ux-spec", but the spec was a doc produced by a spec-doc run on a +// DIFFERENT issue, and no input form reached it — every legal form is same-run +// and same-issue (`.`, `issue.latest.`, the issue forms). So +// the spec reached executors only when the issue body happened to cite it: 33 +// design-qa instances across 3 runs all relying on prose. +// +// RESOLUTION IS ACTIVATION'S, NOT ASSEMBLY'S. Context assembly is pure and +// snapshot-pinned (§6.6): it may not read live state, and "the linked issue's +// latest artifact" is live state — a later run on that issue would change the +// answer mid-run. So activation resolves the relation and the artifact ONCE, +// while it snapshots everything else about the issue, and records the resolved +// ARTIFACT IDS in the issue's `issue_snapshot` under a `linked` key. Artifact +// rows are never mutated (reliability-delta §2.1), so an id is a content pin +// exactly as a `pins` row's hash is — assembly reads the pinned rows by id and +// never asks the live question. And because re-activation never re-snapshots +// (RA2's reasoning), the binding cannot drift under a run already under way. +// +// FAILURE IS LOUD, AT ACTIVATION, per the acceptance criteria: a declared +// relation with no link, or a link whose issues hold no artifact of the kind, +// refuses the whole activation inside the fat transaction — the binding is +// enforced rather than conventional, and nothing is written. + +// linkedSuffix is the snapshot key for one declared form: the declaration +// verbatim, minus the `issue.linked.` prefix — "." exactly as +// the author spelled it, so assembly's lookup is a verbatim match and no +// normalization rule has to agree across two packages. +func linkedSuffix(declared string) string { + return strings.TrimPrefix(declared, workflow.InputIssueLinkedPrefix) +} + +// resolveLinkedInputs resolves every distinct `issue.linked..` +// declaration in a bound definition for ONE issue, at activation. +// +// Per declaration: the linked issue(s) by relation (canonical token = this +// issue is the relation's source; inverse token = its target; the symmetric +// `relates_to` admits both directions), then each linked issue's latest +// recorded artifact of the kind — highest artifact id whose producing step +// recorded its work (done or superseded, recordedProducer's rule), across +// every run in the store, which is the same "highest id is newest" reading +// latestPerProducer applies within a run. +// +// A linked issue with NO artifact of the kind contributes nothing rather than +// failing: relations are overloaded — `depends_on` orders scheduling as well +// as binding specs — so a ui-change issue depending on three implementation +// issues and one spec issue must resolve the spec, not refuse over the +// implementations. The failure modes are the empty ones: no relation at all, +// or a relation none of whose issues holds the kind. +// +// The result maps each declaration's suffix (".", verbatim) to +// the pinned artifact ids, ordered by linked issue id then artifact id — a +// pure function of the store, so two activations of the same state pin the +// same rows. +func resolveLinkedInputs( + tx *sql.Tx, issue *model.Issue, def *workflow.Definition, +) (map[string][]int, error) { + var out map[string][]int + for _, step := range def.Steps { + for _, input := range step.Inputs { + relation, kind, ok := workflow.LinkedInput(input) + if !ok { + continue + } + suffix := linkedSuffix(input) + if _, done := out[suffix]; done { + continue + } + + ids, err := resolveLinkedDeclaration(tx, issue, input, relation, kind) + if err != nil { + return nil, err + } + if out == nil { + out = make(map[string][]int) + } + out[suffix] = ids + } + } + return out, nil +} + +// resolveLinkedDeclaration is one declaration's resolution: linked issues, +// then the latest recorded artifact of the kind per linked issue, with the +// loud refusals. +func resolveLinkedDeclaration( + tx *sql.Tx, issue *model.Issue, declared, relation, kind string, +) ([]int, error) { + linked, err := linkedIssueIDs(tx, issue.ID, relation) + if err != nil { + return nil, err + } + if len(linked) == 0 { + return nil, validationErr( + "issue %s declares input %q, but has no %s relation to any issue; "+ + "link one with `docket issue link add` before activating", + model.FormatID(issue.ID), declared, relation) + } + + var ids []int + for _, linkedID := range linked { + id, err := latestArtifactOfKind(tx, linkedID, kind) + if err != nil { + return nil, err + } + if id > 0 { + ids = append(ids, id) + } + } + if len(ids) == 0 { + return nil, validationErr( + "issue %s declares input %q, but none of its %s-linked issue(s) "+ + "(%s) has a recorded artifact of kind %q; produce and record "+ + "one there before activating", + model.FormatID(issue.ID), declared, relation, + formatIDList(linked), kind) + } + return ids, nil +} + +// linkedIssueIDs resolves the issues one relation token reaches from a +// consuming issue, sorted ascending so the pin order is deterministic. +func linkedIssueIDs(tx *sql.Tx, issueID int, relation string) ([]int, error) { + rt, inverse, err := model.ParseRelationDirection(relation) + if err != nil { + // V11 refuses unknown relation tokens at register, so reaching here + // means a definition was written directly into the table. + return nil, validationErr("input relation %q: %v", relation, err) + } + + // The canonical token reads the relation from its SOURCE ("A depends_on + // B": A's `depends_on` reaches B), the inverse token from its TARGET. The + // symmetric `relates_to` is its own inverse, so it reads both directions. + forward := !inverse || rt == model.RelationRelatesTo + backward := inverse || rt == model.RelationRelatesTo + + seen := make(map[int]bool) + var out []int + collect := func(query string) error { + rows, err := tx.Query(query, issueID, string(rt)) + if err != nil { + return fmt.Errorf("reading %s relations: %w", rt, err) + } + defer rows.Close() + for rows.Next() { + var id int + if err := rows.Scan(&id); err != nil { + return fmt.Errorf("reading %s relations: %w", rt, err) + } + if !seen[id] { + seen[id] = true + out = append(out, id) + } + } + return rows.Err() + } + + if forward { + if err := collect( + `SELECT target_issue_id FROM issue_relations + WHERE source_issue_id = ? AND relation_type = ?`); err != nil { + return nil, err + } + } + if backward { + if err := collect( + `SELECT source_issue_id FROM issue_relations + WHERE target_issue_id = ? AND relation_type = ?`); err != nil { + return nil, err + } + } + sort.Ints(out) + return out, nil +} + +// latestArtifactOfKind is one linked issue's contribution: the highest-id +// artifact of the kind whose producing step recorded its work, across every +// run — or 0 when the issue holds none. +func latestArtifactOfKind(tx *sql.Tx, issueID int, kind string) (int, error) { + var id int + err := tx.QueryRow( + `SELECT a.id FROM artifacts a + JOIN steps s ON s.id = a.step_id + WHERE s.issue_id = ? AND a.kind = ? AND s.status IN (?, ?) + ORDER BY a.id DESC LIMIT 1`, + issueID, kind, db.StepDone, db.StepSuperseded).Scan(&id) + if err == sql.ErrNoRows { + return 0, nil + } + if err != nil { + return 0, fmt.Errorf( + "resolving the latest %q artifact of %s: %w", + kind, model.FormatID(issueID), err) + } + return id, nil +} + +// formatIDList renders issue ids as display ids for an error message. +func formatIDList(ids []int) string { + out := make([]string, 0, len(ids)) + for _, id := range ids { + out = append(out, model.FormatID(id)) + } + return strings.Join(out, ", ") +} + +// resolveLinkedPinned is assembly's half: the ContextInputs for one declared +// `issue.linked` entry, loaded by the artifact ids activation pinned into the +// issue snapshot. +// +// The producer instance is rendered as `/` — the producing +// step belongs to another issue (usually another run), so its bare instance +// name would collide with this run's own namespace and say nothing about where +// the artifact came from. +// +// A missing snapshot entry is an error, not an empty resolution: activation +// pins every declaration the PINNED definition makes, in the same transaction +// that snapshots the issue, so absence means the ledger was edited by hand — +// resolving to nothing would silently reopen the unenforced-citation gap this +// form exists to close. +func resolveLinkedPinned( + tx *sql.Tx, step *db.Step, linked map[string][]int, declared string, +) ([]ContextInput, error) { + ids, ok := linked[linkedSuffix(declared)] + if !ok { + return nil, fmt.Errorf( + "step %s: input %q was not pinned at activation — the issue "+ + "snapshot holds no resolution for it", step.Instance, declared) + } + + out := make([]ContextInput, 0, len(ids)) + for _, id := range ids { + var ( + kind, body string + payload sql.NullString + instance sql.NullString + issueID sql.NullInt64 + ) + err := tx.QueryRow( + `SELECT a.kind, a.body, a.payload, s.instance, s.issue_id + FROM artifacts a + LEFT JOIN steps s ON s.id = a.step_id + WHERE a.id = ?`, id).Scan(&kind, &body, &payload, &instance, &issueID) + if err == sql.ErrNoRows { + return nil, fmt.Errorf( + "step %s: input %q was pinned to artifact %d at activation, "+ + "but the artifact no longer exists", step.Instance, declared, id) + } + if err != nil { + return nil, fmt.Errorf( + "reading pinned artifact %d for %s: %w", id, step.Instance, err) + } + + producer := "" + if instance.Valid && issueID.Valid { + producer = model.FormatID(int(issueID.Int64)) + "/" + instance.String + } + out = append(out, ContextInput{ + Artifact: fmt.Sprintf("ARTIFACT-%d", id), + Kind: kind, + ProducerStep: producer, + Body: body, + Payload: payload.String, + }) + } + return out, nil +} + +// linkedArtifacts is resolveLinkedPinned for a consumer that needs the ROWS — +// ResolveInputArtifacts' reading, sharing the pinned-id lookup so the two +// resolvers cannot disagree about which artifacts a declaration binds. +func linkedArtifacts( + tx *sql.Tx, step *db.Step, linked map[string][]int, declared string, +) ([]*db.Artifact, error) { + ids, ok := linked[linkedSuffix(declared)] + if !ok { + return nil, fmt.Errorf( + "step %s: input %q was not pinned at activation — the issue "+ + "snapshot holds no resolution for it", step.Instance, declared) + } + out := make([]*db.Artifact, 0, len(ids)) + for _, id := range ids { + a, err := db.GetArtifactTx(tx, id) + if err != nil { + return nil, fmt.Errorf( + "reading pinned artifact %d for %s: %w", id, step.Instance, err) + } + out = append(out, a) + } + return out, nil +} diff --git a/internal/engine/lookahead.go b/internal/engine/lookahead.go index d3c3474c..b5e686cb 100644 --- a/internal/engine/lookahead.go +++ b/internal/engine/lookahead.go @@ -106,7 +106,9 @@ func (s *Scheduler) lookaheadOffer(admitted []*db.Step) []offerEntry { for _, step := range admitted { entries = append(entries, offerEntry{step: step}) member[step.ID] = true - cost += step.ExpectedCost + // reservableCost: a vote step's declared cost is already in the floor + // at materialization (DKT-584), so the closure must not re-reserve it. + cost += reservableCost(step) } // ---- Phase A: membership, to a fixed point. --------------------------- @@ -164,7 +166,7 @@ func (s *Scheduler) lookaheadOffer(admitted []*db.Step) []offerEntry { continue } member[cand.ID] = true - cost += cand.ExpectedCost + cost += reservableCost(cand) entries = append(entries, offerEntry{step: cand, staged: true}) changed = true } diff --git a/internal/engine/loop.go b/internal/engine/loop.go index 83a7ee7b..e36c180d 100644 --- a/internal/engine/loop.go +++ b/internal/engine/loop.go @@ -4,6 +4,7 @@ import ( "database/sql" "errors" "fmt" + "sort" "strings" "github.com/ALT-F4-LLC/docket/internal/db" @@ -185,6 +186,245 @@ func newestIssueDiffUpTo(tx *sql.Tx, step *db.Step, ordinal int) (hash, body str return hash, body, nil } +// routingVerdictUnchanged reports whether the step whose routing is about to +// enter the loop recorded the IDENTICAL artifact at its own previous ordinal +// (DKT-589). +// +// It is roundMovedNothing's sibling, measuring the other end of the same +// round. roundMovedNothing asks whether the loop BODY moved any bytes; this +// asks whether the ROUTING STEP reached any new conclusion. RUN-31 is the case +// the first one cannot see: verify@0 reported AC2 and AC5 unmet, fix@1 +// committed an 82,742-byte diff, verify@1 reported AC2 and AC5 unmet with AC9 +// newly regressed, fix@2 committed 2,571 bytes, and verify@2 came back +// BYTE-IDENTICAL to verify@1. Both rounds moved real bytes, so DKT-340's +// byte-based guard correctly stayed silent — and 342,490 output tokens, 45.8% +// of the run, closed zero acceptance criteria. RUN-34 is the same shape with +// four byte-identical ac-reports. +// +// WHAT IS COMPARED IS THE WHOLE ARTIFACT, AS BYTES. The reference corpus's +// verify emits `ac-report` and the issue describes the symptom as "the same +// criterion ids at the same statuses" — but "criterion", "id" and "status" are +// the workflow author's vocabulary, not core's, exactly as `severity` and +// `finding` are (genericity.md, and roundMovedNothing's own reasoning). Core +// reads the kind the step's OWN `emits`/`params.output` declares — +// workflow.ArtifactKind, the same resolution `complete` used to record it and +// input resolution uses to bind it — and compares the recorded sha256, which +// covers body AND payload. No JSON is parsed and no kind is hardcoded: a +// workflow whose gate emits `qa-verdict` or `lint-report` gets the identical +// check for free, and one whose payload shape nobody here has seen cannot +// break it. +// +// IT FAILS TOWARD ENTERING THE ROUND, roundMovedNothing's direction and for its +// reason. Ordinal 0 has no previous round of its own; a step class that records +// no artifact at all (`type = "human"`, `type = "vote"` — ArtifactKind "") has +// nothing to compare, which is what keeps a rejected gate, a lost tally and an +// exhausted attempt budget out of this check entirely; and an artifact with no +// bytes in either channel is an absent measurement, never a matching one. +// +// The CURRENT side is read AT this step's exact ordinal, not "at or below" it +// the way newestIssueDiffUpTo reads the body's diff. The suppression that makes +// "up to" necessary there is `issue.diff`-only (DKT-258/DKT-259) — a declared +// emit always records — so "at or below" would here mean something quite +// different and quite wrong: a routing step that recorded NOTHING at this +// ordinal (an exhausted budget, a rejected gate materialized on another row) +// would read its own previous round's artifact on both sides and park on a +// comparison of one row with itself. The PRIOR side keeps priorRoundHandBack's +// "newest strictly below" so a cluster whose routing step skipped an ordinal +// still compares against the last verdict it actually reached. +func routingVerdictUnchanged( + tx *sql.Tx, step *db.Step, def *workflow.Definition, trigger string, +) (bool, int, error) { + if step.Ordinal < 1 { + return false, 0, nil + } + spec := workflow.StepByName(def, trigger) + if spec == nil { + return false, 0, nil + } + kind := workflow.ArtifactKind(spec) + if kind == "" { + return false, 0, nil + } + now, _, err := newestStepEmit(tx, step, trigger, kind, step.Ordinal, step.Ordinal) + if err != nil || now == "" { + return false, 0, err + } + before, priorOrdinal, err := newestStepEmit(tx, step, trigger, kind, 0, step.Ordinal-1) + if err != nil || before == "" { + return false, 0, err + } + return now == before, priorOrdinal, nil +} + +// newestStepEmit is the fingerprint of the newest artifact of one KIND recorded +// by one STEP NAME's rows within an ordinal range, AND the ordinal it was +// recorded at — "" when that step recorded none there, and "" when the one it +// recorded carries no bytes at all. +// +// The ordinal comes back because the caller's park names it to the operator: a +// range read must report where it landed, not where the caller assumed it +// would. +// +// Scoped by (run, issue, step name, kind), which is priorRoundHandBack's shape +// rather than newestIssueDiffUpTo's: the issue-wide read is right for +// `issue.diff`, which is one cumulative fact about the tree whoever produced +// it, and wrong here, where the question is whether ONE step reached the same +// conclusion twice. Another producer's artifact of the same kind — the +// fixture's `synthesize` and `reconcile` both emit `findings` — says nothing +// about that. +// +// Newest-first by artifact id, matching every other "the latest artifact wins" +// read in the engine: a re-claimed step completing a second time records again, +// and the last record is the verdict that stands. +// +// AN ARTIFACT WITH NO BYTES IS NOT A MEASUREMENT. Two empty records agree with +// each other trivially — not because the step reached the same conclusion, but +// because it recorded no conclusion at all — and parking a loop on that is +// roundMovedNothing's degenerate-diff failure in a new place. +func newestStepEmit( + tx *sql.Tx, step *db.Step, name, kind string, lo, hi int, +) (string, int, error) { + if hi < lo || hi < 0 { + return "", 0, nil + } + var hash, body string + var payload sql.NullString + var ordinal int + err := tx.QueryRow( + `SELECT a.sha256, a.body, a.payload, s.ordinal FROM artifacts a + JOIN steps s ON s.id = a.step_id + WHERE a.run_id = ? AND s.issue_id = ? AND s.step_name = ? AND a.kind = ? + AND s.ordinal >= ? AND s.ordinal <= ? + ORDER BY a.id DESC LIMIT 1`, + step.RunID, step.IssueID, name, kind, lo, hi, + ).Scan(&hash, &body, &payload, &ordinal) + if errors.Is(err, sql.ErrNoRows) { + return "", 0, nil + } + if err != nil { + return "", 0, fmt.Errorf("reading %s's %s from %q at ordinals %d-%d: %w", + model.FormatID(step.IssueID), kind, name, lo, hi, err) + } + if strings.TrimSpace(body) == "" && strings.TrimSpace(payload.String) == "" { + return "", 0, nil + } + return hash, ordinal, nil +} + +// stalledVolume is what volumeStalled reports when the trigger's routed volume +// has plateaued: how many consecutive measured rounds ran without improvement, +// what the newest round routed, the best (smallest) volume any round achieved, +// and the declared tolerance — the four numbers the park's clause names. +type stalledVolume struct { + rounds, current, best, bound int +} + +// volumeStalled reports whether the triggering step's routed volume has gone +// `max_stalled_rounds` consecutive measured rounds without EVER falling below +// the smallest volume any earlier round recorded (DKT-870), or nil when it has +// not — including every case where the check is not in force or cannot measure. +// +// THE MEASUREMENT IS A COUNT OF PAYLOAD ELEMENTS, per round: for each ordinal, +// the newest artifact of the trigger's own declared kind, its payload parsed by +// the same shape rule the threshold's read uses (parsePayload), its length +// taken and nothing else read. That is the engine-visible rendering of the +// corpus's "standing-cluster volume" — one element per cluster is exactly what +// the routed payload is (§7.6) — reached without core learning what a cluster +// is. The read is scoped (run, issue, step name, kind) for newestStepEmit's +// reason: another producer's payload of the same kind says nothing about what +// THIS trigger keeps routing. +// +// "WITHOUT IMPROVEMENT" MEANS NO NEW STRICT MINIMUM, not endpoint-to-endpoint +// decrease. RUN-51's volumes (8, 12, 10, 10, 7, 10, 7, 11, 10, ~8) decrease +// between plenty of ADJACENT rounds while never trending anywhere, so a +// consecutive-pairs reading would stay silent for exactly the run this exists +// for; and a tie with the best round is not progress — RUN-51's round 6 +// re-achieving round 4's seven clusters was two rounds at the same wall. A +// genuinely converging loop sets a new minimum every round or two and never +// accumulates `bound` consecutive non-improving measurements. +// +// IT FAILS TOWARD ENTERING THE ROUND, its siblings' direction for their +// reason: no declared bound, no declared kind, a round with no recorded +// artifact, a payload with no bytes, and unparseable bytes are all absent +// measurements, never flat ones — a guard must not act on absence of evidence. +func volumeStalled( + tx *sql.Tx, step *db.Step, def *workflow.Definition, trigger string, +) (*stalledVolume, error) { + spec := workflow.StepByName(def, trigger) + if spec == nil || spec.MaxStalledRounds == nil || *spec.MaxStalledRounds <= 0 { + return nil, nil + } + kind := workflow.ArtifactKind(spec) + if kind == "" { + return nil, nil + } + + // The newest payload per ordinal, up to the round that just completed. + // Ordered by artifact id so a later row overwrites an earlier one in the + // map — the same "latest artifact wins" read as everywhere else. + rows, err := tx.Query( + `SELECT s.ordinal, a.payload FROM artifacts a JOIN steps s ON s.id = a.step_id + WHERE a.run_id = ? AND s.issue_id = ? AND s.step_name = ? AND a.kind = ? + AND s.ordinal <= ? + ORDER BY a.id`, + step.RunID, step.IssueID, trigger, kind, step.Ordinal) + if err != nil { + return nil, fmt.Errorf("reading %s's routed volumes: %w", + model.FormatID(step.IssueID), err) + } + defer rows.Close() + + byOrdinal := make(map[int]string) + var ordinals []int + for rows.Next() { + var ( + ordinal int + payload sql.NullString + ) + if err := rows.Scan(&ordinal, &payload); err != nil { + return nil, fmt.Errorf("reading a routed volume: %w", err) + } + if _, seen := byOrdinal[ordinal]; !seen { + ordinals = append(ordinals, ordinal) + } + byOrdinal[ordinal] = payload.String + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("reading the routed volumes: %w", err) + } + sort.Ints(ordinals) + + // Walk the measured rounds in order, tracking the best volume and how many + // measurements have arrived since a round last IMPROVED on it. The first + // measurement is the baseline, not an improvement streak of its own, so + // the signal needs at least bound+1 measured rounds before it can fire. + out := &stalledVolume{best: -1, bound: *spec.MaxStalledRounds} + for _, ordinal := range ordinals { + raw := byOrdinal[ordinal] + if strings.TrimSpace(raw) == "" { + // No bytes is not a measurement (newestStepEmit's rule): counting + // it as volume zero would let a round that recorded nothing read + // as total convergence, or a run of them read as a plateau. + continue + } + payloads, perr := parsePayload([]byte(raw)) + if perr != nil { + continue // Unparseable bytes measure nothing either. + } + volume := len(payloads) + if out.best < 0 || volume < out.best { + out.best, out.rounds = volume, 0 + } else { + out.rounds++ + } + out.current = volume + } + if out.best < 0 || out.rounds < out.bound { + return nil, nil + } + return out, nil +} + // EnterLoop performs a `fix-loop` routing, inside the caller's routing // transaction. // @@ -193,16 +433,19 @@ func newestIssueDiffUpTo(tx *sql.Tx, step *db.Step, ordinal int) (hash, body str // transaction with the step update that triggered it, because a partial loop // entry — a counter raised with no bodies instantiated, or bodies instantiated // twice — is not a state any later pass can repair without guessing. -// The routing STEP's own spec is deliberately not a parameter: nothing about a -// loop entry depends on which step routed. The counter is the issue's, the -// bound is the workflow's, the sweep set is `after_loop`'s downstream, and the -// instantiation is the definition's — so a `fix-loop` from `verify` and one -// from `reconcile` must produce identical effects. Taking the spec would invite -// a future reader to make one of those depend on it. +// +// WHICH step routed decides WHICH CLUSTER enters (§11.3 cluster scoping, +// DKT-544): the serving `loop = true` bodies, their `after_loop` roots' +// downstream, and any per-cluster round bound are all read for the TRIGGERING +// step's name — a `fix-loop` from `verify` and one from `reconcile` produce +// identical effects exactly when the same bodies serve both, which is always +// true in a workflow that declares no `serves`. The counter stays the ISSUE's +// either way: every cluster's entries share one ordinal sequence and one +// global ceiling. func EnterLoop( tx *sql.Tx, step *db.Step, def *workflow.Definition, nowMS int64, ) (*LoopOutcome, error) { - return enterLoop(tx, step, def, false, nowMS) + return enterLoop(tx, step, def, false, "", nowMS) } // EnterLoopAuthorized is EnterLoop under an EXPLICIT operator authorization — @@ -217,15 +460,39 @@ func EnterLoop( // The BOUND is not waived here, because it does not need to be: a grant raises // the effective maximum, so the same resolution already gets past it through // LoopGrantsTx. Only the convergence check has no such counter to move. +// +// `note` is the resolution's `-m` text — the operator's steering for the round +// they just paid for (DKT-725). It is stamped onto the round's freshly +// instantiated rows as their entering routing record, which is the ONE channel +// that reliably reaches a new round's rendered packet: issue comments are not +// a context source at all (§6.6's five-source rule), and a mid-run +// `description` edit never renders either, because `body_snapshot` is frozen +// at activation (§9 item 5's mid-run edit immunity). Before this stamp, the +// note died on the SUPERSEDED park's row — RUN-51's fix@9 rendered zero bytes +// of the judge-converged remedy the operator recorded when authorizing it. func EnterLoopAuthorized( - tx *sql.Tx, step *db.Step, def *workflow.Definition, nowMS int64, + tx *sql.Tx, step *db.Step, def *workflow.Definition, note string, nowMS int64, ) (*LoopOutcome, error) { - return enterLoop(tx, step, def, true, nowMS) + return enterLoop(tx, step, def, true, note, nowMS) } func enterLoop( - tx *sql.Tx, step *db.Step, def *workflow.Definition, authorized bool, nowMS int64, + tx *sql.Tx, step *db.Step, def *workflow.Definition, authorized bool, + note string, nowMS int64, ) (*LoopOutcome, error) { + // The TRIGGER is the routing step's name, and it selects the cluster + // (§11.3 cluster scoping): the serving bodies, the scoped sweep set, and + // the per-cluster bound are all read for it. A materialized `-held` + // name maps back to its routing step first (H17), because the definition + // declares clusters over its own names and a held row routes on its + // routing step's behalf. + trigger := step.StepName + if routing, ok := workflow.RoutingStepNameOf(trigger); ok { + trigger = routing + } + bodies := workflow.LoopBodiesFor(def, trigger) + downstream := afterLoopDownstreamFor(def, trigger) + // ---- Clause (1): the issue's counter, and the bound. ------------------- // // The increment happens FIRST and its result is what decides the bound, so @@ -297,6 +564,40 @@ func enterLoop( }, nil } + // THE CLUSTER'S OWN BOUND (§11.3 cluster scoping, DKT-544). A + // `max_fix_loops` declared on a `serves`-scoped body bounds ITS cluster's + // rounds, under the issue-level ceiling above — several independent + // quality gates can then carry different retry budgets without sharing + // one. Rounds are counted from the rows the entries themselves wrote: + // each entry instantiates its serving bodies at a fresh ordinal, so the + // distinct ordinals holding this cluster's scoped bodies ARE its entries, + // and no second counter can drift from the instances it is counting. + // + // The refusal takes the bound's exact shape — nothing superseded, nothing + // instantiated, counter restored, `waiting-human` naming `fix-round` as + // the way out — and an AUTHORIZED entry skips it the way it skips + // non-convergence: the operator who just granted the round has answered + // the question this bound asks. + if clusterMax := clusterMaxFixLoops(def, trigger); clusterMax > 0 && !authorized { + rounds, err := clusterRoundsUsed(tx, step, scopedClusterBodies(def, trigger)) + if err != nil { + return nil, err + } + if rounds+1 > clusterMax { + if err := restoreLoopCount(tx, step, count); err != nil { + return nil, err + } + reason := fmt.Sprintf( + "loop round %d for %q would exceed its cluster's max_fix_loops = %d "+ + "on %s; `docket step resolve --as fix-round` authorizes one more round", + rounds+1, trigger, clusterMax, model.FormatID(step.IssueID)) + return &LoopOutcome{ + Entered: false, Ordinal: count - 1, + Routing: workflow.OnFailWaitingHuman, Reason: reason, + }, nil + } + } + // NON-CONVERGENCE (DKT-340). A round that changed nothing in the issue's // scope cannot have changed what any check measures, so the next round is // handed the identical tree and reaches the identical verdict. The loop is @@ -315,19 +616,93 @@ func enterLoop( // budget raise was needed to survive the churn. Core cannot know that an // acceptance criterion is unmeetable; it can see that a round moved no // bytes the run is scoped to, which is enough to stop and ask. + // + // THE SECOND SIGNAL (DKT-589) is the same refusal read from the other end + // of the round. A round that moved bytes can still change nothing that + // MATTERS: RUN-31's fix@1 committed 82,742 bytes and fix@2 committed 2,571 + // more, and verify@2's ac-report came back byte-identical to verify@1's — + // so roundMovedNothing correctly saw movement and stayed silent while + // 342,490 output tokens, 45.8% of the run, closed zero acceptance + // criteria. When the ROUTING step's own recorded verdict is unchanged, the + // next round is being asked to fix the same thing on the same evidence. + // + // The two signals are OR'd into ONE refusal deliberately: one shape + // (nothing superseded, nothing instantiated, counter restored), one reason + // family, one `--as fix-round` way out, one `authorized` waiver. A second, + // differently-shaped park would be a second thing for an operator to learn + // and a second place for the escape hatch to be forgotten. Only the CLAUSE + // naming what did not change differs, because that is the one thing the + // operator reading the park needs to tell them apart. stalled, err := roundMovedNothing(tx, step, count) if err != nil { return nil, err } + clause := "" + if stalled { + clause = fmt.Sprintf( + "round %d changed nothing in %s's scope, so another round reads the "+ + "same tree and reaches the same verdict", + count-1, model.FormatID(step.IssueID)) + } else { + // The PRIOR ORDINAL the comparison actually matched is what the reason + // names, not `step.Ordinal - 1`: the prior side is the newest verdict + // STRICTLY BELOW this ordinal, and a cluster whose routing step sat out + // an ordinal would otherwise be told about a round it never ran. + repeated, priorOrdinal, err := routingVerdictUnchanged(tx, step, def, trigger) + if err != nil { + return nil, err + } + if repeated { + stalled = true + clause = fmt.Sprintf( + "%q recorded the identical verdict at ordinals %d and %d on %s, "+ + "so another round is handed the same verdict the last one "+ + "already failed to change", + trigger, priorOrdinal, step.Ordinal, model.FormatID(step.IssueID)) + } + } + // THE THIRD SIGNAL (DKT-870) is the one the first two cannot see: rounds + // that each move real bytes and each reach a byte-distinct verdict, while + // the SIZE of the standing set the trigger routes never improves. RUN-51 + // held 8-12 clusters flat across TEN rounds — rounds 8 and 9 alone spent + // ~271k and ~251k output tokens after the run's own ruling that the defect + // was structural — and RUN-50 held 7-10 across six; both ended only by + // operator action, because DKT-340 saw moving trees and DKT-589 saw + // changing verdicts, and nothing read the plateau. The measurement is the + // element COUNT of the trigger's own routed payload per round — packaging + // arithmetic, like `held`'s index list — so core still knows nothing about + // clusters or severities; the tolerance is the author's `max_stalled_rounds` + // (V38), zero or absent meaning the signal never fires. + // + // It joins the SAME refusal, deliberately: one shape, one reason family, + // one `--as fix-round` way out, one `authorized` waiver — the exact + // argument the two signals above make for each other. Only the clause + // differs, naming the flat volume, because that is what the operator + // reading the park needs. + if !stalled { + flat, err := volumeStalled(tx, step, def, trigger) + if err != nil { + return nil, err + } + if flat != nil { + stalled = true + clause = fmt.Sprintf( + "%q has routed %d element(s) after %d consecutive round(s) in "+ + "which the volume never fell below its best of %d "+ + "(max_stalled_rounds = %d) on %s, so the loop is holding "+ + "its standing volume flat rather than converging", + trigger, flat.current, flat.rounds, flat.best, flat.bound, + model.FormatID(step.IssueID)) + } + } if stalled && !authorized { if err := restoreLoopCount(tx, step, count); err != nil { return nil, err } reason := fmt.Sprintf( - "loop %d would repeat: round %d changed nothing in %s's scope, so "+ - "another round reads the same tree and reaches the same verdict; "+ + "loop %d would repeat: %s; "+ "`docket step resolve --as fix-round` authorizes one anyway", - count, count-1, model.FormatID(step.IssueID)) + count, clause) return &LoopOutcome{ Entered: false, Ordinal: count - 1, Routing: workflow.OnFailWaitingHuman, Reason: reason, @@ -338,8 +713,8 @@ func enterLoop( Entered: true, Ordinal: count, Routing: workflow.OnFailFixLoop, } - // ---- Clause (2): the supersede sweep. ---------------------------------- - superseded, err := supersedeSweep(tx, step, def, count, nowMS) + // ---- Clause (2): the supersede sweep, over THIS cluster's set. --------- + superseded, err := supersedeSweep(tx, step, downstream, count, nowMS) if err != nil { return nil, err } @@ -347,11 +722,11 @@ func enterLoop( // ---- Clauses (3) and (4): instantiate at the new ordinal. -------------- // - // The loop BODIES (clause 3) and the re-instantiated `after_loop` chain - // (clause 4) are one expansion at ordinal `count`, not two: they are the - // same set of rows written by the same rules, and separating them would + // The serving loop BODIES (clause 3) and the re-instantiated `after_loop` + // chain (clause 4) are one expansion at ordinal `count`, not two: they are + // the same set of rows written by the same rules, and separating them would // invite two writers disagreeing about which steps belong to ordinal k. - instantiated, err := instantiateOrdinal(tx, step, def, count, nowMS) + instantiated, err := instantiateOrdinal(tx, step, def, bodies, downstream, count, nowMS) if err != nil { return nil, err } @@ -375,10 +750,25 @@ func enterLoop( } out.Instantiated = append(out.Instantiated, repaired...) + // An AUTHORIZED entry's note is the operator's steering for the round, and + // this stamp is what carries it into the round's packets (DKT-725) — see + // EnterLoopAuthorized. Only an operator's `fix-round` note is stamped: an + // automatic entry's reason is machine diagnostics, and stamping it would + // change every looping packet for no ruling anyone issued. + if authorized && note != "" { + if err := stampEntryRouting(tx, step, out.Instantiated, note, nowMS); err != nil { + return nil, err + } + } + + // The event names the TRIGGER alongside the ordinal, because with cluster + // scoping the trigger is what decided which bodies ran — a ledger reader + // asking "whose rounds were these" should not have to re-derive it from + // the instance column. if err := recordEvent(tx, eventRecord{ Kind: EventLoopEntered, RunID: step.RunID, Instance: step.Instance, IssueID: step.IssueID, - Data: fmt.Sprintf(`{"ordinal":%d}`, count), AtMS: nowMS, + Data: fmt.Sprintf(`{"ordinal":%d,"trigger":%q}`, count, trigger), AtMS: nowMS, }); err != nil { return nil, err } @@ -394,9 +784,17 @@ func enterLoop( // loop must be bounded by the same number — a per-step reading would let the // bound depend on which step happened to route, which is not a bound at all. // +// A `serves`-scoped loop body's declaration is NOT this bound: it is that +// CLUSTER's round budget (clusterMaxFixLoops), so it is skipped here — the +// global ceiling stays whatever a non-cluster step declares, exactly as it +// would read without the clusters. +// // Zero means unbounded, which is what a definition declaring nothing means. func maxFixLoops(def *workflow.Definition) int { for _, step := range def.Steps { + if step.Loop && len(step.Serves) > 0 { + continue + } if step.MaxFixLoops != nil && *step.MaxFixLoops > 0 { return *step.MaxFixLoops } @@ -404,12 +802,89 @@ func maxFixLoops(def *workflow.Definition) int { return 0 } -// supersedeSweep is §7.3, exactly. +// scopedClusterBodies is the set of `serves`-SCOPED loop bodies serving one +// trigger — LoopBodiesFor minus the serve-everything bodies. It is the set the +// per-cluster bound is declared on and counted over: an unscoped body +// instantiates on EVERY cluster's entries, so counting its ordinals would +// charge one cluster for another's rounds. +func scopedClusterBodies(def *workflow.Definition, trigger string) []string { + var out []string + for _, step := range def.Steps { + if step.Loop && len(step.Serves) > 0 && workflow.ServesTrigger(step, trigger) { + out = append(out, step.Name) + } + } + return out +} + +// clusterMaxFixLoops reads the per-cluster round bound for one trigger: the +// smallest positive `max_fix_loops` declared on a `serves`-scoped body serving +// it (§11.3 cluster scoping, DKT-544). Zero means the cluster declares no +// bound of its own and only the issue-level ceiling applies. // -// On loop entry at ordinal k-1 -> k: every instance of the `after_loop` -// downstream set, at ordinal < k, that is NOT YET CLAIMED becomes `superseded` -// and is event-logged. Claimed, running, and gated instances are LEFT ALONE to -// finish; their eventual routing is made inert by StaleLineage. +// The SMALLEST when several bodies declare one, for the same reason a bound +// read off "whichever step routed" is not a bound: the cluster's budget must +// not depend on declaration order, and the conservative merge is the only +// deterministic one an author can reason about. +func clusterMaxFixLoops(def *workflow.Definition, trigger string) int { + out := 0 + for _, step := range def.Steps { + if !step.Loop || len(step.Serves) == 0 || !workflow.ServesTrigger(step, trigger) { + continue + } + if step.MaxFixLoops == nil || *step.MaxFixLoops <= 0 { + continue + } + if out == 0 || *step.MaxFixLoops < out { + out = *step.MaxFixLoops + } + } + return out +} + +// clusterRoundsUsed counts the rounds a cluster has already run for one issue: +// the distinct ordinals at which any of its scoped bodies has an instance. +// Every entry instantiates its serving bodies at the entry's fresh ordinal and +// bodies instantiate nowhere else — never at ordinal 0 — so the instances ARE +// the entry record, and a count read from them cannot drift from what actually +// ran the way a separate counter could. A bounded refusal wrote no body rows, +// so refusals are correctly not counted as rounds. +// +// Bodies SHARED between clusters (one `serves` list naming several triggers) +// are counted wherever they ran: a shared body's round is a real round of every +// cluster it serves, which is the conservative reading of a budget. +func clusterRoundsUsed(tx *sql.Tx, step *db.Step, bodies []string) (int, error) { + if len(bodies) == 0 { + return 0, nil + } + placeholders := strings.Repeat("?,", len(bodies)) + placeholders = placeholders[:len(placeholders)-1] + args := []any{step.RunID, step.IssueID} + for _, name := range bodies { + args = append(args, name) + } + var rounds int + err := tx.QueryRow( + `SELECT COUNT(DISTINCT ordinal) FROM steps + WHERE run_id = ? AND issue_id = ? AND step_name IN (`+placeholders+`)`, + args...).Scan(&rounds) + if err != nil { + return 0, fmt.Errorf("counting %s's cluster rounds: %w", + model.FormatID(step.IssueID), err) + } + return rounds, nil +} + +// supersedeSweep is §7.3, exactly, over the TRIGGERING CLUSTER's set. +// +// On loop entry at ordinal k-1 -> k: every instance of the entering cluster's +// `after_loop` downstream set (`downstream`, computed by the caller for the +// triggering step), at ordinal < k, that is NOT YET CLAIMED becomes +// `superseded` and is event-logged. Claimed, running, and gated instances are +// LEFT ALONE to finish; their eventual routing is made inert by StaleLineage. +// Instances OUTSIDE the cluster's set are not candidates at all — another +// cluster's chain keeps its current instances, exactly as a branch parallel to +// the loop always has (DKT-540). // // The status table (§7.7) is the specification and is worth stating as one: // @@ -433,9 +908,8 @@ func maxFixLoops(def *workflow.Definition) int { // an operator was asked. The sweep is therefore written as an explicit // unclaimed-and-pending test rather than as a negation. func supersedeSweep( - tx *sql.Tx, step *db.Step, def *workflow.Definition, ordinal int, nowMS int64, + tx *sql.Tx, step *db.Step, downstream map[string]bool, ordinal int, nowMS int64, ) ([]string, error) { - downstream := afterLoopDownstream(def) if len(downstream) == 0 { return nil, nil } @@ -527,12 +1001,20 @@ func supersedeSweep( return out, nil } -// afterLoopDownstream computes §7.3 (1)'s set: the `after_loop` step and every -// step transitively `after` it. +// afterLoopDownstream computes §7.3 (1)'s MERGED set: every `after_loop` root +// and every step transitively `after` one — the union over every cluster. // // The traversal is over the DEFINITION, not over expanded rows, because the set // is a property of the workflow's shape and must be the same on every entry — // including one where a downstream step has no instance yet. +// +// This merged closure stays the reading of every consumer that asks "is this +// step part of what ANY loop re-runs" — the scheduler's ordering predicates +// (stage.go), the loop-producer input redirect (context.go) — because for them +// a per-cluster answer would have to be re-derived per candidate trigger with +// no trigger in hand. Loop ENTRY itself uses afterLoopDownstreamFor: the entry +// knows its trigger, and sweeping or re-instantiating another cluster's chain +// is exactly what cluster scoping exists to stop (DKT-544). func afterLoopDownstream(def *workflow.Definition) map[string]bool { var roots []string for _, step := range def.Steps { @@ -540,6 +1022,26 @@ func afterLoopDownstream(def *workflow.Definition) map[string]bool { roots = append(roots, step.AfterLoop) } } + return downstreamClosure(def, roots) +} + +// afterLoopDownstreamFor is afterLoopDownstream scoped to one triggering step: +// the closure over only the `after_loop` roots declared by steps SERVING that +// trigger (§11.3 cluster scoping). With no `serves` declared anywhere, every +// declarer serves every trigger and this is afterLoopDownstream exactly. +func afterLoopDownstreamFor(def *workflow.Definition, trigger string) map[string]bool { + var roots []string + for _, step := range def.Steps { + if step.AfterLoop != "" && workflow.ServesTrigger(step, trigger) { + roots = append(roots, step.AfterLoop) + } + } + return downstreamClosure(def, roots) +} + +// downstreamClosure is the shared walk under both root sets: the roots plus +// every step transitively `after` one. +func downstreamClosure(def *workflow.Definition, roots []string) map[string]bool { if len(roots) == 0 { return nil } @@ -573,18 +1075,21 @@ func afterLoopDownstream(def *workflow.Definition) map[string]bool { return set } -// instantiateOrdinal writes the rows clauses (3) and (4) call for: the loop -// bodies and the re-instantiated `after_loop` chain, all at ordinal k. +// instantiateOrdinal writes the rows clauses (3) and (4) call for: the +// triggering cluster's loop bodies and its re-instantiated `after_loop` +// chain, all at ordinal k. `bodies` and `downstream` are the caller's — the +// same trigger-scoped sets the sweep and the bound already read, so one +// entry's four effects cannot disagree about which cluster it is. // // Steps NOT in either set are not re-instantiated. `implement` is the example // worth holding onto: it is upstream of `after_loop`, it never re-runs, and its // artifact stays bound at ordinal 0 — which is exactly why §7.4's fallback is -// per-input rather than per-step. +// per-input rather than per-step. Another cluster's bodies and chain are +// likewise untouched: their rounds are their own triggers' to mint. func instantiateOrdinal( - tx *sql.Tx, step *db.Step, def *workflow.Definition, ordinal int, nowMS int64, + tx *sql.Tx, step *db.Step, def *workflow.Definition, + bodies, downstream map[string]bool, ordinal int, nowMS int64, ) ([]string, error) { - downstream := afterLoopDownstream(def) - // The issue's subject, for `when` predicates — read from the SNAPSHOT, so a // re-instantiation at ordinal k evaluates the same predicate against the // same facts activation froze. Reading the live issue here would make a @@ -595,12 +1100,13 @@ func instantiateOrdinal( return nil, err } - // Clause (3) instantiates `loop = true` steps; clause (4) re-instantiates - // the `after_loop` chain. workflow.ExpandOrdinal applies both rules and - // nothing else, so the ordering and the fanout indices are the same pure - // function ordinary expansion uses (§5.3.1) — a second implementation here - // is how ordinal 1's topology would drift from ordinal 0's. - rows := workflow.ExpandOrdinal(def, subject, ordinal, downstream) + // Clause (3) instantiates the serving `loop = true` steps; clause (4) + // re-instantiates the cluster's `after_loop` chain. workflow.ExpandOrdinal + // applies both rules and nothing else, so the ordering and the fanout + // indices are the same pure function ordinary expansion uses (§5.3.1) — a + // second implementation here is how ordinal 1's topology would drift from + // ordinal 0's. + rows := workflow.ExpandOrdinal(def, subject, ordinal, bodies, downstream) return writeInstances(tx, step, rows, nowMS) } @@ -611,6 +1117,55 @@ func instantiateOrdinal( // below writes rows the same way from a different expansion, and a step row // created by two writers with two ideas of which columns matter is how an // instance ends up missing the `skipped` event or its metadata. +// stampEntryRouting writes an authorized entry's note onto the round's freshly +// instantiated PENDING rows as their entering routing record, +// `fix-loop: ` (DKT-725). +// +// The record lands on each new row's OWN `steps.routing` column, so two +// standing rules carry it the rest of the way with no new machinery: +// resolutionOf renders it as the packet's `== RESOLUTION` section (DKT-247's +// own-row scoping intact — the note was recorded FOR these rows, at their +// minting), and loopEntryOf reports it as the bundle's `loop_entry.routing`, +// which finally matches §11.3's "the routing that entered it". +// +// A `pending` row carrying a routing record is the DKT-247 retry shape +// exactly — human.go returns a ruled-on step to `pending` with its note in +// `routing` — so no reader learns a new state here: readiness reads routing +// off `done` predecessors only, QuorumMisses requires a completed join, and +// the row's own completion overwrites the stamp with its real verdict at +// routing time, exactly as a retry's does. +// +// `skipped` rows are left alone: their `when` predicate said this issue never +// runs them, and a steering note on a step that will never render is a lie in +// the ledger. Every row in `instances` was written by THIS transaction's +// expansion, so reading them back inside it is consistent by construction. +func stampEntryRouting( + tx *sql.Tx, step *db.Step, instances []string, note string, nowMS int64, +) error { + record := routingRecord(workflow.OnFailFixLoop, note) + for _, instance := range instances { + var ( + id int + status string + ) + err := tx.QueryRow( + `SELECT id, status FROM steps + WHERE run_id = ? AND issue_id = ? AND instance = ?`, + step.RunID, step.IssueID, instance, + ).Scan(&id, &status) + if err != nil { + return fmt.Errorf("stamping the fix-round note on %s: %w", instance, err) + } + if status != db.StepPending { + continue + } + if err := db.SetStepRoutingTx(tx, id, record, db.StepPending, nowMS); err != nil { + return err + } + } + return nil +} + func writeInstances( tx *sql.Tx, step *db.Step, rows []workflow.StepInstance, nowMS int64, ) ([]string, error) { diff --git a/internal/engine/loop_cluster_test.go b/internal/engine/loop_cluster_test.go new file mode 100644 index 00000000..59e4e40a --- /dev/null +++ b/internal/engine/loop_cluster_test.go @@ -0,0 +1,326 @@ +package engine + +import ( + "database/sql" + "os" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-544: per-trigger fix-loop scoping. Two independent quality gates each +// carry their own loop cluster — a `serves`-scoped body, its own `after_loop` +// re-entry, and its own round budget — under one issue-level counter and one +// global ceiling. +// +// The fixture is the dotfiles.vorpal shape reduced to two gates: `draft` +// fans into two PARALLEL gates, each of which wants reject -> respawn-the- +// offender -> re-check semantics of its own. Before cluster scoping only one +// loop construct existed per workflow, so entering either gate's loop would +// have instantiated BOTH bodies and re-instantiated the union of both +// downstream chains. +// +// - cluster A: `prd-gate` -> `prd-fix` (serves prd-gate, bound 1, +// re-enters at prd-gate); +// - cluster B: `design-gate` -> `design-fix` (serves design-gate, bound 5, +// re-enters at design-gate); +// - the GLOBAL ceiling is `max_fix_loops = 2` on `prd-gate` — a non-cluster +// declaration, so it stays the issue's bound exactly as before. +const clusterSrc = ` +[pipeline] +name = "two-clusters" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "draft" +executor = "draft" +emits = "doc" + +[[step]] +name = "prd-gate" +after = ["draft"] +executor = "prd-review" +emits = "findings" +inputs = ["draft.doc"] +threshold = { "fix-loop" = "any(status == blocked)" } +max_fix_loops = 2 + +[[step]] +name = "design-gate" +after = ["draft"] +executor = "design-review" +emits = "report" +inputs = ["draft.doc"] +threshold = { "fix-loop" = "any(status == blocked)" } + +[[step]] +name = "prd-fix" +executor = "fix" +emits = "doc" +loop = true +serves = ["prd-gate"] +after_loop = "prd-gate" +max_fix_loops = 1 +inputs = ["prd-gate.findings", "draft.doc"] + +[[step]] +name = "design-fix" +executor = "fix" +emits = "doc" +loop = true +serves = ["design-gate"] +after_loop = "design-gate" +max_fix_loops = 5 +inputs = ["design-gate.report", "draft.doc"] +` + +const blockedPayload = `[{"status":"blocked"}]` +const okPayload = `[{"status":"ok"}]` + +// TestClusterEntryInstantiatesOnlyItsServingBodies is the scoping itself: +// each gate's `fix-loop` instantiates ITS body, re-instantiates ITS re-entry +// chain, sweeps ONLY its own downstream — and leaves the other cluster's +// instances exactly where they were. +func TestClusterEntryInstantiatesOnlyItsServingBodies(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, clusterSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "draft@0", "the draft", "") + + // Cluster A enters: prd-gate@0 routes fix-loop. + claimAndComplete(t, conn, e, "prd-gate@0", "findings", blockedPayload) + + if got := loopCount(t, conn, runID, issue); got != 1 { + t.Fatalf("loop_count = %d after prd-gate's fix-loop, want 1", got) + } + for _, want := range []string{"prd-fix@1", "prd-gate@1"} { + if !stepExists(t, conn, want) { + t.Errorf("%s does not exist; cluster A's entry must instantiate "+ + "its serving body and its after_loop chain", want) + } + } + for _, mustNot := range []string{"design-fix@1", "design-gate@1"} { + if stepExists(t, conn, mustNot) { + t.Errorf("%s exists; cluster A's entry must not instantiate "+ + "cluster B's body or chain", mustNot) + } + } + // Cluster B's pending gate is NOT swept: it is outside cluster A's + // after_loop downstream, so its ordinal-0 instance stays the current one. + if got := stepStatus(t, conn, "design-gate@0"); got != db.StepPending { + t.Errorf("design-gate@0 = %q after cluster A's entry, want %q — "+ + "another cluster's sweep must not reach it", got, db.StepPending) + } + if staleLineage(t, conn, "design-gate@0") { + t.Error("design-gate@0 reads as a stale lineage after cluster A's " + + "entry; nothing replaced it, so its routing must stay live") + } + + // Round 1 of cluster A passes. + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "prd-fix@1", "the fixed draft", "") + claimAndComplete(t, conn, e, "prd-gate@1", "findings", okPayload) + + // Cluster B enters, at the issue's NEXT ordinal — one counter, shared. + claimAndComplete(t, conn, e, "design-gate@0", "report", blockedPayload) + + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after design-gate's fix-loop, want 2", got) + } + for _, want := range []string{"design-fix@2", "design-gate@2"} { + if !stepExists(t, conn, want) { + t.Errorf("%s does not exist; cluster B's entry must instantiate "+ + "its serving body and its after_loop chain", want) + } + } + for _, mustNot := range []string{"prd-fix@2", "prd-gate@2"} { + if stepExists(t, conn, mustNot) { + t.Errorf("%s exists; cluster B's entry must not re-instantiate "+ + "cluster A's body or chain", mustNot) + } + } + // Cluster A's finished chain is untouched by cluster B's sweep. + if got := stepStatus(t, conn, "prd-gate@1"); got != db.StepDone { + t.Errorf("prd-gate@1 = %q after cluster B's entry, want %q", got, db.StepDone) + } + + // The ledger names each entry's trigger. + triggers := loopEnteredTriggers(t, conn, runID) + if len(triggers) != 2 || !strings.Contains(triggers[0], "prd-gate") || + !strings.Contains(triggers[1], "design-gate") { + t.Errorf("loop-entered events carry data %v, want the first to name "+ + "prd-gate and the second design-gate", triggers) + } + + // Round 2 of cluster B passes, and the issue completes over + // highest-ordinal instances per name — prd-fix at 1, design-fix at 2. + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "design-fix@2", "the redesigned draft", "") + claimAndComplete(t, conn, e, "design-gate@2", "report", okPayload) + + if got := issueStatusOf(t, conn, issue); got != "done" { + t.Errorf("issue status = %q with every chain finished, want done", got) + } + if got := runStatusOf(t, conn, runID); got != "done" { + t.Errorf("run status = %q, want done", got) + } +} + +// TestClusterBoundParksItsOwnTriggerOnly: a cluster's own `max_fix_loops` +// refuses ITS next round — waiting-human, counter restored, nothing +// instantiated, the fix-round way out named — while the global budget still +// has room and the other cluster's instances stand untouched. +func TestClusterBoundParksItsOwnTriggerOnly(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, clusterSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "draft@0", "the draft", "") + + // Cluster A's one round is minted... + claimAndComplete(t, conn, e, "prd-gate@0", "findings", blockedPayload) + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "prd-fix@1", "the fixed draft", "") + + // ...and its second is refused by the CLUSTER bound (1), not the global + // ceiling (2), which still has room. + claimAndComplete(t, conn, e, "prd-gate@1", "findings", blockedPayload) + + if got := stepStatus(t, conn, "prd-gate@1"); got != db.StepWaitingHuman { + t.Fatalf("prd-gate@1 = %q after its cluster's rounds are spent, want %q", + got, db.StepWaitingHuman) + } + raw := stepRoutingRaw(t, conn, "prd-gate@1") + if !strings.Contains(raw, "cluster") || !strings.Contains(raw, "max_fix_loops = 1") || + !strings.Contains(raw, "fix-round") { + t.Errorf("prd-gate@1 routing = %q, want it to name the cluster bound "+ + "of 1 and the fix-round way out", raw) + } + if got := loopCount(t, conn, runID, issue); got != 1 { + t.Errorf("loop_count = %d after a refused cluster round, want 1 — "+ + "a refusal is not an ordinal", got) + } + if stepExists(t, conn, "prd-fix@2") { + t.Error("prd-fix@2 exists; a bounded cluster entry must instantiate nothing") + } + // Cluster B's budget and instances are untouched by A's exhaustion: no + // body of B was ever minted, and its gate still waits at ordinal 0. + if stepExists(t, conn, "design-fix@1") || stepExists(t, conn, "design-fix@2") { + t.Error("a design-fix instance exists; cluster A's refusals must not touch cluster B") + } + if got := stepStatus(t, conn, "design-gate@0"); got != db.StepPending { + t.Errorf("design-gate@0 = %q, want %q", got, db.StepPending) + } +} + +// TestGlobalCeilingBoundsAClusterWithBudgetLeft: the issue-level counter is +// the ceiling over EVERY cluster — cluster B has four of its five rounds +// unspent, and the third entry is still refused because the ISSUE has none. +func TestGlobalCeilingBoundsAClusterWithBudgetLeft(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, clusterSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "draft@0", "the draft", "") + + // Two rounds of cluster B spend the whole GLOBAL budget (2). Each gate + // records its OWN report (roundReport): a routing step repeating the + // byte-identical verdict parks the loop under DKT-589, and this test's + // subject is the global ceiling, not the identical-verdict park. + claimAndComplete(t, conn, e, "design-gate@0", roundReport(0), blockedPayload) + driveFixtureRound(t, 1) + claimAndComplete(t, conn, e, "design-fix@1", "the redesigned draft", "") + claimAndComplete(t, conn, e, "design-gate@1", roundReport(1), blockedPayload) + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "design-fix@2", "the re-redesigned draft", "") + + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Fatalf("loop_count = %d after two cluster B rounds, want 2", got) + } + + // The third refusal is the GLOBAL ceiling's, though the cluster's own + // bound (5) has plenty of room. + claimAndComplete(t, conn, e, "design-gate@2", roundReport(2), blockedPayload) + + if got := stepStatus(t, conn, "design-gate@2"); got != db.StepWaitingHuman { + t.Fatalf("design-gate@2 = %q past the global ceiling, want %q", + got, db.StepWaitingHuman) + } + raw := stepRoutingRaw(t, conn, "design-gate@2") + if !strings.Contains(raw, "max_fix_loops = 2") { + t.Errorf("design-gate@2 routing = %q, want it to name the global "+ + "ceiling of 2", raw) + } + if got := loopCount(t, conn, runID, issue); got != 2 { + t.Errorf("loop_count = %d after the global refusal, want 2", got) + } + if stepExists(t, conn, "design-fix@3") { + t.Error("design-fix@3 exists; the global ceiling must instantiate nothing") + } +} + +// TestNoServesDeclaredIsOneClusterExactly pins backward compatibility at the +// set level: with no `serves` anywhere, every trigger's cluster is EVERY body +// and EVERY root's downstream — the merged sets the single-construct reading +// always used. The whole legacy loop suite (loop_test.go and its siblings, +// over the serves-free fixture) is the behavioral half of this check. +func TestNoServesDeclaredIsOneClusterExactly(t *testing.T) { + src, err := os.ReadFile(fixturePath) + testsupport.Must(t, err, "reading the fixture: %v", err) + def, err := workflow.Parse(src) + testsupport.Must(t, err, "parsing the fixture: %v", err) + + for _, trigger := range []string{"reconcile", "verify", "commit-gate"} { + bodies := workflow.LoopBodiesFor(def, trigger) + if len(bodies) != 1 || !bodies["fix"] { + t.Errorf("LoopBodiesFor(%q) = %v, want every body (fix)", trigger, bodies) + } + + scoped := afterLoopDownstreamFor(def, trigger) + merged := afterLoopDownstream(def) + if len(scoped) != len(merged) { + t.Errorf("afterLoopDownstreamFor(%q) has %d members, want the "+ + "merged closure's %d", trigger, len(scoped), len(merged)) + } + for name := range merged { + if !scoped[name] { + t.Errorf("afterLoopDownstreamFor(%q) is missing %q", trigger, name) + } + } + } + + // And the fixture's global bound is still read as the global bound. + if got := maxFixLoops(def); got != 2 { + t.Errorf("maxFixLoops = %d on the serves-free fixture, want 2", got) + } + if got := clusterMaxFixLoops(def, "verify"); got != 0 { + t.Errorf("clusterMaxFixLoops = %d on the serves-free fixture, want 0 "+ + "— no scoped body, no cluster bound", got) + } +} + +// loopEnteredTriggers reads the run's loop-entered event data, in entry order. +func loopEnteredTriggers(t *testing.T, conn *sql.DB, runID int) []string { + t.Helper() + rows, err := conn.Query( + `SELECT data FROM events WHERE run_id = ? AND kind = ? ORDER BY seq`, + runID, EventLoopEntered) + testsupport.Must(t, err, "reading loop-entered events: %v", err) + defer rows.Close() + + var out []string + for rows.Next() { + var data string + testsupport.Must(t, rows.Scan(&data), "scanning loop-entered data") + out = append(out, data) + } + testsupport.Must(t, rows.Err(), "iterating loop-entered events") + return out +} diff --git a/internal/engine/loop_reentry_test.go b/internal/engine/loop_reentry_test.go index 517f6cd7..1ea1a112 100644 --- a/internal/engine/loop_reentry_test.go +++ b/internal/engine/loop_reentry_test.go @@ -17,7 +17,7 @@ func exhaustTheLoop(t *testing.T, conn *sql.DB, e *Engine) { t.Helper() for k := range 3 { driveToVerify(t, conn, e, k) - claimAndComplete(t, conn, e, fmt.Sprintf("verify@%d", k), "report", unmetPayload) + claimAndComplete(t, conn, e, fmt.Sprintf("verify@%d", k), roundReport(k), unmetPayload) } } @@ -107,7 +107,7 @@ func TestFixRoundGrantsOneRoundOnly(t *testing.T) { // Round 3 runs and fails again. The bound — now 2 declared + 1 granted — // is exhausted again, so it parks again rather than looping on. driveToVerify(t, conn, e, 3) - claimAndComplete(t, conn, e, "verify@3", "report", unmetPayload) + claimAndComplete(t, conn, e, "verify@3", roundReport(3), unmetPayload) if stepExists(t, conn, "fix@4") { t.Error("fix@4 exists; one grant must buy exactly one round") diff --git a/internal/engine/loop_starvation_test.go b/internal/engine/loop_starvation_test.go index eda58ba9..63ee54e4 100644 --- a/internal/engine/loop_starvation_test.go +++ b/internal/engine/loop_starvation_test.go @@ -101,8 +101,12 @@ func enterLoopAt(t *testing.T, conn *sql.DB, e *Engine, ordinal int) { } else { claimAndComplete(t, conn, e, fmt.Sprintf("fix@%d", ordinal), "the fix summary", "") } - claimAndComplete(t, conn, e, - fmt.Sprintf("assess@%d", ordinal), "the assessment", unmetPayload) + // And each round's assessment is its OWN, for DKT-589's sibling guard: a + // routing step that records the byte-identical verdict two ordinals running + // parks the loop, so a fixture that reuses one constant assessment would + // never reach the bound this file is about either. + claimAndComplete(t, conn, e, fmt.Sprintf("assess@%d", ordinal), + fmt.Sprintf("the assessment of round %d", ordinal), unmetPayload) } // --------------------------------------------------------------------------- diff --git a/internal/engine/loop_test.go b/internal/engine/loop_test.go index 86e8cdfb..933e5b63 100644 --- a/internal/engine/loop_test.go +++ b/internal/engine/loop_test.go @@ -27,6 +27,20 @@ const unmetPayload = `[{"status":"unmet"}]` // metPayload routes `verify` to `pass`: no threshold matches. const metPayload = `[{"status":"met"}]` +// roundReport is ONE ROUND'S verdict body, naming the ordinal that produced it. +// +// It exists for DKT-589 the way driveFixtureRound exists for DKT-340: a routing +// step that records the BYTE-IDENTICAL artifact at two consecutive ordinals now +// parks the loop, so a fixture reusing one constant `"report"` across rounds +// parks at the first repeat and never reaches the bound, the sweep, or the +// input binding those tests are actually about. A real round's report differs +// from the last one's whenever the round found anything new; these do too. A +// test whose SUBJECT is the identical-verdict park uses one constant +// deliberately — see dkt589_test.go. +func roundReport(ordinal int) string { + return fmt.Sprintf("the report of round %d", ordinal) +} + // --------------------------------------------------------------------------- // §11.3 identity: instance rendering // --------------------------------------------------------------------------- @@ -128,7 +142,7 @@ func TestLoopsAreBoundedByConstruction(t *testing.T) { // Loop 1 and loop 2, both legal at max_fix_loops = 2. for k := range 2 { driveToVerify(t, conn, e, k) - claimAndComplete(t, conn, e, fmt.Sprintf("verify@%d", k), "report", unmetPayload) + claimAndComplete(t, conn, e, fmt.Sprintf("verify@%d", k), roundReport(k), unmetPayload) if got := loopCount(t, conn, run.ID, issue); got != k+1 { t.Fatalf("loop_count = %d after loop %d, want %d", got, k+1, k+1) @@ -140,7 +154,7 @@ func TestLoopsAreBoundedByConstruction(t *testing.T) { // The third attempt: routing fires, and the bound converts it. driveToVerify(t, conn, e, 2) - claimAndComplete(t, conn, e, "verify@2", "report", unmetPayload) + claimAndComplete(t, conn, e, "verify@2", roundReport(2), unmetPayload) if stepExists(t, conn, "fix@3") { t.Error("fix@3 exists; max_fix_loops = 2 must make a third entry impossible") @@ -422,10 +436,11 @@ func TestOrdinalScopedInputBindingFallsBackPerInput(t *testing.T) { // Enter loop 1, then run ordinal 1 as far as `reconcile@1` so both inputs // have candidates and they sit at different ordinals. driveToVerify(t, conn, e, 0) - claimAndComplete(t, conn, e, "verify@0", "report", unmetPayload) + claimAndComplete(t, conn, e, "verify@0", roundReport(0), unmetPayload) // Ordinal 1 is driven by hand here rather than through driveToVerify, so // the stub tree is moved by hand too — otherwise round 1 changes nothing - // and DKT-340's guard correctly refuses to mint `fix@2`. + // and DKT-340's guard correctly refuses to mint `fix@2`. Each verify + // records its own report for DKT-589's sibling guard, for the same reason. driveFixtureRound(t, 1) claimAndComplete(t, conn, e, "fix@1", "the fix summary", "") completeReviewFanout(t, conn, e, 1) @@ -434,7 +449,7 @@ func TestOrdinalScopedInputBindingFallsBackPerInput(t *testing.T) { // A second loop entry, so there is a `fix@2` whose inputs we can inspect // with ordinal-1 and ordinal-0 candidates both present. - claimAndComplete(t, conn, e, "verify@1", "report", unmetPayload) + claimAndComplete(t, conn, e, "verify@1", roundReport(1), unmetPayload) inputs := contextInputs(t, conn, run.ID, "fix@2") @@ -879,7 +894,7 @@ func contextInputs(t *testing.T, conn *sql.DB, runID int, instance string) []Con artifacts, err := db.ListRunArtifactsTx(tx, step.RunID) testsupport.Must(t, err, "ListRunArtifactsTx: %v", err) - inputs, err := resolveInputs(tx, sched, step, spec, ri.BodySnapshot, artifacts) + inputs, err := resolveInputs(tx, sched, step, spec, ri.BodySnapshot, artifacts, nil) testsupport.Must(t, err, "resolveInputs(%s): %v", instance, err) return inputs } diff --git a/internal/engine/metadata.go b/internal/engine/metadata.go index c754ba17..63c55c37 100644 --- a/internal/engine/metadata.go +++ b/internal/engine/metadata.go @@ -25,7 +25,8 @@ import ( // TestMergeMetadataReadsNoKey. A merge that special-cased one key would be core // having an opinion about what a workflow author's bag of strings means. -// MetadataMaxBytes caps one completion's `--metadata` bag. +// MetadataMaxBytes caps one `--metadata` bag, on every verb that takes one — +// `claim`, `complete`, `fail`, `annotate`. // // The bag is opaque, so nothing else bounds it. Without a cap a worker can push // an artifact-sized body into a column the R7 rollup groups BY DISTINCT VALUE, @@ -181,7 +182,17 @@ func validateFailMetadataSize(raw string) error { return validateMetadataSizeWithRemedy(raw, "the note") } -// validateMetadataSizeWithRemedy is the shared cap check both verbs' size +// validateClaimMetadataSize is `claim`'s cap check, the same MetadataMaxBytes +// measured the same way (DKT-592). The remedy differs again because the +// channels do: a claim records nothing but the bag, so the only place bulk +// detail can go is the completion this claim is the start of — and naming +// `--artifact-file` outright would send a dispatcher looking for a flag `step +// claim` does not have. +func validateClaimMetadataSize(raw string) error { + return validateMetadataSizeWithRemedy(raw, "the artifact or the payload at completion") +} + +// validateMetadataSizeWithRemedy is the shared cap check the verbs' size // validators call — one size limit, one message shape, remedy text supplied // by the caller because only the caller knows which channels its verb offers. func validateMetadataSizeWithRemedy(raw, remedy string) error { diff --git a/internal/engine/metadata_test.go b/internal/engine/metadata_test.go index 86510f86..f5013ce5 100644 --- a/internal/engine/metadata_test.go +++ b/internal/engine/metadata_test.go @@ -765,6 +765,438 @@ func TestFailValidatesMetadataBeforeTheTransactionOpens(t *testing.T) { } } +// CLAIM METADATA (docs/tdd/completion-metadata.md §1.7, DKT-592). +// +// The bag a dispatcher knows at claim time was reaching the row only through +// `complete`, so the steps that failed or crashed — the ones an operator most +// wants to characterize — were the only steps carrying nothing. Two runs +// measured it: 76 of 119 steps carried the keys, then 55 of 83. A rollup over +// those keys went blind at precisely the rows that motivated it. +// +// GENERICITY, as above: the bags here use NEUTRAL keys. `tier_requested` and +// `desk_resolved` mirror the SHAPE of a routing pair — one fact known when the +// work is handed out, one known only when it comes back — without naming any +// instance's vocabulary. + +// TestClaimMetadataIsPersisted is the feature reduced to one assertion: the bag +// a dispatcher supplies at claim reaches the step's row IN THE CLAIM, with no +// completion anywhere in the test. +func TestClaimMetadataIsPersisted(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, + Metadata: `{"tier_requested":"a","desk_requested":"front"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + bag := stepMetadata(t, conn, stepID) + if bag["tier_requested"] != "a" { + t.Errorf("tier_requested = %v, want a — the claim's bag was not recorded (DKT-592)", + bag["tier_requested"]) + } + if bag["desk_requested"] != "front" { + t.Errorf("desk_requested = %v, want front", bag["desk_requested"]) + } +} + +// TestClaimMetadataSurvivesFailure is DKT-592's whole point: a step that FAILS +// after being claimed still carries what the dispatcher knew when it handed +// the step out. Before the claim-time write this step recorded nothing, which +// is why the drift row was blind exactly where it mattered. +func TestClaimMetadataSurvivesFailure(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, + Metadata: `{"tier_requested":"a","desk_requested":"front"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + // The failing worker reports nothing of its own — the case a crashed + // executor's supervisor produces. + err = e.FailStep(conn, stepID, claim.Token, "gave up", "", nowMS) + testsupport.Must(t, err, "fail: %v", err) + + bag := stepMetadata(t, conn, stepID) + if bag["tier_requested"] != "a" || bag["desk_requested"] != "front" { + t.Errorf("bag after a failure = %v, want both claim-time keys — a step "+ + "that never completes is the one whose dispatch facts matter most", bag) + } +} + +// TestClaimMetadataSurvivesTheReap is the CRASH case, which is not the failure +// case: nobody calls `fail` at all. The worker dies, the lease lapses, and the +// next claim reaps it. The reap writes status, lease and counters and must +// leave `metadata` alone. +func TestClaimMetadataSurvivesTheReap(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, TTLOverride: 1_000, + Metadata: `{"tier_requested":"a"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + // The holder is gone. The lazy reap fires on the next claim (§6.3). + const after = nowMS + 60_000 + _, err = ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker-2", NowMS: after, + Metadata: `{"tier_requested":"b"}`, + }) + testsupport.Must(t, err, "re-claim: %v", err) + + bag := stepMetadata(t, conn, stepID) + if bag["tier_requested"] != "b" { + t.Errorf("tier_requested = %v, want b — the re-claim's bag overlays the "+ + "dead attempt's, last-write-wins like every other writer of this column", + bag["tier_requested"]) + } +} + +// TestClaimThenCompleteCarriesEveryKey is the AC that the two writes compose: +// a step that completes normally ends up with BOTH pairs — the facts known at +// dispatch and the facts known only at completion. The claim-time keys must not +// be clobbered when the completion bag merges in. +func TestClaimThenCompleteCarriesEveryKey(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, + Metadata: `{"tier_requested":"a","desk_requested":"front"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("body"), NowMS: nowMS, + Metadata: `{"tier_resolved":"b","desk_resolved":"back"}`, + }) + testsupport.Must(t, err, "complete: %v", err) + + bag := stepMetadata(t, conn, stepID) + for key, want := range map[string]string{ + "tier_requested": "a", "desk_requested": "front", + "tier_resolved": "b", "desk_resolved": "back", + } { + if bag[key] != want { + t.Errorf("%s = %v, want %s — a completed step carries the whole pair "+ + "of pairs; the completion merge must not clobber the claim's keys", + key, bag[key], want) + } + } +} + +// TestClaimMetadataMergesOverDefinition is §1.2's merge, exercised through +// claim: a definition-only key survives, the dispatcher's keys are added, and a +// key in both takes the DISPATCHER's value. +func TestClaimMetadataMergesOverDefinition(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + seedStepMetadata(t, conn, stepID, `{"desk":"back","tier":"a"}`) + + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, + Metadata: `{"desk":"front","tier_requested":"b"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + bag := stepMetadata(t, conn, stepID) + if bag["tier"] != "a" { + t.Errorf("tier = %v, want a — a definition-only key must survive", bag["tier"]) + } + if bag["desk"] != "front" { + t.Errorf("desk = %v, want front — the dispatcher's value must win", bag["desk"]) + } + if bag["tier_requested"] != "b" { + t.Errorf("tier_requested = %v, want b — a dispatcher-only key must be added", + bag["tier_requested"]) + } +} + +// TestClaimWithoutMetadataLeavesDefinitionBag is the no-regression assertion: +// every existing claimant passes no `--metadata`, and none of them may see +// their step's bag change — nor pay a row_version bump for a write that has +// nothing to write. +func TestClaimWithoutMetadataLeavesDefinitionBag(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + const seeded = `{"desk":"back","tier":"a"}` + seedStepMetadata(t, conn, stepID, seeded) + + _, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + + if got := rawStepMetadata(t, conn, stepID); got != seeded { + t.Errorf("metadata = %q, want %q byte-identical — an absent flag writes nothing", + got, seeded) + } +} + +// TestClaimMetadataIsInTheContextBundle: the claim that recorded the bag also +// REPORTS it, so a worker reading `context.metadata` sees what it was +// dispatched with rather than the row's state before its own claim. +func TestClaimMetadataIsInTheContextBundle(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: `{"tier_requested":"a"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + + if claim.Context == nil { + t.Fatal("the claim returned no context bundle") + } + if claim.Context.Metadata["tier_requested"] != "a" { + t.Errorf("context.metadata = %v, want tier_requested=a — the bundle must "+ + "report the bag this same claim recorded", claim.Context.Metadata) + } +} + +// TestClaimMetadataReachesTheReportRollup is the AC in the surface the defect +// was observed in: a step that was claimed and then FAILED shows its +// dispatch-time keys in `run report`'s R7 rollup. No read-side change — the +// rollup already reads the column regardless of the step's status. +func TestClaimMetadataReachesTheReportRollup(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: `{"tier_requested":"a"}`, + }) + testsupport.Must(t, err, "claim: %v", err) + testsupport.Must(t, e.FailStep(conn, stepID, claim.Token, "gave up", "", nowMS), + "fail: %v", err) + + report, err := LoadRunReport(conn, run.ID, nowMS) + testsupport.Must(t, err, "LoadRunReport: %v", err) + + found := false + for _, rollup := range report.Metadata { + if rollup.Key != "tier_requested" { + continue + } + for _, vc := range rollup.Values { + if vc.Value == "a" && vc.Count == 1 { + found = true + } + } + } + if !found { + t.Errorf("report.Metadata = %s, want a FAILED step's tier_requested=a rollup — "+ + "the row the drift report was blind on", mustJSON(t, report.Metadata)) + } +} + +// TestClaimMetadataRefusals is §1.1's validation ladder on the claim side. +// Every refusal is pre-transaction, and on THIS path that has an extra edge: +// the claim's transaction performs the lazy reap and the CAS, so a refusal +// that ran inside it would consume an attempt. Each case asserts the step is +// untouched AND still claimable. +func TestClaimMetadataRefusals(t *testing.T) { + cases := []struct { + name string + metadata string + wantIn string + }{ + {"invalid JSON", `{"desk":`, "metadata"}, + {"a JSON array", `[{"desk":"front"}]`, "object"}, + {"a JSON scalar", `"front"`, "object"}, + {"a JSON number", `7`, "object"}, + {"JSON null", `null`, "object"}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + before := stepRowVersion(t, conn, stepID) + + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: tc.metadata, + }) + if err == nil { + t.Fatal("the claim was accepted, want VALIDATION_ERROR") + } + if code, _ := CodeOf(err); code != CodeValidation { + t.Errorf("error code = %q, want %q", code, CodeValidation) + } + if !strings.Contains(err.Error(), tc.wantIn) { + t.Errorf("err = %q, want it to mention %q", err.Error(), tc.wantIn) + } + + // §6.9: a refusal before the transaction writes NOTHING — no claim, + // no attempt, no status move. + if after := stepRowVersion(t, conn, stepID); after != before { + t.Errorf("row_version moved %d -> %d on a refusal", before, after) + } + step, err := db.GetStep(conn, stepID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Status != db.StepPending || step.Attempt != 0 { + t.Errorf("step is %q at attempt %d after a refused claim, want %q at 0", + step.Status, step.Attempt, db.StepPending) + } + + // The step is still claimable: the refusal cost the dispatcher + // nothing but the round trip. + if _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: `{"desk":"front"}`, + }); err != nil { + t.Errorf("the step was not claimable after a refusal: %v", err) + } + }) + } +} + +// TestClaimMetadataCap is §1.1.1 on the claim side. Same constant, same +// inclusive boundary asserted in both directions, and a message naming both +// numbers — with the remedy this verb can actually offer (a claim has no +// artifact or payload channel of its own; its completion does). +func TestClaimMetadataCap(t *testing.T) { + const envelope = `{"desk":""}` + atCap := `{"desk":"` + strings.Repeat("a", MetadataMaxBytes-len(envelope)) + `"}` + if len(atCap) != MetadataMaxBytes { + t.Fatalf("fixture is %d bytes, want exactly %d", len(atCap), MetadataMaxBytes) + } + overCap := `{"desk":"` + strings.Repeat("a", MetadataMaxBytes-len(envelope)+1) + `"}` + + t.Run("exactly at the cap is accepted", func(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: atCap, + }) + testsupport.Must(t, err, "a bag of exactly %d bytes was refused: %v", + MetadataMaxBytes, err) + }) + + t.Run("one byte over is refused", func(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + before := stepRowVersion(t, conn, stepID) + + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: overCap, + }) + if code, _ := CodeOf(err); code != CodeValidation { + t.Fatalf("error code = %q (err %v), want %q", code, err, CodeValidation) + } + for _, want := range []string{"16385", "16384", "completion"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to mention %q", err.Error(), want) + } + } + if after := stepRowVersion(t, conn, stepID); after != before { + t.Errorf("row_version moved %d -> %d on a refusal", before, after) + } + }) + + // The cap measures the RAW INPUT, not the merged result — the same + // property complete's cap has, for the same reason. + t.Run("the cap measures raw input, not the merged result", func(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + stepID := stepIDByInstance(t, conn, "implement@0") + seedStepMetadata(t, conn, stepID, + `{"tier":"`+strings.Repeat("b", MetadataMaxBytes-len(`{"tier":""}`))+`"}`) + + _, err := ClaimStep(conn, stepID, ClaimOptions{ + Owner: "worker", NowMS: nowMS, Metadata: `{"desk":"front"}`, + }) + testsupport.Must(t, err, "a small bag was refused because the MERGED text was large: %v", err) + }) +} + +// TestClaimValidatesMetadataBeforeTheTransactionOpens is +// TestFailValidatesMetadataBeforeTheTransactionOpens's twin, and it is a +// SOURCE-POSITION check for the same reason: `claimStepWithGates` runs under +// `defer tx.Rollback()`, so a refusal made inside the transaction is undone +// identically to one made before it and `row_version` cannot tell the +// placements apart. +// +// The placement matters more here than on `fail`: that transaction performs +// the LAZY REAP and the CAS, so validation drifting after `Begin()` would put +// a malformed bag one code change away from consuming an attempt. +func TestClaimValidatesMetadataBeforeTheTransactionOpens(t *testing.T) { + fset := token.NewFileSet() + file, err := parser.ParseFile(fset, "claim.go", nil, 0) + testsupport.Must(t, err, "parsing claim.go: %v", err) + + var beginPos, sizePos, decodePos token.Pos + + ast.Inspect(file, func(n ast.Node) bool { + fn, ok := n.(*ast.FuncDecl) + if !ok || fn.Name.Name != "claimStepWithGates" || fn.Body == nil { + return true + } + ast.Inspect(fn.Body, func(n ast.Node) bool { + call, ok := n.(*ast.CallExpr) + if !ok { + return true + } + switch fn := call.Fun.(type) { + case *ast.Ident: + switch fn.Name { + case "validateClaimMetadataSize": + sizePos = call.Pos() + case "DecodeMetadataBag": + decodePos = call.Pos() + } + case *ast.SelectorExpr: + if ident, ok := fn.X.(*ast.Ident); ok && ident.Name == "conn" && fn.Sel.Name == "Begin" { + if beginPos == token.NoPos { + beginPos = call.Pos() + } + } + } + return true + }) + return false + }) + + if beginPos == token.NoPos { + t.Fatal("claimStepWithGates no longer calls conn.Begin() — this test needs updating, not deleting") + } + if sizePos == token.NoPos { + t.Fatal("claimStepWithGates no longer calls validateClaimMetadataSize") + } + if decodePos == token.NoPos { + t.Fatal("claimStepWithGates no longer calls DecodeMetadataBag") + } + if sizePos > beginPos { + t.Error("validateClaimMetadataSize runs AFTER conn.Begin() — a refusal " + + "would happen inside the transaction that reaps and claims") + } + if decodePos > beginPos { + t.Error("DecodeMetadataBag runs AFTER conn.Begin() — a refusal would " + + "happen inside the transaction that reaps and claims") + } +} + // --- helpers --------------------------------------------------------------- func stepMetadata(t *testing.T, conn *sql.DB, stepID int) map[string]any { diff --git a/internal/engine/orphan_registration.go b/internal/engine/orphan_registration.go new file mode 100644 index 00000000..8254cf56 --- /dev/null +++ b/internal/engine/orphan_registration.go @@ -0,0 +1,249 @@ +package engine + +import ( + "fmt" + "os" + "sync" + + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// ORPHANED REGISTRATIONS (DKT-609): a registered workflow NAME that no file in +// any instance-config root declares any more. +// +// The state that produced this: a corpus commit renamed +// `security-load-bearing` to `security-change`, and the four registered +// versions of the old name stayed live in the store — registration is a row, +// not a file, and nothing removes a row (TestRegistrationSurvivesTOMLDeletion +// pins that on purpose). The next issue carrying the old label then failed +// activation with "matches 2 workflows: security-change@17, +// security-load-bearing@12", a refusal that named both candidates and said +// nothing about the fact that ONE OF THEM NO LONGER EXISTS ON DISK. The +// remedy — `docket workflow deprecate` — was obvious only after a git +// archaeology session established which of the two names the corpus had +// dropped. +// +// This file computes that missing half: for a registered name, whether a fresh +// scan of the configured roots finds a file declaring it. +// +// IT IS A SIBLING OF source_drift.go, NOT A DUPLICATE. CheckWorkflowSource +// re-reads ONE row's recorded `source_path` and compares bytes — the drift +// case, where the same file changed. A rename leaves no file to re-read: the +// old path is gone and the new file declares a different name, so the question +// has to be asked of the ROOTS rather than of a path. The same rules apply as +// they do there: the verdict is NEVER an error, it is computed on demand, and +// NOTHING IS EVER REPAIRED — deprecating a stranded registration is an +// operator's decision, taken with a verb, never a side effect of reporting. +// +// THE DISCOVERY IS THE SCAN'S, NOT A SECOND ONE. The names come from +// scanConfigDirs (the walk activation already runs) and from +// `[pipeline].name` (the identity registerConfigWorkflowTx already registers +// under), so this cannot disagree with registration about what a root contains +// or what name a file declares. + +// WorkflowOriginIndex answers, for a registered workflow name, whether a file +// in the instance-config roots still declares it. +// +// It is built LAZILY. Activation constructs one per activation and consults it +// only on a binding refusal, so the ordinary path pays nothing: reading and +// parsing every workflow file in the corpus a second time, on every +// activation, to answer a question nobody asked would be a real cost for a +// diagnostic. +type WorkflowOriginIndex struct { + // scan is the discovery this index reads. A nil scan means NO ROOT EXISTS + // (scanConfigDirs' F17 dormancy), which is `unchecked` and never + // `orphaned`. + scan *configScan + + once sync.Once + // declaredBy maps a declared `[pipeline].name` to the first file declaring + // it, in the scan's own precedence order. + declaredBy map[string]string + // err is the failure that stopped the build. Every verdict then reports + // `unchecked` carrying it: a half-read root can only produce false + // orphans, and a false orphan is exactly the claim that costs an operator + // an investigation. + err error +} + +// ScanWorkflowOrigins scans the instance-config roots for the read verbs +// (`workflow list --orphans`). +// +// The error is the SCAN's own refusal — a root that is a regular file, a +// dangling symlink — surfaced rather than swallowed, because those are the +// states in which every registration would otherwise be reported orphaned on +// the strength of having looked nowhere. +func ScanWorkflowOrigins() (*WorkflowOriginIndex, error) { + scan, err := scanConfigDirs(resolvePaths().InstanceConfigDirs()) + if err != nil { + return nil, err + } + return newWorkflowOriginIndex(scan), nil +} + +// newWorkflowOriginIndex wraps a scan an activation already performed. +func newWorkflowOriginIndex(scan *configScan) *WorkflowOriginIndex { + return &WorkflowOriginIndex{scan: scan} +} + +// Scanned reports whether any instance-config root existed to look in. It asks +// nothing of the files themselves — a build failure is Err's to report — so +// the two refusals a caller owes an operator stay separate facts. +// +// A false answer means every verdict this index gives is `unchecked`, and a +// caller whose whole purpose is orphan detection should say so rather than +// render an empty result that reads as "nothing is orphaned". +func (i *WorkflowOriginIndex) Scanned() bool { + return i != nil && i.scan != nil && len(i.scan.roots) > 0 +} + +// Roots are the canonicalized roots that were scanned, in precedence order. +func (i *WorkflowOriginIndex) Roots() []string { + if i == nil || i.scan == nil { + return nil + } + return i.scan.roots +} + +// Err is the failure that stopped the build, if any. +func (i *WorkflowOriginIndex) Err() error { + if i == nil { + return nil + } + i.build() + return i.err +} + +// Status is the verdict for one registered workflow NAME. +// +// A nil index is `unchecked` rather than a panic: the annotation call sites sit +// on refusal paths, where the only thing worse than no annotation is a crash +// while reporting someone else's error. +func (i *WorkflowOriginIndex) Status(name string) *model.WorkflowOriginStatus { + if i == nil || i.scan == nil || len(i.scan.roots) == 0 { + return &model.WorkflowOriginStatus{ + State: model.WorkflowOriginUnchecked, + Reason: "no instance-config root exists to scan, so nothing here " + + "can say whether a registered name still has a definition", + } + } + + i.build() + if i.err != nil { + return &model.WorkflowOriginStatus{ + State: model.WorkflowOriginUnchecked, + Roots: i.scan.roots, + Reason: i.err.Error(), + } + } + + if path, ok := i.declaredBy[name]; ok { + return &model.WorkflowOriginStatus{ + State: model.WorkflowOriginPresent, + Roots: i.scan.roots, + Path: path, + } + } + return &model.WorkflowOriginStatus{ + State: model.WorkflowOriginOrphaned, + Roots: i.scan.roots, + } +} + +// Orphaned is Status reduced to the one bit the refusal annotations need. +func (i *WorkflowOriginIndex) Orphaned(name string) bool { + return i.Status(name).Orphaned() +} + +// build reads and parses every workflow file the scan found, ONCE. +// +// A file that DOES NOT PARSE declares no name — registryIdentityKey's rule, +// for its reason: the refusal for an unparseable definition belongs to +// registerConfigWorkflowTx, which words it precisely, and a second opinion +// here would be the same failure reported twice and worse. The consequence is +// worth stating: while a workflow file in a root is unparseable, the name it +// WOULD declare reads as orphaned — but that same file already refuses the +// next activation outright while adoption is on, so the repo is in a state an +// operator is being told about from a louder place. +// +// A file that cannot be READ is different, and fails the build: it is an +// environment fault rather than an authoring one, and every verdict falls back +// to `unchecked` rather than manufacturing orphans out of a permission error. +func (i *WorkflowOriginIndex) build() { + i.once.Do(func() { + i.declaredBy = make(map[string]string) + if i.scan == nil { + return + } + for _, path := range i.scan.paths { + if isSchemaConfigPath(path) { + continue + } + src, err := os.ReadFile(path) + if err != nil { + i.err = fmt.Errorf( + "reading the instance-config workflow %s: %w", path, err) + i.declaredBy = nil + return + } + def, perr := workflow.Parse(src) + if perr != nil { + continue + } + // FIRST DECLARATION WINS, matching the roots' precedence order. + // Two roots declaring one name with different bytes is already + // refused at registration (crossRootConflictErr); here the + // question is only "does a file declare it", so the first is as + // good an answer as the last. + if _, dup := i.declaredBy[def.Pipeline.Name]; !dup { + i.declaredBy[def.Pipeline.Name] = path + } + } + }) +} + +// DescribeWorkflowOrigin renders one verdict as a human-readable clause, so +// every reader of this fact says the same thing about the same state — the +// shape DescribeWorkflowSource established. +func DescribeWorkflowOrigin(s *model.WorkflowOriginStatus) string { + if s == nil { + return "" + } + switch s.State { + case model.WorkflowOriginPresent: + return fmt.Sprintf( + "present — %s in the instance config still declares this name", s.Path) + case model.WorkflowOriginOrphaned: + return "ORPHANED — no file in any instance-config root declares this " + + "name any more; the registration outlives its definition (a rename " + + "or a deletion), and it still binds until it is deprecated" + default: + return fmt.Sprintf("unchecked — %s", s.Reason) + } +} + +// orphanAnnotation is the clause refList appends to a candidate whose name has +// no definition left on disk. +// +// It is appended ONLY to orphans, never to the healthy side. The refusal it +// decorates already lists both candidates; annotating the one that is +// anomalous is what makes the pair distinguishable at a glance, whereas +// annotating both would double the length of every ordinary two-name refusal +// to say "normal" twice. +const orphanAnnotation = " (no source on disk — orphaned registration, deprecation candidate)" + +// orphanRefusalHint is the sentence appended to a binding refusal in which at +// least one candidate is orphaned: what the state IS, and the one verb that +// clears it. +// +// IT PRESCRIBES, IT DOES NOT ACT. Deprecating a name is a decision about which +// definition the corpus meant to keep, and the engine holds no opinion on +// that — it only knows that one of the two names it is refusing over has +// nothing behind it. +const orphanRefusalHint = "\n\nAn ORPHANED registration is a name no file in any " + + "instance-config root declares any more — usually the residue of a rename, " + + "since registering the new name never retires the old one. Retire it with " + + "`docket workflow deprecate @` (every registered version of " + + "that name — see `docket workflow list --orphans`), or restore the file " + + "that declared it." diff --git a/internal/engine/packet.go b/internal/engine/packet.go index c206ca7e..d516e3f1 100644 --- a/internal/engine/packet.go +++ b/internal/engine/packet.go @@ -9,6 +9,7 @@ import ( "strings" "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" "github.com/ALT-F4-LLC/docket/internal/workflow" ) @@ -95,7 +96,7 @@ const packetFrontmatterFence = "---" // entry living in the shared root resolves identically from any cwd, including a // linked worktree that carries no `.docket/` of its own. func resolvePacketFiles( - pins map[string]string, roots []string, entries []string, + runRef string, pins map[string]string, roots []string, entries []string, ) ([]PacketFile, error) { if len(entries) == 0 { return nil, nil @@ -113,7 +114,7 @@ func resolvePacketFiles( } seen[ref] = true - body, hash, err := readPinnedPacketFile(pins, roots, ref) + body, hash, err := readPinnedPacketFile(runRef, pins, roots, ref) if err != nil { return nil, err } @@ -155,8 +156,13 @@ func resolvePacketFiles( // unpinned one means the file was not in the config directory when the run // activated — reading the live tree would break the byte-identical property // outright. Refusing is the only answer consistent with §8's Properties clause. +// +// That row's MESSAGE then splits by whether the ref resolves on disk (DKT-818): +// a ref nothing ever wrote needs writing, while one sitting under a config root +// needs a run whose pin set includes it. Same code, same refusal — a true +// sentence about which of the two it is. func readPinnedPacketFile( - pins map[string]string, roots []string, ref string, + runRef string, pins map[string]string, roots []string, ref string, ) (body, hash string, err error) { candidates := make([]string, 0, len(roots)) for _, root := range roots { @@ -178,10 +184,31 @@ func readPinnedPacketFile( } } if !pinned { + // THE REFUSAL NAMES THE CAUSE IT ACTUALLY HAS (DKT-818). "Not pinned" + // has two causes with two different remedies, and the message used to + // state only one of them: "add it under an instance-config root and + // start a new run". On RUN-59 both unpinned fragments were ALREADY + // under `~/.docket/config/fragments/` — a repin had adopted contract + // bytes that reached them — so the conductor went looking for a missing + // file, found it present, and had to re-derive the real cause. The pin + // set was frozen at activation; the filesystem was never the problem. + // So the ladder branches HERE, where both facts are in hand: the ref is + // unpinned, and this walk already knows every path it could resolve to. + for _, full := range candidates { + if _, serr := os.Stat(full); serr != nil { + continue + } + return "", "", validationErr( + "packet file %q is not in %s's pin set, which froze at activation; "+ + "the file is on disk at %s, but a run reads only what it "+ + "snapshotted, so its presence cannot admit it here — start a new "+ + "run to pin it, or see `docket run repin --help`", + ref, pinSetOwner(runRef), full) + } return "", "", validationErr( - "packet file %q is not pinned by this run; a packet reads only files the "+ - "run snapshotted at activation, so add it under an instance-config "+ - "root and start a new run", ref) + "packet file %q is not in %s's pin set, which froze at activation, and "+ + "resolves under no instance-config root; add it under one and start "+ + "a new run to pin it", ref, pinSetOwner(runRef)) } // FIRST ROOT THAT HOLDS IT WINS, and the hash check below then applies to @@ -399,5 +426,17 @@ func stepPacketFiles( return nil, err } - return resolvePacketFiles(packetPinsForRun(pins), instanceConfigRoots(), entries) + return resolvePacketFiles( + model.FormatRunID(step.RunID), packetPinsForRun(pins), + instanceConfigRoots(), entries) +} + +// pinSetOwner names the run a refusal is about. A resolution that carries no +// run — a direct call in a unit test, or any future seam holding only a pin +// map — gets the deictic form, so the sentence reads correctly either way. +func pinSetOwner(runRef string) string { + if runRef == "" { + return "this run" + } + return runRef } diff --git a/internal/engine/packet_closure.go b/internal/engine/packet_closure.go new file mode 100644 index 00000000..d0337b5a --- /dev/null +++ b/internal/engine/packet_closure.go @@ -0,0 +1,192 @@ +package engine + +import ( + "os" + "path" + "path/filepath" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// PACKET-CLOSURE PINNING (DKT-581). +// +// Activation used to pin EVERY file the config scan walked. That made the pin +// set the whole corpus, so a corpus install mid-run drifted pins for files the +// run could never read — 7 of 18 terminal runs in the measured week were +// abandoned over exactly that, and `verify-pins` reported drift on contracts +// the bound workflows never render. +// +// The pin set is now the CLOSURE the bound workflows actually reach: +// +// - every step's declared `packet` entries, with `{executor}` substituted +// the way expansion substitutes it — per fanout hint for a fanout step, +// the declared executor otherwise. EVERY declared step contributes, +// including `loop = true` bodies (they instantiate at loop entry) and +// steps a `when` will skip (expansion still requires their entries +// pinned); +// - the files those entries' `packet_includes` frontmatter declares, +// walked transitively — a fragment naming a fragment stays pinned even +// if the resolver later deepens past its current one include level; +// - `policy.toml`, the instance's policy surface, which is read by the +// harness rather than by a step and so appears in no `packet` list. +// +// Registered-object pins (workflows, schemas) were already narrow — one pin +// per BOUND workflow and per schema those workflows' steps declare — so this +// brings the file pins to the same standard: a run pins what it can read, +// and a corpus edit anywhere else is a non-event for `verify-pins`. +// +// `--pin PATH` files are deliberately NOT filtered: they are the operator's +// explicit additions, and the operator saying "pin this" is the closure for +// that file. + +// policyPinRef is the config-relative ref of the instance's policy file. Core +// never reads its content — it is pinned because the harness resolves policy +// from it and a run must be able to say which policy it ran under. +const policyPinRef = "policy.toml" + +// packetClosurePins filters the config scan's pins to the packet closure the +// bound workflows reach. Order is preserved from scan.pins, which is already +// sorted by ref, so the recorded set stays deterministic. +func packetClosurePins( + scan *configScan, runIssues []*db.RunIssue, bindings map[int]*boundDefinition, +) []db.Pin { + needed := packetClosureRefs(scan, runIssues, bindings) + out := make([]db.Pin, 0, len(needed)) + for _, p := range scan.pins { + if needed[path.Clean(filepath.ToSlash(p.Ref))] { + out = append(out, p) + } + } + return out +} + +// packetClosureRefs computes the set of config-relative refs the bound +// workflows can reach, as described at the top of this file. +func packetClosureRefs( + scan *configScan, runIssues []*db.RunIssue, bindings map[int]*boundDefinition, +) map[string]bool { + needed := map[string]bool{policyPinRef: true} + + // The worklist carries every ref whose `packet_includes` still needs + // walking. `needed` doubles as the seen-set, so an include cycle + // terminates and a diamond is walked once. + var queue []string + add := func(ref string) { + ref = path.Clean(filepath.ToSlash(ref)) + if ref == "" || ref == "." || needed[ref] { + return + } + needed[ref] = true + queue = append(queue, ref) + } + + // One walk per bound DEFINITION, not per issue: substitution depends only + // on the step's own executor/fanout declarations, never on the issue's + // subject, so two issues bound to one workflow reach one closure. + seenWorkflow := make(map[int]bool, len(bindings)) + for _, ri := range runIssues { + bound := bindings[ri.IssueID] + if bound == nil || bound.definition == nil { + continue + } + if seenWorkflow[bound.workflow.ID] { + continue + } + seenWorkflow[bound.workflow.ID] = true + + for _, step := range bound.definition.Steps { + for _, entry := range step.Packet { + if len(step.Fanout) > 0 { + // PER HINT, exactly as expansion substitutes per sibling: + // the reachable contracts of a fanout step are its hints'. + for _, hint := range step.Fanout { + add(workflow.SubstitutePacketEntry(entry, hint)) + } + continue + } + add(workflow.SubstitutePacketEntry(entry, step.Executor)) + } + } + } + + for i := 0; i < len(queue); i++ { + for _, include := range packetIncludesOf(scan.roots, queue[i]) { + add(include) + } + } + return needed +} + +// packetIncludesOf reads one closure file's `packet_includes`, best-effort. +// +// BEST-EFFORT IS DELIBERATE. The strict ladder — pinned/unpinned, hash match, +// malformed frontmatter — belongs to resolution (readPinnedPacketFile, +// parsePacketFrontmatter), which refuses at claim/render time with the precise +// message. Refusing here would move that failure into activation for a file +// activation otherwise never opens; skipping here changes nothing the strict +// path would not catch, because a file whose includes could not be read here +// is a file whose render will refuse on the same defect. A ref no root holds +// simply contributes no includes — expansion's unpinned-entry refusal already +// owns that case. +func packetIncludesOf(roots []string, ref string) []string { + for _, root := range roots { + content, err := os.ReadFile(filepath.Join(root, filepath.FromSlash(ref))) + if err != nil { + continue + } + // FIRST ROOT THAT HOLDS IT WINS, matching readPinnedPacketFile — and + // the scan has already refused a ref two roots offer with different + // bytes, so the choice cannot change the answer. + includes, _, perr := parsePacketFrontmatter(ref, string(content)) + if perr != nil { + return nil + } + return includes + } + return nil +} + +// withoutAlreadyPinned drops closure pins whose ref an earlier activation +// already recorded — in the config-relative form, or as a legacy absolute +// walked path that maps onto the same ref. +// +// It exists for RA2: a re-activation inherits the original pin set, and the +// closure recomputed here must only ADD refs a newly-bound workflow needs +// (RA3), never re-list an inherited ref. Re-listing one would carry the +// current file size into the declared-packet index where the inherited pin +// deliberately resolves as present-with-unknown-size — and a re-activation +// must not reject a run that was legal when it started. +func withoutAlreadyPinned(closure []db.Pin, existing []db.Pin, roots []string) []db.Pin { + have := make(map[string]bool, len(existing)) + for _, p := range existing { + if p.Kind != db.PinKindFile { + continue + } + ref := p.Ref + if filepath.IsAbs(ref) { + for _, root := range roots { + if rel, err := filepath.Rel(root, ref); err == nil && + !isOutsideRoot(rel) { + ref = rel + break + } + } + } + have[path.Clean(filepath.ToSlash(ref))] = true + } + + kept := make([]db.Pin, 0, len(closure)) + for _, p := range closure { + if !have[path.Clean(filepath.ToSlash(p.Ref))] { + kept = append(kept, p) + } + } + return kept +} + +// isOutsideRoot reports whether a Rel result escaped the root it was taken +// against. +func isOutsideRoot(rel string) bool { + return rel == ".." || len(rel) >= 3 && rel[:3] == ".."+string(filepath.Separator) +} diff --git a/internal/engine/packet_closure_test.go b/internal/engine/packet_closure_test.go new file mode 100644 index 00000000..081b4a5e --- /dev/null +++ b/internal/engine/packet_closure_test.go @@ -0,0 +1,301 @@ +package engine + +import ( + "database/sql" + "os" + "path/filepath" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// PACKET-CLOSURE PINNING (DKT-581): activation pins the closure the bound +// workflows actually reach — packet entries, their `packet_includes`, and +// policy.toml — not every file under the config roots. 7 of 18 terminal runs +// in the measured week were abandoned over pin drift in corpus files they +// never read; RUN-30's verify-pins named 20 drifted pins of which 11 were +// contracts its workflow never renders. + +// rewriteConfigFile edits a config file in place — the mid-run corpus install +// these tests simulate. +func rewriteConfigFile(t *testing.T, configDir, rel, body string) { + t.Helper() + err := os.WriteFile(filepath.Join(configDir, rel), []byte(body), 0o644) + testsupport.Must(t, err, "rewriting %s: %v", rel, err) +} + +// closureRun activates one issue against a config tree whose workflow declares +// `packet = ["contracts/used.md"]`, with used.md carrying a `packet_includes` +// to fragments/style.md, beside an UNREFERENCED contracts/unused.md and a +// policy.toml. +func closureRun(t *testing.T) (conn *sql.DB, configDir string, runID int) { + t.Helper() + conn, configDir = configRepo(t) + writeConfigFile(t, configDir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/used.md\"]\n") + writeConfigFile(t, configDir, "contracts/used.md", + "---\npacket_includes:\n - fragments/style.md\n---\nthe used contract\n") + writeConfigFile(t, configDir, "fragments/style.md", "the included fragment\n") + writeConfigFile(t, configDir, "contracts/unused.md", "referenced by nothing\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "closure subject", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + return conn, configDir, run.ID +} + +// TestVerifyPinsIgnoresCorpusEditOutsideTheClosure is DKT-581's first +// acceptance criterion, verbatim: a corpus edit to a contract no step in an +// active run references does not fail verify-pins for that run. +func TestVerifyPinsIgnoresCorpusEditOutsideTheClosure(t *testing.T) { + conn, configDir, runID := closureRun(t) + + // The pin set is the closure: the declared contract, its included + // fragment, and policy.toml — and NOT the unreferenced contract. + refs := map[string]bool{} + for _, p := range pinsByKind(t, conn, runID, db.PinKindFile) { + refs[filepath.ToSlash(p.Ref)] = true + } + for _, want := range []string{"contracts/used.md", "fragments/style.md", "policy.toml"} { + if !refs[want] { + t.Errorf("pin refs %v do not include %q; the closure must cover it", refs, want) + } + } + if refs["contracts/unused.md"] { + t.Fatalf("pin refs %v include the unreferenced contract; the corpus edit "+ + "below would then drift a file this run never reads", refs) + } + + // The mid-run corpus install, hitting ONLY the unreferenced file. + rewriteConfigFile(t, configDir, "contracts/unused.md", "REWRITTEN BY THE INSTALL\n") + + report, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if !report.Sound() { + t.Errorf("verify-pins reports drift after an edit outside the closure: %s", + PinReportReason(report)) + } +} + +// TestVerifyPinsStillReportsDriftInsideTheClosure is the second acceptance +// criterion — narrowing the pin set must not silence legitimate detection. A +// directly declared contract AND a fragment reachable only through +// `packet_includes` both still drift. +func TestVerifyPinsStillReportsDriftInsideTheClosure(t *testing.T) { + conn, configDir, runID := closureRun(t) + + rewriteConfigFile(t, configDir, "contracts/used.md", "the install's new contract\n") + rewriteConfigFile(t, configDir, "fragments/style.md", "the install's new fragment\n") + + report, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if report.Changed != 2 { + t.Fatalf("verify-pins reports %d changed pin(s), want 2 (the contract and "+ + "its included fragment): %+v", report.Changed, report.Pins) + } + changed := map[string]bool{} + for _, v := range report.Pins { + if v.Status == PinChanged { + changed[filepath.ToSlash(v.Ref)] = true + } + } + if !changed["contracts/used.md"] { + t.Error("a drifted contract a step's packet directly declares was not reported") + } + if !changed["fragments/style.md"] { + t.Error("a drifted fragment reachable via packet_includes was not reported") + } + + // And repin — the recovery verb — still adopts exactly those two. + outcome, err := RepinRun(conn, runID, "corpus install", nowMS) + testsupport.Must(t, err, "RepinRun: %v", err) + if len(outcome.Repinned) != 2 { + t.Fatalf("repinned %d pin(s), want the 2 drifted ones: %+v", + len(outcome.Repinned), outcome.Repinned) + } + after, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins after repin: %v", err) + if !after.Sound() { + t.Errorf("the run is still unsound after repin: %s", PinReportReason(after)) + } +} + +// closureLoopWorkflowSrc exercises every substitution branch the closure walk +// has: a literal entry, a `{executor}` entry on a FANOUT step (per hint), a +// `{executor}` entry on a plain step (declared executor), a `loop = true` +// body's entry (loop bodies never appear in ordinary expansion but do render +// at loop entry), and a step a false `when` will skip (expansion still +// requires its entries pinned). +const closureLoopWorkflowSrc = ` +[pipeline] +name = "closure-loop" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" +after = [] +packet = ["contracts/implement.md"] + +[[step]] +name = "review" +after = ["implement"] +fanout = ["judge-a", "judge-b"] +emits = "findings" +inputs = ["implement.change-summary"] +packet = ["contracts/{executor}.md"] + +[[step]] +name = "verify" +after = ["review"] +executor = "verify" +emits = "ac-report" +inputs = ["implement.change-summary"] +packet = ["contracts/{executor}.md"] +threshold = { "fix-loop" = "any(status == unmet)" } +max_fix_loops = 1 + +[[step]] +name = "fix" +executor = "fix" +loop = true +emits = "change-summary" +inputs = ["implement.change-summary"] +packet = ["contracts/fix.md"] +after_loop = "review" + +[[step]] +name = "extra" +after = ["implement"] +executor = "extra" +emits = "note" +when = "labels contains never-applied" +packet = ["contracts/extra.md"] +` + +// TestClosureCoversHintsLoopBodiesAndSkippedSteps pins the walk's breadth: +// every declared step contributes its entries — fanout per hint, loop bodies, +// `when`-skipped steps — while a contract no declaration reaches stays out. +func TestClosureCoversHintsLoopBodiesAndSkippedSteps(t *testing.T) { + conn, dir := configRepo(t) + writeConfigFile(t, dir, "workflows/closure-loop.toml", closureLoopWorkflowSrc) + for _, rel := range []string{ + "contracts/implement.md", "contracts/judge-a.md", "contracts/judge-b.md", + "contracts/verify.md", "contracts/fix.md", "contracts/extra.md", + "contracts/unreachable.md", + } { + writeConfigFile(t, dir, rel, "contract body for "+rel+"\n") + } + + issue := createIssue(t, conn, "hints and loops", "a body", "task", nil) + run := startRun(t, conn, issue) + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + refs := map[string]bool{} + for _, p := range pinsByKind(t, conn, run.ID, db.PinKindFile) { + refs[filepath.ToSlash(p.Ref)] = true + } + for _, want := range []string{ + "contracts/implement.md", // literal entry + "contracts/judge-a.md", // fanout hint substitution, first sibling + "contracts/judge-b.md", // fanout hint substitution, second sibling + "contracts/verify.md", // {executor} against the declared executor + "contracts/fix.md", // loop = true body + "contracts/extra.md", // `when`-skipped step: expansion still checks it + } { + if !refs[want] { + t.Errorf("pin refs %v do not include %q", refs, want) + } + } + if refs["contracts/unreachable.md"] { + t.Errorf("pin refs %v include a contract no step declares", refs) + } + if result.PinsFromConfig != 6 { + t.Errorf("pinned %d config files, want the 6 reachable contracts", + result.PinsFromConfig) + } +} + +// TestReactivationPinsANewlyBoundWorkflowsClosure is RA3 under DKT-581: an +// issue added mid-run binds a workflow whose files the FIRST activation's +// closure never reached, so the re-activation must ADD those pins — while +// RA2's inheritance keeps every already-pinned ref exactly where it was, even +// after a mid-run edit. +func TestReactivationPinsANewlyBoundWorkflowsClosure(t *testing.T) { + conn, dir := configRepo(t) + // Both workflows are registered by the FIRST activation's scan (F15: a + // re-activation never re-scans), but only auto-dev binds then, so only + // its contract is in the first closure. + writeConfigFile(t, dir, "workflows/auto-dev.toml", + autoWorkflowSrc+"packet = [\"contracts/a.md\"]\n") + other := ` +[pipeline] +name = "auto-bug" +version = 1 + +[match] +kind = ["bug"] + +[[step]] +name = "diagnose" +executor = "w" +emits = "out" +after = [] +packet = ["contracts/b.md"] +` + writeConfigFile(t, dir, "workflows/auto-bug.toml", other) + writeConfigFile(t, dir, "contracts/a.md", "contract a, original\n") + writeConfigFile(t, dir, "contracts/b.md", "contract b\n") + + first := createIssue(t, conn, "first", "a body", "task", nil) + run := startRun(t, conn, first) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "the first activation: %v", err) + + refs := map[string]string{} + for _, p := range pinsByKind(t, conn, run.ID, db.PinKindFile) { + refs[filepath.ToSlash(p.Ref)] = p.SHA256 + } + if _, ok := refs["contracts/b.md"]; ok { + t.Fatal("premise: the unbound workflow's contract must not be in the " + + "first activation's closure") + } + pinnedA := refs["contracts/a.md"] + if pinnedA == "" { + t.Fatal("premise: the bound workflow's contract must be pinned") + } + + // The mid-run corpus edit RA2 makes a non-event. + rewriteConfigFile(t, dir, "contracts/a.md", "contract a, edited mid-run\n") + + second := createIssue(t, conn, "second", "a body", "bug", nil) + err = db.AddRunIssue(conn, run.ID, second) + testsupport.Must(t, err, "adding the second issue: %v", err) + + result, err := activate(conn, run.ID) + testsupport.Must(t, err, "the re-activation: %v", err) + if !result.Reactivation { + t.Fatal("premise: the second activation must be a re-activation") + } + + after := map[string]string{} + for _, p := range pinsByKind(t, conn, run.ID, db.PinKindFile) { + after[filepath.ToSlash(p.Ref)] = p.SHA256 + } + if after["contracts/b.md"] == "" { + t.Error("RA3: the newly-bound workflow's contract was not pinned, so its " + + "steps could never render") + } + if after["contracts/a.md"] != pinnedA { + t.Errorf("RA2: the inherited pin moved from %s to %s; a mid-run edit must "+ + "be a non-event", pinnedA, after["contracts/a.md"]) + } +} diff --git a/internal/engine/packet_test.go b/internal/engine/packet_test.go index b0c5a7e5..1846d57e 100644 --- a/internal/engine/packet_test.go +++ b/internal/engine/packet_test.go @@ -172,7 +172,7 @@ func TestPacketIncludesAreOneLevelDeep(t *testing.T) { "---\npacket_includes:\n - fragments/deeper.md\n---\nMIDDLE\n") writeFixture(t, root, "fragments/deeper.md", "DEEPEST\n") - files, err := resolvePacketFiles(testPinSet(t, root, + files, err := resolvePacketFiles("RUN-1", testPinSet(t, root, "checklists/proofing.md", "fragments/house-style.md", "fragments/deeper.md"), []string{root}, []string{"checklists/proofing.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) @@ -198,7 +198,7 @@ func TestPacketIncludesDedupe(t *testing.T) { "---\npacket_includes:\n - fragments/shared.md\n---\nB\n") writeFixture(t, root, "fragments/shared.md", "SHARED\n") - files, err := resolvePacketFiles(testPinSet(t, root, + files, err := resolvePacketFiles("RUN-1", testPinSet(t, root, "checklists/a.md", "checklists/b.md", "fragments/shared.md"), []string{root}, []string{"checklists/a.md", "checklists/b.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) @@ -234,7 +234,8 @@ func TestPacketResolutionLadder(t *testing.T) { root := t.TempDir() writeFixture(t, root, "checklists/a.md", "BODY\n") files, err := resolvePacketFiles( - testPinSet(t, root, "checklists/a.md"), []string{root}, []string{"checklists/a.md"}) + "RUN-1", testPinSet(t, root, "checklists/a.md"), + []string{root}, []string{"checklists/a.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) if len(files) != 1 || strings.TrimSpace(files[0].Body) != "BODY" { t.Errorf("files = %+v, want the file's bytes inlined", files) @@ -252,7 +253,8 @@ func TestPacketResolutionLadder(t *testing.T) { writeFixture(t, root, "checklists/a.md", "EDITED\n") - _, err := resolvePacketFiles(pins, []string{root}, []string{"checklists/a.md"}) + _, err := resolvePacketFiles( + "RUN-1", pins, []string{root}, []string{"checklists/a.md"}) if err == nil { t.Fatal("an edited file resolved, want CONFLICT") } @@ -271,7 +273,7 @@ func TestPacketResolutionLadder(t *testing.T) { err := os.Remove(filepath.Join(root, "checklists/a.md")) testsupport.Must(t, err, "removing the fixture: %v", err) - _, err = resolvePacketFiles(pins, []string{root}, []string{"checklists/a.md"}) + _, err = resolvePacketFiles("RUN-1", pins, []string{root}, []string{"checklists/a.md"}) if code, _ := CodeOf(err); code != CodeNotFound { t.Errorf("code = %q (err %v), want %q", code, err, CodeNotFound) } @@ -281,7 +283,9 @@ func TestPacketResolutionLadder(t *testing.T) { root := t.TempDir() writeFixture(t, root, "checklists/a.md", "BODY\n") - _, err := resolvePacketFiles(map[string]string{}, []string{root}, []string{"checklists/a.md"}) + _, err := resolvePacketFiles( + "RUN-1", map[string]string{}, []string{root}, + []string{"checklists/a.md"}) if err == nil { t.Fatal("an unpinned file was read, want a refusal") } @@ -296,7 +300,8 @@ func TestPacketResolutionLadder(t *testing.T) { "---\npacket_includes:\n - fragments/missing.md\n---\nA\n") _, err := resolvePacketFiles( - testPinSet(t, root, "checklists/a.md"), []string{root}, []string{"checklists/a.md"}) + "RUN-1", testPinSet(t, root, "checklists/a.md"), + []string{root}, []string{"checklists/a.md"}) if err == nil { t.Fatal("a dangling include was skipped — that is DKT-70's failure signature") } @@ -316,7 +321,8 @@ func TestPacketResolutionIsDeterministic(t *testing.T) { var first string for i := 0; i < 16; i++ { - files, err := resolvePacketFiles(pins, []string{root}, []string{"checklists/a.md"}) + files, err := resolvePacketFiles( + "RUN-1", pins, []string{root}, []string{"checklists/a.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) var b strings.Builder for _, f := range files { diff --git a/internal/engine/passfloor.go b/internal/engine/passfloor.go new file mode 100644 index 00000000..a3de66b9 --- /dev/null +++ b/internal/engine/passfloor.go @@ -0,0 +1,86 @@ +package engine + +import ( + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// The `pass_floor` exit bar (DKT-870): a step whose routing resolved to `pass` +// while its own recorded payload still holds work at or above a declared +// position does not exit — it parks, and an operator decides. +// +// It exists because "converged" in the ledger sometimes meant "dispositioned": +// a threshold reads whatever field its author chose, and a routing step's +// self-reported disposition can contradict the evidence it recorded beside it. +// RUN-58's reconcile@1 routed `pass` and the loop exited with all 16 clusters +// open, SIX at the order's high position, none held and none operator-resolved +// — including the fail-open siblings of the class round 0 had closed. Nothing +// engine-visible stood between that pass and issue completion. +// +// THE FLOOR IS THE AUTHOR'S, AND EVERY VALUE IS COMPARED BY POSITION. `field` +// and `at` are opaque tokens positioned in the step's PINNED schema order — +// `route_at`'s exact discipline (aggregate.go) — so core acquires no opinion +// about severities: a workflow whose order ranks ripeness or confidence gets +// the identical check. `held` and `operator_resolved` are the engine's own +// packaging vocabulary (§7.6), and elements carrying either are exempt because +// both already route through a decision channel: a held cluster gates the step +// for an operator, and a resolved one records the operator's acceptance — the +// approved-hold resume must not re-park on the decision it just received. +type passFloorResult struct { + // Standing counts the payload elements at or above the floor with neither + // exemption — the number the park's reason names. + Standing int +} + +// passFloorStanding counts the elements a `pass` would leave standing above +// the declared floor, or 0 when the floor is absent or cannot measure. +// +// IT FAILS TOWARD THE PASS, the direction every completion-side guard here +// fails (roundMovedNothing, DKT-588's hand-back): a nil resolver (V37 makes +// this unreachable through `workflow register`; reachable from a restored +// database), a floor value outside the declared order, an element without the +// field, and an element whose value has no position are all absent +// measurements, never breaches — the guard must not act on evidence it cannot +// read, and overriding a routing on a guess would be a silent misroute wearing +// a park's clothes. +func passFloorStanding( + floor *workflow.PassFloor, payloads []map[string]any, order OrderResolver, +) passFloorResult { + var out passFloorResult + if floor == nil || order == nil { + return out + } + want, ok := order.Position(floor.Field, floor.At) + if !ok { + return out + } + for _, element := range payloads { + if flagged(element, KeyHeld) || flagged(element, KeyOperatorResolved) { + continue + } + raw, present := element[floor.Field] + if !present || raw == nil { + continue + } + value, err := normalizeScalar(raw) + if err != nil { + continue // A composite value has no position to compare. + } + got, ok := order.Position(floor.Field, value) + if !ok { + continue + } + if got >= want { + out.Standing++ + } + } + return out +} + +// flagged reads one of the aggregate's boolean packaging keys off an element. +// Only an explicit JSON `true` counts: the keys are the engine's own output +// vocabulary (§7.6), absent on payloads no aggregate produced, and absence of +// a decision is not a decision. +func flagged(element map[string]any, key string) bool { + v, _ := element[key].(bool) + return v +} diff --git a/internal/engine/pending_closure.go b/internal/engine/pending_closure.go new file mode 100644 index 00000000..0a0450c9 --- /dev/null +++ b/internal/engine/pending_closure.go @@ -0,0 +1,302 @@ +package engine + +import ( + "database/sql" + "errors" + "fmt" + "os" + "path" + "path/filepath" + "sort" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// PENDING PACKET CLOSURE (DKT-582). +// +// DKT-581 narrowed the pin set at ACTIVATION to the closure the bound workflows +// reach. This is the same walk asked of a run MID-FLIGHT, and about a strictly +// smaller set: not "what could this run ever read" but "what can this run still +// read from here" — the packet closure of its NON-TERMINAL steps. +// +// It exists for repin's drop disposition. A ref that no longer resolves at all +// has no bytes to adopt, so the only safe question is whether anything left to +// run would ever open it; if nothing would, retiring the pin costs the run +// nothing, and refusing costs it every remaining step. RUN-42 and RUN-36 both +// died on refs deleted from the corpus that no pending step named. +// +// THE WALK IS THE SAME ONE, SOURCED DIFFERENTLY. Activation walks declared +// steps because no step rows exist yet; here the step ROWS exist and carry the +// executor hint expansion already substituted, so a fanout sibling contributes +// its own hint's contract rather than all of them — exactly what stepPacketFiles +// re-derives at render time, and for the same reason (one fact, one source). +// +// TWO THINGS ARE DELIBERATELY OVER-COUNTED, because a false "still referenced" +// only refuses a drop while a false "unreferenced" would wedge a later render: +// +// - an issue whose phase has NOT been expanded contributes its bound +// definition's WHOLE declared closure, per DKT-581's substitution rules — +// its step rows do not exist yet, so there is nothing narrower to ask; +// - `policy.toml` is always referenced. The harness resolves policy from it +// rather than any step's packet, so it appears in no `packet` list and no +// step-sourced walk could ever find it. +type pendingClosure map[string]*closureReach + +// closureReach is what reaches one ref: the pending steps (and unexpanded +// phases) whose packets can still open it, and the closure FILES whose +// `packet_includes` name it. +// +// The two are kept apart because they answer different questions. Repin's drop +// disposition asks "would anything left to run open this", which only the steps +// answer. DKT-821's closure check asks "who wrote the reference", and a fragment +// reached three includes deep is named by no step's packet — the file that +// includes it is the thing an operator edits or re-pins. +type closureReach struct { + steps []string + includes []string +} + +// referencedBy returns the pending steps (and unexpanded phases) that can still +// reach a ref, in a stable order, or nil when nothing can. +func (c pendingClosure) referencedBy(ref string) []string { + return sortedCopy(c[normalizePinRef(ref)].stepsOf()) +} + +// includedBy returns the closure files whose `packet_includes` name a ref, in a +// stable order, or nil when a step's own `packet` entry is what reaches it. +func (c pendingClosure) includedBy(ref string) []string { + return sortedCopy(c[normalizePinRef(ref)].includesOf()) +} + +func (r *closureReach) stepsOf() []string { + if r == nil { + return nil + } + return r.steps +} + +func (r *closureReach) includesOf() []string { + if r == nil { + return nil + } + return r.includes +} + +// sortedCopy is the stable-order form every reacher list is handed out in — +// nil for empty, so a caller can gate on length alone. +func sortedCopy(in []string) []string { + if len(in) == 0 { + return nil + } + out := append([]string(nil), in...) + sort.Strings(out) + return out +} + +// normalizePinRef is the one spelling refs are compared in — the config-relative +// slash form packet entries, pin rows, and an operator's `--drop` argument all +// have to agree on. +func normalizePinRef(ref string) string { + return path.Clean(filepath.ToSlash(ref)) +} + +// pendingPacketClosure computes the refs a run's non-terminal work can still +// read, each mapped to what reaches it. +func pendingPacketClosure(conn *sql.DB, runID int, roots []string) (pendingClosure, error) { + steps, err := db.ListRunSteps(conn, runID) + if err != nil { + return nil, err + } + defs, err := StepDefinitions(conn, runID) + if err != nil { + return nil, err + } + runIssues, err := db.ListRunIssues(conn, runID) + if err != nil { + return nil, err + } + + closure := pendingClosure{} + // The worklist carries every ref whose `packet_includes` still needs + // walking; membership in `closure` doubles as the seen-set, so an include + // cycle terminates and a diamond is walked once. + var queue []string + add := func(ref, by string) *closureReach { + ref = normalizePinRef(ref) + if ref == "" || ref == "." { + return nil + } + reach, seen := closure[ref] + if !seen { + reach = &closureReach{} + closure[ref] = reach + queue = append(queue, ref) + } + reach.steps = appendOnce(reach.steps, by) + return reach + } + + add(policyPinRef, "the harness (policy)") + + for _, s := range steps { + if db.StepTerminal(s.Status) { + continue + } + spec := stepSpec(defs, s, holdTally{}) + if spec == nil { + continue + } + for _, entry := range spec.Packet { + add(workflow.SubstitutePacketEntry(entry, s.Executor), s.Instance) + } + } + + for _, ri := range runIssues { + if ri.Expanded() || ri.WorkflowID == nil { + continue + } + def := defs[*ri.WorkflowID] + if def == nil { + continue + } + by := fmt.Sprintf("issue %d's unexpanded phase", ri.IssueID) + for _, step := range def.Steps { + for _, entry := range step.Packet { + if len(step.Fanout) > 0 { + for _, hint := range step.Fanout { + add(workflow.SubstitutePacketEntry(entry, hint), by) + } + continue + } + add(workflow.SubstitutePacketEntry(entry, step.Executor), by) + } + } + } + + for i := 0; i < len(queue); i++ { + ref := queue[i] + // Snapshot the reachers before walking: `add` appends to the map, and + // a self-include would otherwise range a slice it is extending. + reachers := append([]string(nil), closure[ref].steps...) + for _, include := range packetIncludesOf(roots, ref) { + for _, by := range reachers { + // The INCLUDE EDGE is recorded beside the step that inherits + // it: the step says whether the ref still matters, the file + // says who asked for it, and DKT-821 needs both in one + // sentence. + if reach := add(include, by); reach != nil { + reach.includes = appendOnce(reach.includes, ref) + } + } + } + } + return closure, nil +} + +// appendOnce keeps a reacher list a set without paying for a map per ref — +// these lists are single-digit in every real closure. +func appendOnce(list []string, add string) []string { + if add == "" { + return list + } + for _, have := range list { + if have == add { + return list + } + } + return append(list, add) +} + +// UNPINNED CLOSURE REFS — the one walk two verbs ask (DKT-805, DKT-821). +// +// Repin asks it to decide what to CREATE when it adopts new bytes; verify-pins +// asks it to decide what to REPORT when it is handed a run whose pins all +// match. Both questions are "what does this closure reach that this pin set +// does not hold", and a second implementation of it would be a second answer: +// RUN-59 is exactly the run where the repin verb and the read verb disagreed, +// one pinning nothing and the other calling the result healthy. +type unpinnedRef struct { + ref string + requiredBy []string + includedBy []string + // path is the first config root that holds the file, empty when none does. + path string + // sha256 is the file's current disk bytes, empty when it does not resolve. + sha256 string + // readErr is set when the file IS there and could not be read — a + // permission to fix, never an absence. + readErr error +} + +// unpinnedClosureRefs returns every ref the pending closure reaches that the +// run's pin set does not cover, in ref order, each resolved against the roots. +// +// `policy.toml` is excluded, for the reason repin's additions exclude it: the +// HARNESS reads it, no step's packet ever renders it, so its absence from a pin +// set cannot make a step unrenderable — and treating it as a hole would have +// every run that chose not to pin policy report a hole it does not have. +func unpinnedClosureRefs( + pins []PinVerdict, closure pendingClosure, roots []string, +) []unpinnedRef { + // The run's held file refs, in the one spelling refs are compared in. A + // legacy pin that recorded the full walked path maps onto its + // config-relative ref, the same way withoutAlreadyPinned maps it for RA3 — + // treating an inherited ref as unheld is the mistake both walks avoid. + have := make(map[string]bool, len(pins)) + for _, v := range pins { + if v.Kind != db.PinKindFile { + continue + } + ref := v.Ref + if filepath.IsAbs(ref) { + for _, root := range roots { + if rel, err := filepath.Rel(root, ref); err == nil && !isOutsideRoot(rel) { + ref = rel + break + } + } + } + have[normalizePinRef(ref)] = true + } + + // Sorted, so the pins repin adds and the rows verify-pins prints land in a + // deterministic order — the same golden-stability discipline the pin report + // follows. + refs := make([]string, 0, len(closure)) + for ref := range closure { + refs = append(refs, ref) + } + sort.Strings(refs) + + var out []unpinnedRef + for _, ref := range refs { + if have[ref] || ref == policyPinRef { + continue + } + u := unpinnedRef{ + ref: ref, + requiredBy: closure.referencedBy(ref), + includedBy: closure.includedBy(ref), + } + for _, root := range roots { + full := filepath.Join(root, filepath.FromSlash(ref)) + content, err := os.ReadFile(full) + if errors.Is(err, os.ErrNotExist) { + continue + } + // FIRST ROOT THAT HOLDS IT WINS, matching readPinnedPacketFile — + // falling through to a later root here would describe a stale copy + // of the file the closure actually resolves to. + u.path = full + if err != nil { + u.readErr = err + } else { + u.sha256 = workflow.SHA256(content) + } + break + } + out = append(out, u) + } + return out +} diff --git a/internal/engine/pin_staleness.go b/internal/engine/pin_staleness.go new file mode 100644 index 00000000..2a9cb38a --- /dev/null +++ b/internal/engine/pin_staleness.go @@ -0,0 +1,408 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "fmt" + "sort" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// THE PIN SECTIONS OF `run report` (DKT-594). +// +// Two questions an analyst had to answer with git and a hand-join, on a corpus +// that took 41 commits in 4.3 days: +// +// 1. HOW FAR HAS THE CORPUS MOVED SINCE THIS RUN FROZE. Every reader of +// RUN-32's report had to go to git to learn that its `ui-change@8` was five +// registered versions behind before they would trust a finding from it. The +// report published the run's pins nowhere and the registry's head nowhere, +// so the subtraction was not available from any read verb. +// +// 2. WHICH AGREEMENT DID THIS STEP ACTUALLY RUN UNDER. After a mid-run repin +// the `pins` table says what the REMAINING steps will read, and says +// nothing about the bytes the completed ones consumed. RUN-39's post-mortem +// recovered that by correlating repin event seqs (5375/5376) against step +// ids (STEP-1350/1353) by hand. +// +// NO NEW PERSISTED STATE ANSWERS EITHER. The first is a subtraction over +// `pins.ref` (a workflow pin's ref IS `name@version` — activation writes +// `bound.workflow.Ref()`) and the `workflows` table. The second is exactly the +// reconstruction repin.go's own architecture note reserves for the event log: +// "the event log is already the package's history mechanism … a parallel +// pin-history table would be a second source of the same fact", and +// EventRunRepinned's comment already states the reconstruction rule — "steps +// recorded before this event's seq worked under `old_sha256`, steps claimed +// after it work under `new_sha256`". This file performs that reconstruction +// instead of asking an operator to. + +// PinnedWorkflowStaleness is criterion 1: one pinned workflow, and how far the +// registry has moved past it. +// +// It covers WORKFLOW pins only. A `schema` pin is also a registered +// `name@version` and could carry the same diff, but the issue asks about +// workflows and a schema's version moves for different reasons; a FILE pin +// (a contract, a fragment, `policy.toml`) has no version at all — its "11 +// versions stale" in the retro means corpus git commits, which live outside +// this store and which core cannot count. `verify-pins` is the verb that +// answers the file question, by hash. +type PinnedWorkflowStaleness struct { + // Ref is the pin's ref verbatim — `name@version`, what the run froze. + Ref string `json:"ref"` + Name string `json:"name"` + // PinnedVersion is the version this run is expanded from. + PinnedVersion int `json:"pinned_version"` + // CurrentVersion is the highest registered version of this name that STILL + // BINDS — the version a run activated now would pin. Zero when no version + // of the name binds any more (every one retired, or the name is gone from + // this project's registry), which is a fact worth publishing rather than + // rounding to the pinned version. + CurrentVersion int `json:"current_version"` + // Behind is THE STALENESS COUNT: how many binding versions are registered + // STRICTLY ABOVE the pinned one. + // + // A count of rows rather than `current - pinned`, and never negative. The + // subtraction would go negative the moment a higher version is retired + // while a run holds it, and would over-count where a name's versions are + // not contiguous — a registry with @1 and @8 and nothing between is one + // version of advance, not seven. + Behind int `json:"behind"` +} + +// PinEpoch is criterion 2's timeline: one segment of a run's life during which +// its pin agreement did not move. +// +// Epoch 1 is ACTIVATION — the agreement the run froze. Every later epoch is one +// `run repin`, which is the only thing in the engine that rewrites a `pins` row +// (the two statements in repin.go are the whole set; activation's own writes are +// `INSERT OR IGNORE`, so re-activation ADDS refs and never moves one). +type PinEpoch struct { + // Epoch numbers from 1, so a zero on a step means "no epoch", never + // "the first one". + Epoch int `json:"epoch"` + // FromSeq is the event seq this agreement took effect at — the + // `run-activated` event for epoch 1, the repin's first event for the rest. + // Zero for epoch 1 on a run whose activation event has been pruned away. + FromSeq int64 `json:"from_seq"` + AtMS int64 `json:"at_ms,omitempty"` + // Origin is `activation` or `repin`. + Origin string `json:"origin"` + // Reason is the operator's `--reason`, verbatim, on a repin epoch. + Reason string `json:"reason,omitempty"` + // Changes is what MOVED entering this epoch, one entry per changed ref, + // carrying both hashes exactly as the event recorded them. Empty on epoch 1: + // nothing moved, the run froze. + Changes []RepinChange `json:"changes,omitempty"` +} + +// PinEpochOrigin values. +const ( + PinEpochActivation = "activation" + PinEpochRepin = "repin" +) + +// pinnedWorkflowStaleness answers criterion 1 for a whole run. +// +// It reads through the POOL, before the report's snapshot opens, for the reason +// LoadRunReport's own comment gives: internal/db caps the pool at one +// connection, so a pool read from inside the open transaction deadlocks +// permanently. Nothing here needs the snapshot — a pin set and a registry head +// read a moment early are the same two numbers. +func pinnedWorkflowStaleness( + conn *sql.DB, projectID, runID int, +) ([]PinnedWorkflowStaleness, error) { + pins, err := db.ListPins(conn, runID) + if err != nil { + return nil, err + } + + type pinned struct { + ref string + name string + version int + } + var wanted []pinned + names := make([]string, 0, len(pins)) + seen := make(map[string]bool, len(pins)) + for _, p := range pins { + if p.Kind != db.PinKindWorkflow { + continue + } + name, version, err := workflow.ParsePayloadRef(p.Ref) + if err != nil { + // A ref that does not parse as `name@version` is not a workflow pin + // this diff can speak about. It is REPORTED BY `verify-pins` (which + // calls it missing) and skipped here rather than failing the whole + // report: R10 says the report works on a run in any state, and a + // rollup that refused would be unavailable during exactly the + // incident that produced the odd row. + continue + } + wanted = append(wanted, pinned{ref: p.Ref, name: name, version: version}) + if !seen[name] { + seen[name] = true + names = append(names, name) + } + } + if len(wanted) == 0 { + return nil, nil + } + + registered, err := db.WorkflowVersionsFor(conn, projectID, names) + if err != nil { + return nil, err + } + binding := make(map[string][]int, len(names)) + for _, v := range registered { + if v.Binds { + binding[v.Name] = append(binding[v.Name], v.Version) + } + } + + out := make([]PinnedWorkflowStaleness, 0, len(wanted)) + for _, p := range wanted { + row := PinnedWorkflowStaleness{ + Ref: p.ref, Name: p.name, PinnedVersion: p.version, + } + for _, v := range binding[p.name] { + if v > row.CurrentVersion { + row.CurrentVersion = v + } + if v > p.version { + row.Behind++ + } + } + out = append(out, row) + } + // A TOTAL order (R9), by the ref the run holds, so two reports of one + // unchanged run are byte-identical. + sort.Slice(out, func(i, j int) bool { return out[i].Ref < out[j].Ref }) + return out, nil +} + +// runPinEpochsTx reconstructs the run's pin-agreement timeline from the event +// log, newest last. +// +// It returns NOTHING when the run never repinned. One epoch is not a timeline: +// every step of such a run ran under the agreement the `pins` table still +// states, the reader needs no reconciliation, and publishing an `epoch 1` on +// every step of every report would be a column of constants. +// +// GROUPING. A repin writes one event PER CHANGED REF, all in its own +// transaction, so three moved contracts are three events and ONE boundary. +// Events of a single transaction are consecutive in `seq` (SQLite serializes +// writers, so no other transaction's insert can land between them) and share +// the `nowMS` their caller passed, so a maximal contiguous same-`at_ms` run of +// this run's repin events is that transaction. The one shape this cannot split +// is two repin transactions that committed in the same MILLISECOND with nothing +// between them — reachable only with a frozen clock, and harmless where it is: +// no step can have executed inside a zero-length interval, so no step's epoch +// moves, only the count of boundaries. +func runPinEpochsTx(tx *sql.Tx, runID int) ([]PinEpoch, error) { + rows, err := tx.Query( + `SELECT seq, at_ms, kind, data FROM events + WHERE run_id = ? AND kind IN (?, ?) ORDER BY seq`, + runID, EventRunActivated, EventRunRepinned) + if err != nil { + return nil, fmt.Errorf("reading the run's pin history: %w", err) + } + type eventRow struct { + seq int64 + atMS int64 + kind string + data string + } + events, err := scanTxRows(rows, func(r *sql.Rows) (eventRow, error) { + var e eventRow + if err := r.Scan(&e.seq, &e.atMS, &e.kind, &e.data); err != nil { + return eventRow{}, fmt.Errorf("reading a pin-history event: %w", err) + } + return e, nil + }) + if err != nil { + return nil, err + } + + activation := PinEpoch{Epoch: 1, Origin: PinEpochActivation} + var epochs []PinEpoch + for _, e := range events { + if e.kind == EventRunActivated { + // THE FIRST activation. A re-activation writes a second one, and it + // is not a boundary: RA2 INHERITS the pin set, and the only rows a + // re-activation writes are `INSERT OR IGNORE` additions for refs the + // run did not hold — no existing pin's bytes move, so no completed + // step's provenance changes. + if activation.FromSeq == 0 { + activation.FromSeq, activation.AtMS = e.seq, e.atMS + } + continue + } + var payload struct { + RepinChange + Reason string `json:"reason"` + } + if err := json.Unmarshal([]byte(e.data), &payload); err != nil { + // Core's own payload, but a malformed one costs the CHANGE LINE and + // not the boundary: that this run repinned here is the event's + // existence, and a report that refused over an unparseable row would + // be unavailable during the incident that produced it. + payload = struct { + RepinChange + Reason string `json:"reason"` + }{} + } + last := len(epochs) - 1 + if last >= 0 && epochs[last].AtMS == e.atMS && + e.seq == epochs[last].FromSeq+int64(len(epochs[last].Changes)) { + epochs[last].Changes = append(epochs[last].Changes, payload.RepinChange) + continue + } + epochs = append(epochs, PinEpoch{ + Epoch: len(epochs) + 2, // epoch 1 is activation + FromSeq: e.seq, + AtMS: e.atMS, + Origin: PinEpochRepin, + Reason: payload.Reason, + Changes: []RepinChange{payload.RepinChange}, + }) + } + if len(epochs) == 0 { + return nil, nil + } + return append([]PinEpoch{activation}, epochs...), nil +} + +// stepEpochAnchorKinds are the events that say A STEP'S RECORDED WORK HAPPENED +// HERE — the seq an epoch lookup is keyed on. +// +// `step-claimed` is the load-bearing one: a claim is where the packet is +// rendered, and repin's quiescence guard refuses while ANY step is claimed, so +// a claim and its completion can never straddle a boundary. The terminal kinds +// carry the steps that never get claimed at all — a vote step (permanently +// attempt 0), a skipped one, one a cascade terminated. +// +// Deliberately EXCLUDED: `step-ready` (a step can be ready for an epoch and run +// in the next), `step-heartbeat` and `lease-reaped` (a reaped attempt recorded +// nothing, and the step's re-claim is the anchor), and `step-annotated` (it +// runs AFTER completion, potentially after a later repin, and annotating a +// record is not consuming bytes). +var stepEpochAnchorKinds = map[string]bool{ + EventStepClaimed: true, + EventStepRecorded: true, + EventGateRecorded: true, + EventStepRouted: true, + EventStepFailed: true, + EventStepSkipped: true, + EventStepSuperseded: true, + EventStepResolved: true, + EventStepApproved: true, + EventStepRejected: true, + EventVoteTallied: true, +} + +// stepPinEpochsTx maps each step id onto the epoch its recorded work ran under. +// +// The LAST anchor wins. A step claimed, reaped, and re-claimed after a repin +// consumed the NEW bytes, and the attempt that was thrown away is not the one +// the report is describing. +func stepPinEpochsTx(tx *sql.Tx, runID int, epochs []PinEpoch) (map[int]int, error) { + if len(epochs) < 2 { + return nil, nil + } + rows, err := tx.Query( + `SELECT step_id, kind, seq FROM events + WHERE run_id = ? AND step_id IS NOT NULL ORDER BY seq`, runID) + if err != nil { + return nil, fmt.Errorf("reading the run's step events: %w", err) + } + defer rows.Close() + + anchors := make(map[int]int64) + for rows.Next() { + var ( + stepID int + kind string + seq int64 + ) + if err := rows.Scan(&stepID, &kind, &seq); err != nil { + return nil, fmt.Errorf("reading a step event: %w", err) + } + if stepEpochAnchorKinds[kind] && seq > anchors[stepID] { + anchors[stepID] = seq + } + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("reading the run's step events: %w", err) + } + + out := make(map[int]int, len(anchors)) + for stepID, seq := range anchors { + out[stepID] = epochAt(epochs, seq) + } + return out, nil +} + +// epochAt is the lookup EventRunRepinned's own comment describes: the last +// agreement that took effect at or before this seq. +func epochAt(epochs []PinEpoch, seq int64) int { + epoch := epochs[0].Epoch + for _, e := range epochs { + if seq >= e.FromSeq { + epoch = e.Epoch + } + } + return epoch +} + +// stepReportsAnEpoch decides whether a step's effective status makes "the +// agreement it ran under" a fact rather than a forecast. +// +// A `pending` or `ready` step has not consumed anything: whatever its history +// (a reaped claim leaves anchor events behind), the bytes it will read are the +// ones the `pins` table holds when it is finally claimed, and stamping it with +// the epoch of an attempt that recorded nothing would answer a question about +// the future with a fact about the past. +func stepReportsAnEpoch(status string) bool { + switch status { + case db.StepClaimed, db.StepDone, db.StepWaitingHuman, + db.StepSkipped, db.StepSuperseded, db.StepFailedRouted: + return true + } + return false +} + +// annotatePinEpochs stamps each attempt row with the epoch its work ran under. +// +// It runs INSIDE the report's snapshot, like annotateVoteOutcomes and for the +// same two reasons: the one-connection pool makes a pool read from in here a +// permanent deadlock, and the annotation then describes the same instant as the +// statuses beside it. +func annotatePinEpochs( + tx *sql.Tx, runID int, attempts []StepAttempt, +) ([]PinEpoch, error) { + epochs, err := runPinEpochsTx(tx, runID) + if err != nil { + return nil, err + } + if len(epochs) < 2 { + return nil, nil + } + byStep, err := stepPinEpochsTx(tx, runID, epochs) + if err != nil { + return nil, err + } + for i := range attempts { + if !stepReportsAnEpoch(attempts[i].Status) { + continue + } + id, err := model.ParseStepID(attempts[i].Step) + if err != nil { + continue + } + attempts[i].PinEpoch = byStep[id] + } + return epochs, nil +} diff --git a/internal/engine/pregate.go b/internal/engine/pregate.go index 0b6ccbc1..5d9e9808 100644 --- a/internal/engine/pregate.go +++ b/internal/engine/pregate.go @@ -102,11 +102,11 @@ func resolvedTargetFor( if spec == nil || len(spec.Inputs) == 0 { return "", "", nil } - issue, err := contextIssue(tx, step.RunID, step.IssueID) + issue, linked, err := contextIssue(tx, step.RunID, step.IssueID) if err != nil { return "", "", err } - inputs, err := resolveInputs(tx, sched, step, spec, issue.BodySnapshot, artifacts) + inputs, err := resolveInputs(tx, sched, step, spec, issue.BodySnapshot, artifacts, linked) if err != nil { return "", "", err } diff --git a/internal/engine/pregate_test.go b/internal/engine/pregate_test.go index 199e2c53..92d181e2 100644 --- a/internal/engine/pregate_test.go +++ b/internal/engine/pregate_test.go @@ -400,6 +400,169 @@ func TestUnmatchedPreGateKeepsBothItsEvents(t *testing.T) { } } +// TestPreGateEventsMarkTheVerdictThatRoutedNothing is DKT-862's events half. +// +// `gate-recorded ... detail=ac-commands exit=2 verdict=fail` said the same +// thing whether the failure BLOCKED the step or was an advisory input to it — +// and a pre-gate never routes: §11.1 runs it at claim, PG4 keeps it out of the +// saga's verdict. On RUN-61 three pre-gate failures sat in the feed beside the +// `step-routed pass` that contradicted them, and a conductor nearly reported a +// fix round as burned on one. +// +// The assertion is on the STORED PAYLOAD, because that is what both readers +// see: `events list` renders `data` as sorted key=value pairs and interprets +// nothing, and `--json` hands the object straight to a program. +func TestPreGateEventsMarkTheVerdictThatRoutedNothing(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + repoRoot := t.TempDir() + + // A pre-gate that matches and FAILS — RUN-61's shape. `/usr/bin/false` + // exits 1, so the row is a `fail` that routed nothing. + argv := []string{"/usr/bin/false"} + e := testEngine() + runner := NewExecRunner(testRepoPaths(repoRoot)) + runner.LoadStore = sandboxTrust(t, trust.Entry{ + Name: "ac-commands", Argv: argv, ArgvSHA256: trust.ArgvSHA256(argv), + Repo: mustResolve(repoRoot), + }) + e.Gates = runner + + stepID := advanceToVerify(t, conn, e) + _, err := e.ClaimStepWithGates(conn, stepID, ClaimOptions{ + Owner: "w", NowMS: nowMS, + }) + testsupport.Must(t, err, "an advisory pre-gate failure must not refuse the claim: %v", err) + + data := gateRecordedData(t, conn, stepID, "ac-commands") + if data["verdict"] != VerdictFail { + t.Fatalf("the pre-gate's gate-recorded verdict = %v, want %q — this "+ + "case is about a FAILING advisory gate", data["verdict"], VerdictFail) + } + if data["pre"] != true { + t.Errorf("gate-recorded for a pre-gate carries pre = %v, want true; "+ + "payload %v renders as a blocking failure in `events list`", + data["pre"], data) + } + + // The marker is not bought by dropping what was already there (DKT-63). + if data["detail"] != "ac-commands" { + t.Errorf("the gate name left `detail`: %v", data) + } + + // AND a BLOCKING gate stays unmarked. `implement@0`'s gates run through + // advanceToVerify's pass-through runner and route the step for real, so + // their absence of a `pre` key is what makes the pre-gate's presence mean + // something rather than being a field every gate event carries. + implementID := stepIDByInstance(t, conn, "implement@0") + for _, blocking := range []string{"build", "tests"} { + // gateRecordedData FATALS on a missing event, which is what keeps this + // half from passing vacuously: a payload that carries no `pre` key + // because it was never written proves nothing. + if _, ok := gateRecordedData(t, conn, implementID, blocking)["pre"]; ok { + t.Errorf("gate-recorded for the blocking gate %q carries a pre "+ + "marker; only a gate that routed nothing may be marked", blocking) + } + } +} + +// TestGateRecordedEventCarriesStubMarker is DKT-983: a stub-trusted command's +// pass is already marked stub:true in the trust store, in `gate_results` rows +// (`step gates --json`), and in `run report` — but the `gate-recorded` EVENT +// carried none of it, so a stub pass was byte-identical to a real measurement +// on the event stream. RUN-63 / vorpal.git seq 13052 is this exactly: +// `gate-recorded ... detail=ac-commands exit=0 pre=true verdict=pass` beside a +// `step gates --json` that showed `stub=true` for the same record. +// +// The assertion is on the STORED PAYLOAD, exactly as +// TestPreGateEventsMarkTheVerdictThatRoutedNothing's is, because that is what +// both readers see: `events list` renders `data` as sorted key=value pairs, +// and `--json` hands the object straight to a program. +func TestGateRecordedEventCarriesStubMarker(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + repoRoot := t.TempDir() + + // A trust entry that declares `stub = true` (DKT-265): the command that + // runs is a placeholder, not the check its name implies. + argv := []string{"/usr/bin/true"} + e := testEngine() + runner := NewExecRunner(testRepoPaths(repoRoot)) + runner.LoadStore = sandboxTrust(t, trust.Entry{ + Name: "ac-commands", Argv: argv, ArgvSHA256: trust.ArgvSHA256(argv), + Repo: mustResolve(repoRoot), Stub: true, + }) + e.Gates = runner + + stepID := advanceToVerify(t, conn, e) + _, err := e.ClaimStepWithGates(conn, stepID, ClaimOptions{ + Owner: "w", NowMS: nowMS, + }) + testsupport.Must(t, err, "claim: %v", err) + + // The row already carries it (this is not new — it is the baseline the + // event is being brought up to). + rows, err := db.GateResultsForStep(conn, stepID) + testsupport.Must(t, err, "GateResultsForStep: %v", err) + if len(rows) != 1 || !rows[0].StubEntry { + t.Fatalf("recorded rows = %+v, want exactly one stub-marked row", rows) + } + + data := gateRecordedData(t, conn, stepID, "ac-commands") + if data["stub"] != true { + t.Errorf("gate-recorded for a stub-trusted gate carries stub = %v, "+ + "want true; payload %v is indistinguishable from a real measurement "+ + "on the event stream", data["stub"], data) + } + // The marker is not bought by dropping what was already there (DKT-63). + if data["verdict"] != VerdictPass { + t.Errorf("verdict = %v, want %q — the stub marker must not change the "+ + "outcome it decorates", data["verdict"], VerdictPass) + } + if data["detail"] != "ac-commands" { + t.Errorf("the gate name left `detail`: %v", data) + } + + // AND a NON-stub gate stays unmarked: implement@0's gates run through + // advanceToVerify's pass-through runner, whose stub marker is a different + // field entirely (the legacy S3-migration `Stub`, never set by any live + // path) — so their `gate-recorded` events must carry no `stub` key. + implementID := stepIDByInstance(t, conn, "implement@0") + for _, blocking := range []string{"build", "tests"} { + if _, ok := gateRecordedData(t, conn, implementID, blocking)["stub"]; ok { + t.Errorf("gate-recorded for the non-stub gate %q carries a stub "+ + "marker; only a gate whose trust entry declared `stub` may be "+ + "marked", blocking) + } + } +} + +// gateRecordedData reads the `gate-recorded` payload for one step's gate, and +// fails the test when there is none. +func gateRecordedData( + t *testing.T, conn *sql.DB, stepID int, gate string, +) map[string]any { + t.Helper() + rows, err := conn.Query( + `SELECT data FROM events WHERE step_id = ? AND kind = ? ORDER BY seq`, + stepID, EventGateRecorded) + testsupport.Must(t, err, "reading gate-recorded events: %v", err) + defer rows.Close() + + for rows.Next() { + var raw string + testsupport.Must(t, rows.Scan(&raw), "scanning a gate-recorded event: %v", nil) + var fields map[string]any + testsupport.Must(t, json.Unmarshal([]byte(raw), &fields), + "the gate event payload is not an object: %v", nil) + if fields["detail"] == gate { + return fields + } + } + t.Fatalf("no gate-recorded event for gate %q on step %d", gate, stepID) + return nil +} + // advanceToVerify drives the fixture's run until `verify@0` — the step that // declares the `ac-commands` pre-gate — is claimable, and returns its id. // diff --git a/internal/engine/ready.go b/internal/engine/ready.go index 1862100f..5bfb2f58 100644 --- a/internal/engine/ready.go +++ b/internal/engine/ready.go @@ -1444,7 +1444,7 @@ func (s *Scheduler) claimablePass(sorted []*db.Step, evicted map[int]bool) []*db admitted[step.Class]++ } s.grantScope(step, granted) - admittedCost += step.ExpectedCost + admittedCost += reservableCost(step) out = append(out, step) } return out @@ -1528,7 +1528,10 @@ func (s *Scheduler) offerBudget(step *db.Step, admittedCost float64) bool { if s.budget.unlimited() { return true } - return s.budget.spend()+admittedCost+step.ExpectedCost <= s.budget.cap + // reservableCost, not ExpectedCost: a vote step's declared cost is already + // in the floor at materialization (DKT-584), so the offer must not reserve + // it a second time. + return s.budget.spend()+admittedCost+reservableCost(step) <= s.budget.cap } func (s *Scheduler) priorityOf(step *db.Step) int { diff --git a/internal/engine/reap_ack.go b/internal/engine/reap_ack.go index 2db5e580..91ff43e5 100644 --- a/internal/engine/reap_ack.go +++ b/internal/engine/reap_ack.go @@ -187,9 +187,16 @@ func reapOneTx( // the assertion, recorded with `--reason`, that the holder is gone. It is not // an eviction primitive a bystander reaches casually — a forced reap of a // LIVE worker has exactly the risks a lease expiry has, which is why every -// consequence is the expiry reap's own: same event kind (`lease-reaped`, with -// `data.forced` and the reason distinguishing it), same write-class headroom -// hold, same return of the step to the pool. +// SCHEDULING consequence is the expiry reap's own: same event kind +// (`lease-reaped`, with `data.forced` and the reason distinguishing it), same +// write-class headroom hold, same return of the step to the pool. +// +// The one deliberate divergence is the ATTEMPT BUDGET (DKT-585): a reap +// carrying `data.forced` does not count the reaped attempt against +// `max_attempts`, because the verb's premise is that the holder died before +// an executor could fail — see the exemption below, and +// db.ExemptStepAttemptFromBudgetTx for why it is a +1 nudge of the base and +// never a touch of `attempt` itself. func ForceReapStep(conn *sql.DB, stepID int, reason string, nowMS int64) error { if reason == "" { return validationErr( @@ -242,6 +249,22 @@ func ForceReapStep(conn *sql.DB, stepID int, reason string, nowMS int64) error { if err := reapOneTx(tx, sched, step.RunID, target, string(data), nowMS); err != nil { return err } + // A forced reap does not consume the attempt budget (DKT-585). The claim + // already spent an `attempt` when it was minted — the counter increments at + // claim, and ReapStepTx rightly leaves it alone — but the SPEND is a relay's + // assertion that the holder died, not a measurement that an executor + // failed, so the exhaustion math (`attempt - attempt_base` against + // `max_attempts`, the `step fail` branch) must not count it. + // + // This lives HERE and not in reapOneTx, per ReapStepTx's own mechanism/ + // classification split: the shared reap is mechanism, and the two callers + // classify differently. An ordinary TTL expiry may still be an executor + // that went unresponsive under its own load — a judgment the expiry path + // keeps charging for, unchanged — while the forced path carries an explicit + // assertion (`data.forced`, `--reason`) that no executor ever got to work. + if err := db.ExemptStepAttemptFromBudgetTx(tx, step.ID, nowMS); err != nil { + return err + } if err := tx.Commit(); err != nil { return fmt.Errorf("reaping %s: %w", step.Instance, err) } diff --git a/internal/engine/registry_audit.go b/internal/engine/registry_audit.go new file mode 100644 index 00000000..041a7776 --- /dev/null +++ b/internal/engine/registry_audit.go @@ -0,0 +1,428 @@ +package engine + +import ( + "database/sql" + "fmt" + "os" + "sort" + "sync" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// CROSS-PROJECT REGISTRY DRIFT (DKT-614): auto-registration adopts the corpus's +// current contents, but ONLY as a side effect of activating a run IN THAT +// PROJECT. A project nobody has activated lately goes stale against a corpus +// every other project already moved past, and nothing says so. +// +// The measured state that produced this: of eleven projects sharing one store, +// exactly one carried every workflow at its current corpus version. Seven were +// several versions behind across most names and carried no registration at all +// for the two newest ones — and the only way to learn that was to cd into each +// checkout in turn, or to query sqlite directly. +// +// This file answers the question ONCE FOR THE WHOLE STORE. The corpus is +// SHARED — `~/.docket/config` is not per-project — so what "current" means is +// computed from ONE scan and then compared against every project's rows, rather +// than rescanned eleven times to get eleven identical answers. +// +// IT IS A SIBLING OF source_drift.go AND orphan_registration.go, NOT A +// REPLACEMENT for either: +// +// - CheckWorkflowSource asks whether THE FILE ONE ROW POINTS AT still holds +// the registered bytes. That is a hash question about a path. +// - WorkflowOriginIndex asks whether ANY file still declares a registered +// NAME. That is an existence question about a name, and this file asks the +// same one — of both registries, for every project at once. +// - This file adds the third question neither answers: is the version this +// project registered for that name the version the corpus NOW DECLARES? +// That is a comparison of two NUMBERS, and a hash check cannot stand in for +// it: a project sitting on investigation@4 while the corpus declares +// investigation@8 has a row whose own source file is long gone, so the +// hash question returns `unreadable` and says nothing about the eight. +// +// AND IT REPAIRS NOTHING. Adopting a bumped definition is what activation does, +// deliberately and inside a transaction, with the validation and the collision +// rules that go with it. A report that quietly registered rows on the way past +// would be doing an activation's work with none of an activation's guarantees. + +// CorpusEntry is the CURRENT declaration of one registry name in the shared +// instance-config roots: the highest version any root declares, and the file +// declaring it. +type CorpusEntry struct { + // Kind is RegistrationKindWorkflow or RegistrationKindSchema. + Kind string `json:"kind"` + Name string `json:"name"` + Version int `json:"version"` + Path string `json:"path"` +} + +// CorpusIndex is one scan of the instance-config roots, read as "what does the +// corpus declare RIGHT NOW, for both registries". +// +// It is WorkflowOriginIndex's shape, with the two differences the audit needs: +// it carries the VERSION rather than only the path, and it covers SCHEMAS as +// well as workflows. The two indexes are kept apart rather than merged because +// WorkflowOriginIndex is built lazily on activation's refusal path, where +// paying for a schema pass to answer a workflow question would be a cost with +// no reader. +type CorpusIndex struct { + // scan is the discovery this index reads. A nil scan means NO ROOT EXISTS + // (scanConfigDirs' F17 dormancy) — the state in which every registered name + // in the store would trivially look orphaned, which is why Scanned() is a + // question every caller has to ask before believing a verdict. + scan *configScan + + once sync.Once + // current maps `kind\x00name` to the highest version declared for it. + current map[string]CorpusEntry + // err is the failure that stopped the build. Nothing is classified after + // one: a half-read root can only manufacture false orphans and false + // lag, and both cost an operator an investigation. + err error +} + +// ScanCorpus scans the instance-config roots for the audit. +// +// The error is the SCAN's own refusal — a root that is a regular file, a +// dangling symlink — surfaced rather than swallowed, for orphan_registration's +// reason: those are exactly the states in which having looked nowhere would be +// reported as every registration being stale. +func ScanCorpus() (*CorpusIndex, error) { + scan, err := scanConfigDirs(resolvePaths().InstanceConfigDirs()) + if err != nil { + return nil, err + } + return &CorpusIndex{scan: scan}, nil +} + +// Scanned reports whether any instance-config root existed to look in. +func (i *CorpusIndex) Scanned() bool { + return i != nil && i.scan != nil && len(i.scan.roots) > 0 +} + +// Roots are the canonicalized roots that were scanned, in precedence order. +func (i *CorpusIndex) Roots() []string { + if i == nil || i.scan == nil { + return nil + } + return i.scan.roots +} + +// Err is the failure that stopped the build, if any. +func (i *CorpusIndex) Err() error { + if i == nil { + return nil + } + i.build() + return i.err +} + +// Current returns what the corpus declares for one kind and name. +func (i *CorpusIndex) Current(kind, name string) (CorpusEntry, bool) { + if i == nil { + return CorpusEntry{}, false + } + i.build() + if i.err != nil { + return CorpusEntry{}, false + } + entry, ok := i.current[kind+"\x00"+name] + return entry, ok +} + +// Entries are every declaration the scan found, ordered by kind then name, for +// a caller that wants to render what "current" meant. +func (i *CorpusIndex) Entries() []CorpusEntry { + if i == nil { + return nil + } + i.build() + out := make([]CorpusEntry, 0, len(i.current)) + for _, e := range i.current { + out = append(out, e) + } + sort.Slice(out, func(a, b int) bool { + if out[a].Kind != out[b].Kind { + return out[a].Kind < out[b].Kind + } + return out[a].Name < out[b].Name + }) + return out +} + +// build derives the identity of every registry file the scan found, ONCE. +// +// THE HIGHEST VERSION WINS, not the first file seen. A corpus bumps a workflow +// IN PLACE — the version lives in the body, so `investigation.toml` simply says +// 8 where it used to say 4 — but schemas are versioned IN THE FILENAME, so +// `findings@1.json` and `findings@2.json` sit side by side and only the second +// is what a project ought to be carrying. Taking the maximum is the one rule +// that reads both layouts correctly. +// +// A file that DOES NOT PARSE declares no identity — configIdentity's rule, for +// registryIdentityKey's reason: the refusal for an unparseable definition +// belongs to registration, which words it precisely. The consequence is worth +// stating, as it is there: while a workflow file is unparseable, the name it +// WOULD declare reads as orphaned here. +// +// A workflow that cannot be READ fails the build instead, and every verdict +// becomes unavailable. That is an environment fault rather than an authoring +// one, and manufacturing an orphan out of a permission error is the one output +// this report must never produce. +func (i *CorpusIndex) build() { + i.once.Do(func() { + i.current = make(map[string]CorpusEntry) + if i.scan == nil { + return + } + for _, path := range i.scan.paths { + // A SCHEMA'S IDENTITY IS ITS FILENAME, so its document is never + // opened here. Reading every schema in the corpus to learn a name + // already spelled in the path would be IO with no reader. + var src []byte + if !isSchemaConfigPath(path) { + body, err := os.ReadFile(path) + if err != nil { + i.err = fmt.Errorf( + "reading the instance-config workflow %s: %w", path, err) + i.current = nil + return + } + src = body + } + + kind, name, version, ok := configIdentity(path, src) + if !ok { + continue + } + key := kind + "\x00" + name + if prev, dup := i.current[key]; dup && prev.Version >= version { + continue + } + i.current[key] = CorpusEntry{ + Kind: kind, Name: name, Version: version, Path: path, + } + } + }) +} + +// RegistryLag is one registered name whose highest registered version is BEHIND +// what the corpus now declares. +type RegistryLag struct { + Kind string `json:"kind"` + Name string `json:"name"` + // RegisteredVersion is the highest version this project holds a row for, + // retired versions INCLUDED: the question is what the registry contains, + // not what would bind, and a retired row is still a registration. + RegisteredVersion int `json:"registered_version"` + CurrentVersion int `json:"current_version"` + // CurrentPath is the corpus file declaring the current version, so the + // operator can read the definition they are behind without a second search. + CurrentPath string `json:"current_path,omitempty"` +} + +// RegistryOrphan is one registered name that NO file in any scanned root +// declares any more — the state a rename leaves behind, since registering the +// new name never retires the old one. +type RegistryOrphan struct { + Kind string `json:"kind"` + Name string `json:"name"` + // Versions are every registered version of the stranded name, ascending. + // A rename typically strands several at once, and `workflow deprecate` + // takes them one at a time. + Versions []int `json:"versions"` + // Retired is true when EVERY version listed is already deprecated — the + // orphan an operator has finished with. It is reported rather than filtered + // out, for `workflow list --orphans`' reason: hiding an already-retired + // orphan would leave a cleanup pass unable to see its own work. Schemas + // carry no deprecation, so it is always false for them. + Retired bool `json:"retired"` +} + +// ProjectRegistryAudit is one project's verdict. +type ProjectRegistryAudit struct { + ProjectID int `json:"project_id"` + Project string `json:"project"` + Prefix string `json:"prefix,omitempty"` + Identity string `json:"identity,omitempty"` + // Compared is how many distinct registered names were classified. It is + // carried so "no findings" is distinguishable from "nothing registered + // here" — two clean-looking results with nothing in common. + Compared int `json:"compared"` + Behind []RegistryLag `json:"behind"` + Orphaned []RegistryOrphan `json:"orphaned"` +} + +// Clean reports whether this project matched the corpus on every name. +func (p ProjectRegistryAudit) Clean() bool { + return len(p.Behind) == 0 && len(p.Orphaned) == 0 +} + +// RegistryAudit is the whole store's verdict, plus where "current" was read +// from. +type RegistryAudit struct { + // Roots are the instance-config roots the one scan walked, in precedence + // order, so a reader can tell WHERE current was read from — and can see + // when a root they expected is missing. + Roots []string `json:"roots"` + Projects []ProjectRegistryAudit `json:"projects"` + // BehindTotal and OrphanedTotal are store-wide counts of the FINDINGS, not + // of the projects carrying them. + BehindTotal int `json:"behind_total"` + OrphanedTotal int `json:"orphaned_total"` +} + +// RegistryAuditOptions narrows the audit. +type RegistryAuditOptions struct { + // ProjectID audits ONE project; zero audits every project in the store, + // which is the whole point of the verb and therefore the default. + ProjectID int +} + +// AuditRegistries compares every project's registered workflow and schema rows +// against ONE scan of the shared corpus. +// +// The index is passed in rather than scanned here so the CALLER owns the two +// refusals that precede any verdict — no root exists, and a root that would not +// scan — and can word them for an operator. A report that silently classified +// against an empty index would name every registration in the store. +func AuditRegistries( + conn *sql.DB, index *CorpusIndex, opts RegistryAuditOptions, +) (*RegistryAudit, error) { + projects, err := db.ListProjects(conn) + if err != nil { + return nil, fmt.Errorf("listing projects: %w", err) + } + + audit := &RegistryAudit{Roots: index.Roots()} + for _, p := range projects { + if opts.ProjectID != 0 && p.ID != opts.ProjectID { + continue + } + one, err := auditProject(conn, index, p) + if err != nil { + return nil, err + } + audit.BehindTotal += len(one.Behind) + audit.OrphanedTotal += len(one.Orphaned) + audit.Projects = append(audit.Projects, *one) + } + return audit, nil +} + +// registered is one name's rows in one project's registry, reduced to the three +// facts the comparison needs. +type registered struct { + highest int + versions []int + // live is true once any version of the name is still eligible to bind. + live bool +} + +func auditProject( + conn *sql.DB, index *CorpusIndex, p *model.Project, +) (*ProjectRegistryAudit, error) { + out := &ProjectRegistryAudit{ + ProjectID: p.ID, Project: p.Name, Prefix: p.Prefix, Identity: p.Identity, + } + + // EVERY registered version is listed, retired ones included: `workflow + // list`'s default hides them because they no longer BIND, and this verb is + // asking what the registry HOLDS. An orphan whose versions were all retired + // last week is a finished cleanup, and reporting it as finished is more use + // than not reporting it at all. + workflows, _, err := db.ListWorkflows(conn, db.WorkflowListOptions{ProjectID: p.ID}) + if err != nil { + return nil, fmt.Errorf("listing workflows for project %d: %w", p.ID, err) + } + byName := make(map[string]*registered, len(workflows)) + for _, wf := range workflows { + fold(byName, wf.Name, wf.Version, !wf.Deprecated()) + } + out.Compared += classify(out, index, RegistrationKindWorkflow, byName) + + schemas, _, err := db.ListSchemas(conn, db.SchemaListOptions{ProjectID: p.ID}) + if err != nil { + return nil, fmt.Errorf("listing schemas for project %d: %w", p.ID, err) + } + byName = make(map[string]*registered, len(schemas)) + for _, s := range schemas { + // THE BUILTIN IS NOT CORPUS-BACKED. `aggregate@1` ships in the binary, + // is visible to every project by design, and no file in any root + // declares it — so classifying it would report one unfixable orphan for + // every project in the store, on every run of this verb. + if s.Builtin { + continue + } + fold(byName, s.Name, s.Version, true) + } + out.Compared += classify(out, index, RegistrationKindSchema, byName) + + sortLags(out.Behind) + sortOrphans(out.Orphaned) + return out, nil +} + +func fold(index map[string]*registered, name string, version int, live bool) { + r, ok := index[name] + if !ok { + r = ®istered{} + index[name] = r + } + r.versions = append(r.versions, version) + if version > r.highest { + r.highest = version + } + r.live = r.live || live +} + +// classify sorts one kind's registered names into the two findings, and returns +// how many names it looked at. +// +// THE VERDICT IS PER NAME, NOT PER VERSION — the rule `workflow list --orphans` +// established. A superseded version whose name is still declared is ordinary +// lineage and not a finding, and a project holding four versions of one stale +// name is one thing to fix, not four. +func classify( + out *ProjectRegistryAudit, index *CorpusIndex, kind string, + names map[string]*registered, +) int { + for name, r := range names { + sort.Ints(r.versions) + entry, declared := index.Current(kind, name) + switch { + case !declared: + out.Orphaned = append(out.Orphaned, RegistryOrphan{ + Kind: kind, Name: name, Versions: r.versions, Retired: !r.live, + }) + case r.highest < entry.Version: + out.Behind = append(out.Behind, RegistryLag{ + Kind: kind, Name: name, + RegisteredVersion: r.highest, + CurrentVersion: entry.Version, + CurrentPath: entry.Path, + }) + } + } + return len(names) +} + +func sortLags(lags []RegistryLag) { + sort.Slice(lags, func(a, b int) bool { + if lags[a].Kind != lags[b].Kind { + return lags[a].Kind < lags[b].Kind + } + return lags[a].Name < lags[b].Name + }) +} + +func sortOrphans(orphans []RegistryOrphan) { + sort.Slice(orphans, func(a, b int) bool { + if orphans[a].Kind != orphans[b].Kind { + return orphans[a].Kind < orphans[b].Kind + } + return orphans[a].Name < orphans[b].Name + }) +} diff --git a/internal/engine/registry_audit_test.go b/internal/engine/registry_audit_test.go new file mode 100644 index 00000000..1aebb529 --- /dev/null +++ b/internal/engine/registry_audit_test.go @@ -0,0 +1,319 @@ +package engine + +import ( + "database/sql" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-614: the cross-project audit — every project's registry compared against +// ONE scan of the shared corpus. +// +// The state it exists for: of eleven projects sharing one store, exactly one +// carried every workflow at its current corpus version, and the only way to +// learn that was to cd into each checkout in turn or to read sqlite directly. + +// investigationV8 is the corpus's CURRENT declaration. A project registered at +// version 4 is behind it — the shape the measured drift actually took, since a +// corpus bumps a workflow in place rather than adding a second file. +const investigationV8 = ` +[pipeline] +name = "investigation" +version = 8 + +[match] +kind = ["task"] + +[[step]] +name = "look" +executor = "w" +emits = "out" +after = [] +` + +const findingsSchemaV1 = `{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "findings@1", + "type": "object", + "properties": { "risk": { "type": "string" } } +}` + +const findingsSchemaV2 = `{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "findings@2", + "type": "object", + "properties": { "risk": { "type": "string" }, "note": { "type": "string" } } +}` + +// auditCorpus writes the two-registry corpus every test here compares against: +// investigation@8 (workflow) and findings@1 + findings@2 (schemas). +func auditCorpus(t *testing.T, configDir string) { + t.Helper() + writeConfigFile(t, configDir, "workflows/investigation.toml", investigationV8) + writeConfigFile(t, configDir, "schemas/findings@1.json", findingsSchemaV1) + writeConfigFile(t, configDir, "schemas/findings@2.json", findingsSchemaV2) +} + +// registerWorkflowRow puts one workflow row in one project's registry, without +// going anywhere near activation: the audit's whole subject is registries that +// activation has NOT visited lately. +func registerWorkflowRow(t *testing.T, conn *sql.DB, projectID int, name string, version int) { + t.Helper() + body := "registered bytes for " + name + _, _, err := db.InsertWorkflow(conn, &model.Workflow{ + ProjectID: projectID, Name: name, Version: version, + SourcePath: "/nowhere/" + name + ".toml", SourceSHA256: workflow.SHA256([]byte(body)), + Body: body, Parsed: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "registering %s@%d in project %d: %v", name, version, projectID, err) +} + +func registerSchemaRow(t *testing.T, conn *sql.DB, projectID int, name string, version int) { + t.Helper() + body := findingsSchemaV1 + _, _, err := db.InsertSchema(conn, &model.Schema{ + ProjectID: projectID, Name: name, Version: version, + SourcePath: "/nowhere/" + name + ".json", SourceSHA256: workflow.SHA256([]byte(body)), + Body: body, Ordered: "{}", + }, model.NowMS()) + testsupport.Must(t, err, "registering schema %s@%d in project %d: %v", + name, version, projectID, err) +} + +// auditOnce scans the corpus and audits the store, failing the test on either +// of the refusals the CLI words for an operator. +func auditOnce(t *testing.T, conn *sql.DB, opts RegistryAuditOptions) *RegistryAudit { + t.Helper() + index, err := ScanCorpus() + testsupport.Must(t, err, "scanning the corpus: %v", err) + if !index.Scanned() { + t.Fatal("no instance-config root was scanned; every name would look orphaned") + } + testsupport.Must(t, index.Err(), "building the corpus index: %v", index.Err()) + + audit, err := AuditRegistries(conn, index, opts) + testsupport.Must(t, err, "auditing: %v", err) + return audit +} + +// projectAudit picks one project's verdict out of the store-wide report. +func projectAudit(t *testing.T, audit *RegistryAudit, id int) ProjectRegistryAudit { + t.Helper() + for _, p := range audit.Projects { + if p.ProjectID == id { + return p + } + } + t.Fatalf("project %d is missing from the audit; it covers %d project(s)", + id, len(audit.Projects)) + return ProjectRegistryAudit{} +} + +// TestRegistryAuditReportsEveryProjectFromOneScan is DKT-614's headline: one +// invocation, every project, both registries — the stale project named without +// having stood in it, and the current one not named alongside it. +func TestRegistryAuditReportsEveryProjectFromOneScan(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + // TWO REAL PROJECT ROWS. The first identity to register CLAIMS the default + // row (db.ensureProject), so naming only one and assuming the default is + // the other would put both registries in one project and prove nothing. + stale, err := db.EnsureProject(conn, "/repo/stale.git", "stale.git", model.NowMS()) + testsupport.Must(t, err, "creating the stale project: %v", err) + current, err := db.EnsureProject(conn, "/repo/current.git", "current.git", model.NowMS()) + testsupport.Must(t, err, "creating the current project: %v", err) + if stale == current { + t.Fatalf("both identities resolved to project %d", stale) + } + + // The stale project: behind on both registries, plus a name a corpus + // rename stranded. + registerWorkflowRow(t, conn, stale, "investigation", 4) + registerWorkflowRow(t, conn, stale, "security-load-bearing", 12) + registerSchemaRow(t, conn, stale, "findings", 1) + + // The current one: everything at the corpus's version. + registerWorkflowRow(t, conn, current, "investigation", 8) + registerSchemaRow(t, conn, current, "findings", 2) + + audit := auditOnce(t, conn, RegistryAuditOptions{}) + + if len(audit.Roots) == 0 { + t.Error("the audit names no scanned roots, so a reader cannot tell " + + "where `current` was read from") + } + + got := projectAudit(t, audit, stale) + if len(got.Behind) != 2 { + t.Fatalf("the stale project reports %d behind, want 2 (the workflow AND "+ + "the schema): %+v", len(got.Behind), got.Behind) + } + // Ordered by kind then name, so schema precedes workflow. + if got.Behind[0].Kind != RegistrationKindSchema || got.Behind[0].Name != "findings" || + got.Behind[0].RegisteredVersion != 1 || got.Behind[0].CurrentVersion != 2 { + t.Errorf("schema lag = %+v, want findings registered 1 / current 2 — "+ + "the audit must cover schemas, not workflows only", got.Behind[0]) + } + if got.Behind[1].Name != "investigation" || got.Behind[1].RegisteredVersion != 4 || + got.Behind[1].CurrentVersion != 8 { + t.Errorf("workflow lag = %+v, want investigation registered 4 / current 8", + got.Behind[1]) + } + if got.Behind[1].CurrentPath == "" { + t.Error("the lag names no corpus file, so the operator cannot read the " + + "definition they are behind without a second search") + } + + if len(got.Orphaned) != 1 || got.Orphaned[0].Name != "security-load-bearing" { + t.Fatalf("orphans = %+v, want the one stranded name", got.Orphaned) + } + if len(got.Orphaned[0].Versions) != 1 || got.Orphaned[0].Versions[0] != 12 { + t.Errorf("the orphan lists %v, want the registered version(s) to deprecate", + got.Orphaned[0].Versions) + } + if got.Orphaned[0].Retired { + t.Error("the orphan reads as already retired; it still binds") + } + + if other := projectAudit(t, audit, current); !other.Clean() { + t.Errorf("the up-to-date project carries findings: %+v", other) + } else if other.Compared == 0 { + t.Error("the up-to-date project compared nothing, so `clean` says nothing") + } + + if audit.BehindTotal != 2 || audit.OrphanedTotal != 1 { + t.Errorf("store-wide totals = %d behind / %d orphaned, want 2 / 1", + audit.BehindTotal, audit.OrphanedTotal) + } +} + +// TestRegistryAuditVerdictIsPerNameNotPerVersion: a superseded version whose +// name is registered at the current version too is ordinary lineage. The rule +// `workflow list --orphans` established, applied to the lag question — a +// project holding four versions of one name has one thing to fix, not four. +func TestRegistryAuditVerdictIsPerNameNotPerVersion(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + registerWorkflowRow(t, conn, db.DefaultProjectID, "investigation", 4) + registerWorkflowRow(t, conn, db.DefaultProjectID, "investigation", 8) + + got := projectAudit(t, auditOnce(t, conn, RegistryAuditOptions{}), db.DefaultProjectID) + if len(got.Behind) != 0 { + t.Errorf("a project holding the current version is reported behind on "+ + "the older one it also holds: %+v", got.Behind) + } +} + +// TestRegistryAuditTakesTheHighestVersionTheCorpusDeclares: schemas are +// versioned IN THE FILENAME, so `findings@1.json` and `findings@2.json` sit +// side by side and only the second is current. Taking the first file scanned +// would report every up-to-date project as ahead of a corpus it matches. +func TestRegistryAuditTakesTheHighestVersionTheCorpusDeclares(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + index, err := ScanCorpus() + testsupport.Must(t, err, "scanning: %v", err) + entry, ok := index.Current(RegistrationKindSchema, "findings") + if !ok { + t.Fatal("the corpus declares findings in two files and the index found neither") + } + if entry.Version != 2 { + t.Errorf("current findings = @%d, want @2 — the highest version any "+ + "root declares", entry.Version) + } + _ = conn +} + +// TestRegistryAuditIgnoresTheBuiltinSchema: `aggregate@1` ships in the binary, +// is visible to every project by design, and no file in any root declares it. +// Classifying it would report one unfixable orphan for every project in the +// store on every run of this verb. +func TestRegistryAuditIgnoresTheBuiltinSchema(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + got := projectAudit(t, auditOnce(t, conn, RegistryAuditOptions{}), db.DefaultProjectID) + for _, orphan := range got.Orphaned { + if orphan.Name == "aggregate" { + t.Fatalf("the builtin schema is reported orphaned: %+v", orphan) + } + } +} + +// TestRegistryAuditMarksAnOrphanWhoseVersionsAreAllRetired: the orphan an +// operator has finished with stays listed, marked — a cleanup pass has to be +// able to see its own work, which is why `workflow list --orphans` shows +// deprecated rows too. +func TestRegistryAuditMarksAnOrphanWhoseVersionsAreAllRetired(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + registerWorkflowRow(t, conn, db.DefaultProjectID, "security-load-bearing", 11) + registerWorkflowRow(t, conn, db.DefaultProjectID, "security-load-bearing", 12) + for _, version := range []int{11, 12} { + _, err := db.DeprecateWorkflow( + conn, db.DefaultProjectID, "security-load-bearing", version, model.NowMS()) + testsupport.Must(t, err, "deprecating @%d: %v", version, err) + } + + got := projectAudit(t, auditOnce(t, conn, RegistryAuditOptions{}), db.DefaultProjectID) + if len(got.Orphaned) != 1 { + t.Fatalf("orphans = %+v, want the retired one still listed", got.Orphaned) + } + if !got.Orphaned[0].Retired { + t.Error("every version is deprecated and the orphan is not marked retired, " + + "so a finished cleanup reads as outstanding work") + } + if len(got.Orphaned[0].Versions) != 2 { + t.Errorf("versions = %v, want both retired registrations listed", + got.Orphaned[0].Versions) + } +} + +// TestRegistryAuditNarrowsToOneProject: the default is every project — that is +// the verb's whole point — but `--project` exists for the operator who already +// knows which one they are asking about. +func TestRegistryAuditNarrowsToOneProject(t *testing.T) { + conn, configDir := configRepo(t) + auditCorpus(t, configDir) + + first, err := db.EnsureProject(conn, "/repo/first.git", "first.git", model.NowMS()) + testsupport.Must(t, err, "creating the first project: %v", err) + other, err := db.EnsureProject(conn, "/repo/other.git", "other.git", model.NowMS()) + testsupport.Must(t, err, "creating the second project: %v", err) + registerWorkflowRow(t, conn, other, "investigation", 4) + registerWorkflowRow(t, conn, first, "investigation", 4) + + audit := auditOnce(t, conn, RegistryAuditOptions{ProjectID: other}) + if len(audit.Projects) != 1 || audit.Projects[0].ProjectID != other { + t.Fatalf("--project audited %d project(s), want only %d", len(audit.Projects), other) + } + if audit.BehindTotal != 1 { + t.Errorf("behind_total = %d, want 1: the totals count the AUDITED "+ + "population, not the store", audit.BehindTotal) + } +} + +// TestCorpusIndexReportsNothingScannedWithNoRoot is the state every caller has +// to check before believing a verdict: with no root, every registered name in +// the store trivially has no file declaring it. +func TestCorpusIndexReportsNothingScannedWithNoRoot(t *testing.T) { + docketDir := t.TempDir() + t.Setenv("DOCKET_PATH", docketDir) + + index, err := ScanCorpus() + testsupport.Must(t, err, "scanning an empty store: %v", err) + if index.Scanned() { + t.Fatal("a store with no config directory reports a scanned corpus") + } + if _, ok := index.Current(RegistrationKindWorkflow, "investigation"); ok { + t.Error("an unscanned index answered a `current version` question") + } +} diff --git a/internal/engine/render.go b/internal/engine/render.go index 67484582..d7518b18 100644 --- a/internal/engine/render.go +++ b/internal/engine/render.go @@ -24,6 +24,30 @@ import ( // artifacts each delimited and labeled in DECLARED order, the pinned-file list // with hashes, and the output instruction — and it names NO instance concept. // The genericity gate checks its bytes like any other core surface. +// +// WHAT A PACKET DRAWS FROM — the exhaustive list, for anyone deciding where +// to put words they need a worker to read (DKT-725). A packet is the context +// bundle (§6.6's five sources: the pinned step definition, the issue's +// ACTIVATION-FROZEN body_snapshot and issue_snapshot, recorded input +// artifacts, and the pin list) plus the step's declared packet files and the +// step's OWN routing record, rendered as `== RESOLUTION`. Two consequences +// operators repeatedly discover the hard way: +// +// - issue COMMENTS never render. They are an audit surface, not a context +// source, and no template can reach them. +// - a mid-run `description` edit never renders. The packet reads +// `body_snapshot`, frozen at activation — §9 item 5's edit immunity. +// - a mid-run `--scope` edit never renders either, and for the same reason: +// the brief's `scope:` line reads `issue_snapshot`, and so does the scope +// the step's `issue.diff` is recorded over (DKT-741). There is no verb +// that refreshes it; an authorized mid-run widen is made real by taking +// the issue out of the run and re-planning it, and `issue edit --scope` +// says so when it lands on an issue with live steps in a live run. +// +// The sanctioned steering channel is the resolve note: `step resolve --as +// retry|rerun-gates -m` renders on the same step's re-execution, and `--as +// fix-round -m` is stamped onto the new round's rows (stampEntryRouting) so +// the authorization's remedy reaches the round it paid for. // defaultPacket is the shipped template. It lives under internal/ because the // Vorpal build's include list requires embeds there — the same constraint the @@ -85,10 +109,15 @@ func RenderStep( // to the declared hint alone, so a label-resolved executor could never // receive its own contract: the corpus shipped per-resolved-hint files that // no packet could ever name. The resolved hint arrives here, at render time, -// which is where substitution ALREADY re-derives — activation pinned the -// whole config tree, so the resolved contract verifies against its pin like -// any other entry, and a hint whose contract does not exist refuses loudly -// naming the exact path. +// which is where substitution ALREADY re-derives — and the resolved contract +// verifies against its pin like any other entry, while a hint whose contract +// does not exist refuses loudly naming the exact path. Since DKT-581, +// activation pins the packet CLOSURE rather than the whole config tree — +// entries substituted with the declared executor and fanout hints — so a +// resolved hint outside the workflow's own declarations resolves only if the +// corpus declares it somewhere in the bound definition (the shipped corpus +// declares label-resolved executors as `when`-gated steps, which the closure +// covers) or the operator pinned its contract with `--pin`. // // An empty executor is the declared behavior, unchanged. The override also // lands on the rendered step row's `executor`, so the packet's `target:` diff --git a/internal/engine/repin.go b/internal/engine/repin.go index d4bd0bfa..22a6af94 100644 --- a/internal/engine/repin.go +++ b/internal/engine/repin.go @@ -50,6 +50,57 @@ import ( // even after the current agreement moves. The event log is already the // package's history mechanism (§9 item 2); a parallel pin-history table // would be a second source of the same fact. +// +// DROPPING A REF THAT NO LONGER RESOLVES (DKT-582). "Adopt current disk bytes" +// has nothing to say about a ref with no current bytes, so repin refused the +// whole set on one NOT_FOUND — and corpus commits that DELETE a contract are as +// ordinary as commits that edit one. RUN-42 died on `contracts/test-infra.md` +// and `contracts/pr-comment-author.md` being deleted; RUN-36 died the same way +// after surviving two earlier repins. Neither run had a step left that would +// ever open either file. +// +// So there is a second disposition, and its guard is the pending packet closure +// (pending_closure.go): a NOT_FOUND FILE pin that NO non-terminal step can +// reach may be RETIRED — the row deleted, a `run-repinned` event recorded with +// a null `new_sha256`. The properties above are unchanged by it. Completed +// steps' provenance still is not rewritten (the old sha is in the event, and +// their own rows are untouched), and the run stays whole rather than partially +// recovered, because a ref nothing left to run can read is not part of the +// agreement the remaining steps work under. +// +// It is OPT-IN — `--drop REF` or `--drop-unresolvable` — for the same reason +// repin itself is a separate verb: deciding a pinned file is gone for good is a +// person's decision, and an engine that made it silently would delete the pin +// that was about to explain a refusal. A NOT_FOUND ref a pending step DOES +// read still refuses, naming the steps, however the flags are spelled. +// +// ADOPTED BYTES BRING THEIR OWN CLOSURE (DKT-805). "Adopt current disk bytes" +// used to re-hash the run's EXISTING pin set and nothing more — but the pin set +// is the packet closure of the bytes the run froze (DKT-581), and the adopted +// bytes can have a DIFFERENT closure. A corpus edit that makes a contract +// include a new fragment left repin reporting full success while every step +// reading that contract became unrenderable: the fragment was not pinned, so +// render refused (VALIDATION_ERROR) with "start a new run" as its remedy — the +// disposition repin exists to avoid. RUN-56 lost a whole 18-row dispatch to +// exactly that, the day after an operator-approved repin reported +// "Repinned 10 pin(s) (0 dropped, 19 already matched)". +// +// So a repin that adopts changed bytes also walks the pending packet closure +// those bytes reach (pending_closure.go — the same walk the drop guard uses) +// and PINS every reachable file ref the run does not already hold, at its +// current disk hash, in the same transaction. Adding a pin can never rewrite +// completed provenance: no recorded step ever read an unpinned ref (render +// refuses on exactly that), so there is no history for the new row to +// contradict — the same argument RA3's re-activation additions rest on. Each +// added ref records its own `run-repinned` event with a null `old_sha256` and +// `added: true`, the mirror of the drop's signature: "there were no pinned +// bytes before" is a different fact from "the bytes were these". +// +// A newly-required ref with NO bytes on disk refuses the whole set up front, +// naming the ref and the steps that read it — pinning it is impossible and +// repinning around it would trade the CONFLICT for a render-time refusal, which +// is the exact failure this closure walk exists to close. The dispositions are +// the missing-refusal's: restore the file, or abandon the run and re-plan. type RepinChange struct { Kind string `json:"kind"` Ref string `json:"ref"` @@ -58,6 +109,18 @@ type RepinChange struct { // Path is where the file pin resolved on disk, "" for registered-object // pins — the same field PinVerdict carries, for the same operator. Path string `json:"path,omitempty"` + // Dropped marks the DKT-582 disposition: the ref no longer resolves at all + // and nothing non-terminal reads it, so the pin was retired rather than + // moved. NewSHA256 is "" on exactly these, and the recorded event carries a + // null `new_sha256` — "there are no bytes now" said in the trail, which is + // a different fact from "the bytes are these". + Dropped bool `json:"dropped,omitempty"` + // Added marks the DKT-805 disposition: the adopted bytes' packet closure + // reaches a file ref the run never pinned, so a pin was created rather than + // moved. OldSHA256 is "" on exactly these, and the recorded event carries a + // null `old_sha256` — "there were no pinned bytes before" said in the + // trail, the mirror of the drop's signature. + Added bool `json:"added,omitempty"` } // RepinOutcome reports what RepinRun did. @@ -66,13 +129,48 @@ type RepinOutcome struct { // Repinned is one entry per pin whose recorded hash moved, in the // (kind, ref) order the report walks — empty (never nil) on a no-op. Repinned []RepinChange `json:"repinned"` + // Dropped is one entry per pin RETIRED because its ref no longer resolves + // and no non-terminal step reads it — empty (never nil) when none were. + Dropped []RepinChange `json:"dropped"` + // Added is one entry per pin CREATED because the adopted bytes' packet + // closure reaches a ref the run never pinned (DKT-805) — empty (never nil) + // when the adopted closure was already covered. + Added []RepinChange `json:"added"` // Unchanged counts the pins that already matched disk. Unchanged int `json:"unchanged"` } +// RepinOptions carries the operator's dispositions for pins that "adopt the +// current disk bytes" cannot fix on its own (DKT-582). +// +// DROPPING IS OPT-IN, ALWAYS. A ref that stopped resolving is either a deletion +// the run can survive or a deletion that wedges it, and only the pending +// closure can tell the two apart — so the verb refuses by default and these +// fields are how a person says "retire it", never how the engine decides to. +type RepinOptions struct { + // Reason is --reason, recorded verbatim on every event. + Reason string + // Drop names refs to retire individually. A named ref that still resolves, + // or that this run does not pin, is a refusal rather than a no-op: it means + // the operator and the run disagree about what is wrong. + Drop []string + // DropUnresolvable applies the same disposition to EVERY currently drifted + // file pin that no longer resolves and that no non-terminal step reads, + // without naming each. Pins that resolve to different bytes are untouched + // by it — those are repin's ordinary business. + DropUnresolvable bool +} + // RepinRun re-pins a run's drifted pins to what their refs resolve to now. func RepinRun(conn *sql.DB, runID int, reason string, nowMS int64) (*RepinOutcome, error) { - return repinRunIn(conn, runID, reason, nowMS, instanceConfigRoots()) + return RepinRunWith(conn, runID, RepinOptions{Reason: reason}, nowMS) +} + +// RepinRunWith is RepinRun with the DKT-582 dispositions. +func RepinRunWith( + conn *sql.DB, runID int, opts RepinOptions, nowMS int64, +) (*RepinOutcome, error) { + return repinRunOptsIn(conn, runID, opts, nowMS, instanceConfigRoots()) } // repinRunIn is RepinRun over an explicit root list — the same seam @@ -81,6 +179,14 @@ func RepinRun(conn *sql.DB, runID int, reason string, nowMS int64) (*RepinOutcom func repinRunIn( conn *sql.DB, runID int, reason string, nowMS int64, roots []string, ) (*RepinOutcome, error) { + return repinRunOptsIn(conn, runID, RepinOptions{Reason: reason}, nowMS, roots) +} + +// repinRunOptsIn is repinRunIn with dispositions. +func repinRunOptsIn( + conn *sql.DB, runID int, opts RepinOptions, nowMS int64, roots []string, +) (*RepinOutcome, error) { + reason := opts.Reason // The status gate runs BEFORE the disk walk so a repin of a finished run // refuses cleanly rather than reporting drift it would then refuse to fix. // It is re-checked inside the transaction below; this read exists for the @@ -104,23 +210,99 @@ func repinRunIn( return nil, err } - // MISSING REFUSES THE WHOLE SET. Repin's contract is "adopt current disk - // bytes", and a missing ref has no bytes to adopt; repinning the changed - // pins around it would report recovery while the run stays wedged on the - // rest. All-or-nothing is the same rule activation's own pinning follows - // ("pinning is never partial"). + // MISSING REFUSES THE WHOLE SET, UNLESS IT IS DROPPED. Repin's contract is + // "adopt current disk bytes", and a missing ref has no bytes to adopt; + // repinning the changed pins around it would report recovery while the run + // stays wedged on the rest. All-or-nothing is the same rule activation's own + // pinning follows ("pinning is never partial"). DKT-582 adds the one + // disposition that is not a partial recovery but a complete one: a ref + // nothing left to run can read is not a hole in the agreement, and retiring + // it — on the operator's explicit say-so — leaves the run whole. var missing []string // A changed verdict that repinning cannot fix: no resolved hash, or the // same hash (a schema that matches its pin byte-for-byte but no longer // compiles reports `changed` with Found == Pinned — new bytes are not the // remedy for that). var unfixable []string + // A ref covered by --drop/--drop-unresolvable that a non-terminal step can + // still read: the drop that would wedge the run, refused by name. + var stillNeeded []string var changes []RepinChange + var drops []RepinChange + + dropNamed := make(map[string]bool, len(opts.Drop)) + for _, ref := range opts.Drop { + dropNamed[normalizePinRef(ref)] = true + } + dropMatched := make(map[string]bool, len(opts.Drop)) + + // The pending closure is computed ONLY when a disposition could use it — a + // plain repin must not start reading step definitions and config files it + // has no question for. + var closure pendingClosure + if len(dropNamed) > 0 || opts.DropUnresolvable { + closure, err = pendingPacketClosure(conn, runID, roots) + if err != nil { + return nil, err + } + } + for _, v := range report.Pins { switch v.Status { case PinMissing: - missing = append(missing, fmt.Sprintf("%s %s", v.Kind, v.Ref)) + named := dropNamed[normalizePinRef(v.Ref)] + if named { + dropMatched[normalizePinRef(v.Ref)] = true + } + if !named && !opts.DropUnresolvable { + missing = append(missing, fmt.Sprintf("%s %s", v.Kind, v.Ref)) + continue + } + // ONLY FILE PINS ARE DROPPABLE. A workflow or schema pin names a + // REGISTERED object every step of the run is expanded from or + // validates against; the packet closure says nothing about those, + // so there is no reading of "unreferenced" that could justify + // retiring one. --drop-unresolvable leaves them to the refusal + // below rather than quietly widening its own meaning. + if v.Kind != db.PinKindFile { + if named { + return nil, validationErr( + "--drop %s names a %s pin; only file pins can be dropped — a "+ + "registered %s is what the run's steps are expanded from or "+ + "validated against, and no packet closure can say it is unread", + v.Ref, v.Kind, v.Kind) + } + missing = append(missing, fmt.Sprintf("%s %s", v.Kind, v.Ref)) + continue + } + // A file pin can also read `missing` because the file IS there and + // could not be READ — verifyFilePin reports the path in exactly that + // case and in no other. That is a permission to restore, not a + // deletion to retire, and retiring it would throw away the pin that + // still describes the file sitting on disk. + if v.Path != "" { + return nil, conflictErr( + "%s cannot be dropped: %s exists but could not be read; fix its "+ + "permissions — dropping retires a ref with no file at all", + v.Ref, v.Path) + } + if by := closure.referencedBy(v.Ref); len(by) > 0 { + stillNeeded = append(stillNeeded, fmt.Sprintf( + "%s (read by %s)", v.Ref, strings.Join(by, ", "))) + continue + } + drops = append(drops, RepinChange{ + Kind: v.Kind, Ref: v.Ref, + OldSHA256: v.Pinned, NewSHA256: "", Path: v.Path, Dropped: true, + }) case PinChanged: + if dropNamed[normalizePinRef(v.Ref)] { + return nil, conflictErr( + "--drop %s names a ref that still resolves (pinned %s, on disk %s); "+ + "drop retires a ref with no current bytes at all — repin adopts "+ + "this one's new bytes without it", + v.Ref, v.Pinned, v.Found) + } if v.Found == "" || v.Found == v.Pinned { unfixable = append(unfixable, fmt.Sprintf("%s %s", v.Kind, v.Ref)) continue @@ -129,13 +311,38 @@ func repinRunIn( Kind: v.Kind, Ref: v.Ref, OldSHA256: v.Pinned, NewSHA256: v.Found, Path: v.Path, }) + case PinOK: + if dropNamed[normalizePinRef(v.Ref)] { + return nil, conflictErr( + "--drop %s names a ref that still resolves and still matches its "+ + "pin; there is nothing to retire", v.Ref) + } + } + } + + // A --drop that matched no pin at all is the operator and the run + // disagreeing about what this run holds — never a silent no-op. + for _, ref := range opts.Drop { + if !dropMatched[normalizePinRef(ref)] { + return nil, notFoundErr(nil, + "--drop %s names a ref %s does not pin; run `docket run verify-pins %s` "+ + "for the refs this run actually holds", ref, run.Ref(), run.Ref()) } } + if len(stillNeeded) > 0 { + return nil, conflictErr( + "%s cannot drop %s: the ref(s) no longer resolve, but non-terminal steps "+ + "still read them — restore the file(s); dropping retires only a ref "+ + "nothing left to run can open", + run.Ref(), strings.Join(stillNeeded, "; ")) + } if len(missing) > 0 { return nil, notFoundErr(nil, "%s cannot be repinned: %s no longer resolve(s) at all — restore the "+ - "file(s) or abandon the run; repin adopts current disk bytes and "+ - "a missing ref has none", run.Ref(), strings.Join(missing, ", ")) + "file(s), abandon the run, or retire the ref(s) with `--drop`/"+ + "`--drop-unresolvable` if no pending step reads them; repin adopts "+ + "current disk bytes and a missing ref has none", + run.Ref(), strings.Join(missing, ", ")) } if len(unfixable) > 0 { return nil, conflictErr( @@ -144,11 +351,33 @@ func repinRunIn( run.Ref(), strings.Join(unfixable, ", "), run.Ref()) } + // THE ADOPTED BYTES' OWN CLOSURE (DKT-805). Adopting changed bytes adopts + // their packet closure too, and that closure can reach refs the run never + // pinned — the walk below finds them, so the transaction can pin them or + // this verb can refuse NOW, naming them, instead of reporting success and + // letting the next render refuse with "start a new run". Computed only when + // bytes were actually adopted: a drop cannot introduce a reference, and a + // no-op repin must stay a no-op. + var adds []RepinChange + if len(changes) > 0 { + if closure == nil { + closure, err = pendingPacketClosure(conn, runID, roots) + if err != nil { + return nil, err + } + } + adds, err = closureAdditions(report, closure, roots, run.Ref()) + if err != nil { + return nil, err + } + } + outcome := &RepinOutcome{ - Run: run.Ref(), Repinned: []RepinChange{}, - Unchanged: len(report.Pins) - len(changes), + Run: run.Ref(), Repinned: []RepinChange{}, Dropped: []RepinChange{}, + Added: []RepinChange{}, + Unchanged: len(report.Pins) - len(changes) - len(drops), } - if len(changes) == 0 { + if len(changes) == 0 && len(drops) == 0 { // Nothing drifted: a clean no-op, so running repin twice is safe and // the second run says so instead of inventing an event. return outcome, nil @@ -213,12 +442,148 @@ func repinRunIn( outcome.Repinned = append(outcome.Repinned, c) } + for _, d := range drops { + // The DELETE is keyed on the OLD hash for the same reason the UPDATE is: + // a pin that moved between the disk walk and this write is an agreement + // this verb never inspected, and retiring it blind would be exactly the + // silent removal the opt-in gate exists to prevent. + res, err := tx.Exec( + `DELETE FROM pins WHERE run_id = ? AND kind = ? AND ref = ? AND sha256 = ?`, + runID, d.Kind, d.Ref, d.OldSHA256, + ) + if err != nil { + return nil, fmt.Errorf("dropping %s %s: %w", d.Kind, d.Ref, err) + } + if n, err := res.RowsAffected(); err == nil && n == 0 { + return nil, conflictErr( + "%s %s changed while repinning %s; re-run `docket run verify-pins %s` and retry", + d.Kind, d.Ref, fresh.Ref(), fresh.Ref()) + } + + // A NULL `new_sha256` is the drop's signature in the trail: the same + // event kind, the same old sha preserved for completed steps' + // provenance, and the one field that says there are no current bytes + // rather than these current bytes. `dropped` states it in a form a + // reader can branch on without inferring intent from an absence. + data, err := json.Marshal(map[string]any{ + "kind": d.Kind, + "ref": d.Ref, + "old_sha256": d.OldSHA256, + "new_sha256": nil, + "path": d.Path, + "dropped": true, + "reason": reason, + }) + if err != nil { + return nil, fmt.Errorf("recording the drop of %s %s: %w", d.Kind, d.Ref, err) + } + if err := recordEvent(tx, eventRecord{ + Kind: EventRunRepinned, RunID: runID, Data: string(data), AtMS: nowMS, + }); err != nil { + return nil, err + } + outcome.Dropped = append(outcome.Dropped, d) + } + + for _, a := range adds { + // A plain INSERT would race a concurrent activation's `INSERT OR + // IGNORE` of the same ref; the OR IGNORE plus the zero-rows check makes + // the collision a visible CONFLICT rather than either a constraint + // error or a silent adoption of a hash this verb never inspected — + // the same discipline the UPDATE's compare-and-swap enforces. + res, err := tx.Exec( + `INSERT OR IGNORE INTO pins (run_id, kind, ref, sha256) VALUES (?, ?, ?, ?)`, + runID, a.Kind, a.Ref, a.NewSHA256, + ) + if err != nil { + return nil, fmt.Errorf("pinning %s %s: %w", a.Kind, a.Ref, err) + } + if n, err := res.RowsAffected(); err == nil && n == 0 { + return nil, conflictErr( + "%s %s changed while repinning %s; re-run `docket run verify-pins %s` and retry", + a.Kind, a.Ref, fresh.Ref(), fresh.Ref()) + } + + // A NULL `old_sha256` is the addition's signature in the trail, the + // mirror of the drop's null `new_sha256`: same event kind, and the one + // field that says the run held no bytes for this ref before. `added` + // states it in a form a reader can branch on, and `required_by` names + // what reaches the ref, so provenance says WHY the agreement grew. + data, err := json.Marshal(map[string]any{ + "kind": a.Kind, + "ref": a.Ref, + "old_sha256": nil, + "new_sha256": a.NewSHA256, + "path": a.Path, + "added": true, + "required_by": closure.referencedBy(a.Ref), + "reason": reason, + }) + if err != nil { + return nil, fmt.Errorf("recording the pin of %s %s: %w", a.Kind, a.Ref, err) + } + if err := recordEvent(tx, eventRecord{ + Kind: EventRunRepinned, RunID: runID, Data: string(data), AtMS: nowMS, + }); err != nil { + return nil, err + } + outcome.Added = append(outcome.Added, a) + } + if err := tx.Commit(); err != nil { return nil, fmt.Errorf("repinning %s: %w", run.Ref(), err) } return outcome, nil } +// closureAdditions computes the pins a repin must CREATE: every file ref the +// pending packet closure reaches that the run does not already hold, resolved +// and hashed at its current disk bytes (DKT-805). +// +// The walk itself is unpinnedClosureRefs, shared verbatim with `verify-pins`' +// closure check (DKT-821) — the detection verb and the recovery verb must never +// disagree about which refs a pin set is missing, which is the disagreement +// RUN-59 was. +// +// It refuses — before any write — when a reachable ref has no bytes on disk at +// all: there is nothing to pin, and a repin that proceeded around it would +// report success while the steps reading it stay unrenderable, which is the +// exact outcome this walk exists to close. +func closureAdditions( + report *PinReport, closure pendingClosure, roots []string, runRef string, +) ([]RepinChange, error) { + var adds []RepinChange + var unpinnable []string + for _, u := range unpinnedClosureRefs(report.Pins, closure, roots) { + if u.readErr != nil { + return nil, conflictErr( + "%s cannot be repinned: the adopted bytes require %q, and %s "+ + "exists but could not be read; fix its permissions", + runRef, u.ref, u.path) + } + if u.path == "" { + unpinnable = append(unpinnable, fmt.Sprintf( + "%s (read by %s)", u.ref, strings.Join(u.requiredBy, ", "))) + continue + } + adds = append(adds, RepinChange{ + Kind: db.PinKindFile, Ref: u.ref, + OldSHA256: "", NewSHA256: u.sha256, + Path: u.path, Added: true, + }) + } + if len(unpinnable) > 0 { + return nil, notFoundErr(nil, + "%s cannot be repinned: the adopted bytes require ref(s) this run does "+ + "not pin and that do not resolve on disk — %s; restore the file(s) "+ + "under an instance-config root, or abandon the run and re-plan; a "+ + "repin that proceeded would leave those steps unable to render "+ + "their packets", + runRef, strings.Join(unpinnable, "; ")) + } + return adds, nil +} + // repinStatusGuard refuses every run status with no pending-side work. // // `done` and `abandoned` are the acceptance criterion's hard case stated as a @@ -304,11 +669,17 @@ func repinQuiescenceGuard(tx *sql.Tx, runID int, runRef string) error { // surface states when it warns (DKT-408's remedy 2). nil when every pin is // sound, so callers can gate rendering on emptiness alone. // -// It is VerifyPins minus the sound rows rather than its own walk, so the +// It is the pin check minus the sound rows rather than its own walk, so the // warning a surface prints and the report `verify-pins` exits 4 on can never // name different pins. +// +// It is the PER-PIN half only. `verify-pins` also reports refs the pinned bytes +// reference and the pin set does not hold (DKT-821), and that question has no +// per-pin verdict to render here — it is the verb's answer, not a drift line — +// so this advisory keeps to the pins and keeps to a hash-per-pin of work on +// every `run status`. func PinDrift(conn *sql.DB, runID int) ([]PinVerdict, error) { - report, err := VerifyPins(conn, runID) + report, err := verifyPinsIn(conn, runID, instanceConfigRoots()) if err != nil { return nil, err } diff --git a/internal/engine/repin_drop_test.go b/internal/engine/repin_drop_test.go new file mode 100644 index 00000000..cb8542cc --- /dev/null +++ b/internal/engine/repin_drop_test.go @@ -0,0 +1,462 @@ +package engine + +import ( + "database/sql" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-582: a pinned ref that no longer RESOLVES had exactly one disposition — +// refuse the whole repin — and corpus commits that delete a contract are as +// ordinary as commits that edit one. RUN-42 died on cc92e38/93ed1e9 deleting +// `contracts/test-infra.md` and `contracts/pr-comment-author.md`; RUN-36 died +// the same way after surviving two earlier repins. Neither had a step left that +// would ever open either file. +// +// These tests pin the new disposition and its guard: a NOT_FOUND file pin no +// NON-TERMINAL step's packet closure reaches can be retired on the operator's +// say-so, and one a pending step still reads cannot be, however it is asked for. + +// dropWorkflowSrc declares one packet file per step, so "referenced by a +// pending step" and "referenced only by a completed step" are two different +// files rather than two readings of one. +const dropWorkflowSrc = ` +[pipeline] +name = "drop-dev" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" +after = [] +packet = ["contracts/implement.md"] + +[[step]] +name = "review" +after = ["implement"] +executor = "review" +emits = "findings" +inputs = ["implement.change-summary"] +packet = ["contracts/review.md"] +` + +// dropRun activates drop-dev and completes `implement`, leaving `review` +// pending. From here `contracts/implement.md` is read only by history and +// `contracts/review.md` only by work still to come. +func dropRun(t *testing.T) (conn *sql.DB, configDir string, runID int) { + t.Helper() + conn, configDir = configRepo(t) + writeConfigFile(t, configDir, "workflows/drop-dev.toml", dropWorkflowSrc) + writeConfigFile(t, configDir, "contracts/implement.md", "the implement contract\n") + writeConfigFile(t, configDir, "contracts/review.md", "the review contract\n") + writeConfigFile(t, configDir, "policy.toml", "opaque = \"instance policy\"\n") + + issue := createIssue(t, conn, "drop subject", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + // implement records, with the artifact review consumes — the completed + // provenance that must survive every drop below. + doneID := stepID(t, conn, run.ID, "implement@0") + execSQL(t, conn, `UPDATE steps SET status = ? WHERE id = ?`, db.StepDone, doneID) + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + _, err = db.InsertArtifactTx(tx, db.Artifact{ + RunID: run.ID, StepID: doneID, Kind: "change-summary", + Body: "did the thing", SHA256: "abc123", + }, nowMS) + testsupport.Must(t, err, "InsertArtifactTx: %v", err) + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + + return conn, configDir, run.ID +} + +// pinCount counts a run's pin rows naming a ref — 0 after a drop, 1 otherwise. +func pinCount(t *testing.T, conn *sql.DB, runID int, ref string) int { + t.Helper() + var n int + err := conn.QueryRow( + `SELECT COUNT(*) FROM pins WHERE run_id = ? AND ref = ?`, runID, ref).Scan(&n) + testsupport.Must(t, err, "counting pin %s: %v", ref, err) + return n +} + +// deleteConfigFile is the corpus commit these tests are about. +func deleteConfigFile(t *testing.T, configDir, rel string) { + t.Helper() + testsupport.Must(t, os.Remove(filepath.Join(configDir, rel)), "deleting %s", rel) +} + +// TestRepinDropsADeletedRefNoPendingStepReads is the acceptance criterion +// verbatim: a run whose only drifted pins are deleted refs unreferenced by +// pending steps can be repinned and resumed. Both spellings of the ask — +// --drop-unresolvable and an explicit --drop — reach it. +func TestRepinDropsADeletedRefNoPendingStepReads(t *testing.T) { + cases := []struct { + name string + opts RepinOptions + }{ + {"--drop-unresolvable", RepinOptions{ + Reason: "cc92e38 deleted it", DropUnresolvable: true}}, + {"--drop by name", RepinOptions{ + Reason: "cc92e38 deleted it", Drop: []string{"contracts/implement.md"}}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + conn, configDir, runID := dropRun(t) + pinned := pinSHA(t, conn, runID, "contracts/implement.md") + + // The corpus commit: the completed step's contract is deleted. It is + // the run's ONLY drift, and no pending step's packet names it. + deleteConfigFile(t, configDir, "contracts/implement.md") + before, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if before.Missing != 1 || before.Changed != 0 { + t.Fatalf("premise: want exactly one missing pin and no changed one, "+ + "got %d missing / %d changed: %s", + before.Missing, before.Changed, PinReportReason(before)) + } + + outcome, err := RepinRunWith(conn, runID, tc.opts, nowMS) + testsupport.Must(t, err, "RepinRunWith: %v", err) + + if len(outcome.Repinned) != 0 { + t.Errorf("repinned %+v; a deleted ref has no bytes to adopt", + outcome.Repinned) + } + if len(outcome.Dropped) != 1 { + t.Fatalf("dropped %d pin(s), want 1: %+v", len(outcome.Dropped), + outcome.Dropped) + } + d := outcome.Dropped[0] + if d.Ref != "contracts/implement.md" || d.OldSHA256 != pinned || + d.NewSHA256 != "" || !d.Dropped { + t.Errorf("drop = %+v, want contracts/implement.md %s -> (nothing), "+ + "marked dropped", d, pinned) + } + + // THE PIN IS RETIRED, not carried as perpetual drift: the row is gone, + // so verify-pins has nothing left to fail on and the run is sound. + if n := pinCount(t, conn, runID, "contracts/implement.md"); n != 0 { + t.Errorf("%d pin row(s) still name the dropped ref; verify-pins would "+ + "report it forever", n) + } + after, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins after the drop: %v", err) + if !after.Sound() { + t.Errorf("the run is still unsound after the drop: %s", + PinReportReason(after)) + } + + // THE OLD SHA SURVIVES IN THE TRAIL, with a NULL new_sha256 saying + // there are no current bytes rather than naming some. + events := repinEvents(t, conn, runID) + if len(events) != 1 { + t.Fatalf("recorded %d run-repinned event(s), want 1", len(events)) + } + e := events[0] + if e["ref"] != "contracts/implement.md" || e["old_sha256"] != pinned { + t.Errorf("event = %v, want contracts/implement.md at %s", e, pinned) + } + if e["new_sha256"] != nil { + t.Errorf("event new_sha256 = %v, want null — a drop has no new bytes", + e["new_sha256"]) + } + if e["dropped"] != true { + t.Errorf("event dropped = %v, want true; a reader must not have to "+ + "infer the disposition from an absence", e["dropped"]) + } + if e["reason"] != tc.opts.Reason { + t.Errorf("event reason = %v, want the operator's --reason verbatim", + e["reason"]) + } + + // AND THE RUN RESUMES. The pending step's packet still resolves, and + // the next wave's manifest offers it with no drift advisory. + steps, err := db.ListRunSteps(conn, runID) + testsupport.Must(t, err, "ListRunSteps: %v", err) + defs, err := StepDefinitions(conn, runID) + testsupport.Must(t, err, "StepDefinitions: %v", err) + for _, s := range steps { + if s.Instance != "review@0" { + continue + } + files, ferr := stepPacketFiles(conn, s, stepSpec(defs, s, holdTally{}), "") + testsupport.Must(t, ferr, "rendering review@0's packet: %v", ferr) + if len(files) != 1 || files[0].Path != "contracts/review.md" { + t.Errorf("review@0's packet = %+v, want its own contract", files) + } + } + m := openDispatch(t, conn, runID, 0, nowMS) + if len(m.PinDrift) != 0 { + t.Errorf("the manifest still advises drift after the drop: %+v", + m.PinDrift) + } + offered := false + for _, r := range m.Rows { + if r.Instance == "review@0" { + offered = true + } + } + if !offered { + t.Errorf("the next wave offers %+v; the pending step must be "+ + "claimable again", m.Rows) + } + }) + } +} + +// TestRepinRefusesToDropARefAPendingStepReads is the guard: dropping is opt-in, +// and opting in never buys the removal of something still needed. Both flag +// spellings refuse, and the refusal names the step. +func TestRepinRefusesToDropARefAPendingStepReads(t *testing.T) { + cases := []struct { + name string + opts RepinOptions + }{ + {"--drop-unresolvable", RepinOptions{ + Reason: "the install", DropUnresolvable: true}}, + {"--drop by name", RepinOptions{ + Reason: "the install", Drop: []string{"contracts/review.md"}}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + conn, configDir, runID := dropRun(t) + pinned := pinSHA(t, conn, runID, "contracts/review.md") + + // The deletion that WOULD wedge the run: review@0 is pending and its + // packet declares exactly this file. + deleteConfigFile(t, configDir, "contracts/review.md") + + _, err := RepinRunWith(conn, runID, tc.opts, nowMS) + if err == nil { + t.Fatal("the repin dropped a ref a pending step still reads") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + for _, want := range []string{"contracts/review.md", "review@0"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("refusal %q does not name %q", err, want) + } + } + if got := pinSHA(t, conn, runID, "contracts/review.md"); got != pinned { + t.Errorf("the pin moved to %q under a refusal; nothing may change", got) + } + if events := repinEvents(t, conn, runID); len(events) != 0 { + t.Errorf("%d run-repinned event(s) recorded by a refused repin", + len(events)) + } + }) + } +} + +// TestRepinDropsAndUpdatesInOneCall is the mixed case: one drifted ref resolves +// to new bytes and one no longer resolves at all. The first is adopted, the +// second retired, and the run comes out sound — a NOT_FOUND ref must not cost +// the changed ones their recovery. +func TestRepinDropsAndUpdatesInOneCall(t *testing.T) { + conn, configDir, runID := dropRun(t) + deletedPin := pinSHA(t, conn, runID, "contracts/implement.md") + changedPin := pinSHA(t, conn, runID, "contracts/review.md") + policyPin := pinSHA(t, conn, runID, policyPinRef) + + deleteConfigFile(t, configDir, "contracts/implement.md") + rewriteConfigFile(t, configDir, "contracts/review.md", "the install's new contract\n") + + outcome, err := RepinRunWith(conn, runID, RepinOptions{ + Reason: "corpus install: one edit, one deletion", DropUnresolvable: true, + }, nowMS) + testsupport.Must(t, err, "RepinRunWith: %v", err) + + if len(outcome.Repinned) != 1 || outcome.Repinned[0].Ref != "contracts/review.md" { + t.Fatalf("repinned %+v, want the one ref that resolves to new bytes", + outcome.Repinned) + } + if outcome.Repinned[0].OldSHA256 != changedPin || + outcome.Repinned[0].NewSHA256 == changedPin { + t.Errorf("the changed ref's pin did not move: %+v", outcome.Repinned[0]) + } + if len(outcome.Dropped) != 1 || outcome.Dropped[0].Ref != "contracts/implement.md" { + t.Fatalf("dropped %+v, want the one deleted ref", outcome.Dropped) + } + if outcome.Dropped[0].OldSHA256 != deletedPin { + t.Errorf("the drop event's old sha is %q, want %q — the completed step's "+ + "agreement rides on it", outcome.Dropped[0].OldSHA256, deletedPin) + } + + // The sound pin was not touched by either disposition. + if got := pinSHA(t, conn, runID, policyPinRef); got != policyPin { + t.Errorf("the sound pin moved %s -> %s", policyPin, got) + } + report, err := VerifyPins(conn, runID) + testsupport.Must(t, err, "VerifyPins: %v", err) + if !report.Sound() { + t.Errorf("the run is still unsound after the mixed repin: %s", + PinReportReason(report)) + } + + // One event per ref, each self-contained, and the two dispositions are + // distinguishable in the trail. + events := repinEvents(t, conn, runID) + if len(events) != 2 { + t.Fatalf("recorded %d run-repinned event(s), want one per ref", len(events)) + } + seen := map[string]any{} + for _, e := range events { + ref, _ := e["ref"].(string) + seen[ref] = e["new_sha256"] + } + if seen["contracts/implement.md"] != nil { + t.Errorf("the dropped ref's event carries new_sha256 %v, want null", + seen["contracts/implement.md"]) + } + if seen["contracts/review.md"] == nil { + t.Errorf("the updated ref's event carries a null new_sha256; only a drop does") + } +} + +// TestRepinRefusesADropThatDescribesTheWrongThing: --drop is a claim about the +// run's state, and a claim the run contradicts is a refusal rather than a +// silent no-op — the operator and the run disagree about what is broken. +func TestRepinRefusesADropThatDescribesTheWrongThing(t *testing.T) { + t.Run("the named ref still resolves", func(t *testing.T) { + conn, configDir, runID := dropRun(t) + rewriteConfigFile(t, configDir, "contracts/review.md", "edited, not deleted\n") + + _, err := RepinRunWith(conn, runID, RepinOptions{ + Reason: "r", Drop: []string{"contracts/review.md"}}, nowMS) + if err == nil { + t.Fatal("--drop retired a ref that still resolves") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + if pinCount(t, conn, runID, "contracts/review.md") != 1 { + t.Error("the pin was retired under a refusal") + } + }) + + t.Run("the named ref is there but unreadable", func(t *testing.T) { + if os.Geteuid() == 0 { + t.Skip("root reads a 0000 file, so the unreadable case cannot be staged") + } + conn, configDir, runID := dropRun(t) + path := filepath.Join(configDir, "contracts/implement.md") + testsupport.Must(t, os.Chmod(path, 0o000), "chmod") + t.Cleanup(func() { _ = os.Chmod(path, 0o644) }) + + _, err := RepinRunWith(conn, runID, RepinOptions{ + Reason: "r", DropUnresolvable: true}, nowMS) + if err == nil { + t.Fatal("a file that exists but cannot be read was retired as deleted") + } + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("error code = %v, want CONFLICT: %v", code, err) + } + if pinCount(t, conn, runID, "contracts/implement.md") != 1 { + t.Error("the pin was retired for a file still sitting on disk") + } + }) + + t.Run("the named ref is not pinned at all", func(t *testing.T) { + conn, configDir, runID := dropRun(t) + deleteConfigFile(t, configDir, "contracts/implement.md") + + _, err := RepinRunWith(conn, runID, RepinOptions{ + Reason: "r", Drop: []string{"contracts/never-pinned.md"}}, nowMS) + if err == nil { + t.Fatal("--drop accepted a ref this run does not pin") + } + if code, ok := CodeOf(err); !ok || code != CodeNotFound { + t.Errorf("error code = %v, want NOT_FOUND: %v", code, err) + } + if !strings.Contains(err.Error(), "contracts/never-pinned.md") { + t.Errorf("refusal %q does not name the ref", err) + } + }) +} + +// TestRepinNeverDropsARegisteredObjectPin: --drop-unresolvable is about FILES. +// A workflow or schema pin names a registered object every step is expanded +// from or validated against, and no packet closure can call one unread — so a +// missing one still refuses, and naming it explicitly is a validation error +// rather than a wider reading of the flag. +func TestRepinNeverDropsARegisteredObjectPin(t *testing.T) { + conn, _, runID := dropRun(t) + + // A schema pin whose registry row does not exist: verify-pins reports it + // missing, exactly as a de-registered schema would. + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + testsupport.Must(t, db.InsertPinTx(tx, db.Pin{ + RunID: runID, Kind: db.PinKindSchema, Ref: "gone@1", SHA256: "deadbeef", + }), "InsertPinTx") + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + + _, err = RepinRunWith(conn, runID, RepinOptions{ + Reason: "r", DropUnresolvable: true}, nowMS) + if err == nil { + t.Fatal("--drop-unresolvable retired a schema pin") + } + if code, ok := CodeOf(err); !ok || code != CodeNotFound { + t.Errorf("error code = %v, want NOT_FOUND: %v", code, err) + } + if !strings.Contains(err.Error(), "gone@1") { + t.Errorf("refusal %q does not name the missing schema", err) + } + + _, err = RepinRunWith(conn, runID, RepinOptions{ + Reason: "r", Drop: []string{"gone@1"}}, nowMS) + if err == nil { + t.Fatal("--drop retired a schema pin") + } + if code, ok := CodeOf(err); !ok || code != CodeValidation { + t.Errorf("error code = %v, want VALIDATION_ERROR: %v", code, err) + } + if pinCount(t, conn, runID, "gone@1") != 1 { + t.Error("the schema pin was retired under a refusal") + } +} + +// TestPendingClosureCoversIncludesAndSpares completed-only refs: the walk that +// decides a drop must follow `packet_includes` from a PENDING step (or the +// fragment under a still-needed contract would look unread), and must not +// follow them from a terminal one (or nothing would ever be droppable). +func TestPendingClosureCoversIncludesAndSparesCompletedOnlyRefs(t *testing.T) { + conn, configDir, runID := dropRun(t) + rewriteConfigFile(t, configDir, "contracts/review.md", + "---\npacket_includes:\n - fragments/style.md\n---\nthe review contract\n") + writeConfigFile(t, configDir, "fragments/style.md", "the included fragment\n") + + closure, err := pendingPacketClosure(conn, runID, instanceConfigRoots()) + testsupport.Must(t, err, "pendingPacketClosure: %v", err) + + if by := closure.referencedBy("contracts/review.md"); len(by) == 0 || + by[0] != "review@0" { + t.Errorf("contracts/review.md is reached by %v, want review@0", by) + } + if by := closure.referencedBy("fragments/style.md"); len(by) == 0 { + t.Error("a fragment reachable only through a pending step's " + + "packet_includes reads as unreferenced; dropping it would wedge the render") + } + if by := closure.referencedBy("contracts/implement.md"); len(by) != 0 { + t.Errorf("contracts/implement.md is reached by %v, but only a terminal "+ + "step declares it", by) + } + // policy.toml is the harness's, not any step's, and no step-sourced walk + // could find it — so it is never droppable. + if by := closure.referencedBy(policyPinRef); len(by) == 0 { + t.Error("policy.toml reads as unreferenced; the harness resolves policy from it") + } +} diff --git a/internal/engine/repin_test.go b/internal/engine/repin_test.go index c7255358..faaed6ac 100644 --- a/internal/engine/repin_test.go +++ b/internal/engine/repin_test.go @@ -454,12 +454,21 @@ func TestDispatchOpenNamesPinDrift(t *testing.T) { testsupport.Must(t, os.MkdirAll(filepath.Join(configRoot, "contracts"), 0o755), "mkdir") testsupport.Must(t, os.WriteFile( filepath.Join(configRoot, "contracts", "x.md"), []byte("OLD\n"), 0o644), "write") + // The workflow DECLARES the contract, so it is in DKT-581's pin closure — + // an undeclared file would (correctly) no longer be pinned at all. + testsupport.Must(t, os.MkdirAll(filepath.Join(configRoot, "workflows"), 0o755), "mkdir") + testsupport.Must(t, os.WriteFile( + filepath.Join(configRoot, "workflows", "auto-dev.toml"), + []byte(autoWorkflowSrc+"packet = [\"contracts/x.md\"]\n"), 0o644), "write") t.Setenv("DOCKET_PATH", store) conn := mustDB(t) - // Activation scans the config root and pins contracts/x.md itself — the - // same path a real run's pin takes. - run, _ := activatedRun(t, conn) + // Activation scans the config root, registers the workflow, and pins + // contracts/x.md itself — the same path a real run's pin takes. + issue := createIssue(t, conn, "drift subject", "a body", "task", nil) + run := startRun(t, conn, issue) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) report, err := VerifyPins(conn, run.ID) testsupport.Must(t, err, "VerifyPins: %v", err) if !report.Sound() { diff --git a/internal/engine/report.go b/internal/engine/report.go index 7c4eb945..edbb944b 100644 --- a/internal/engine/report.go +++ b/internal/engine/report.go @@ -69,8 +69,29 @@ type RunBudgetReport struct { UsageCap float64 `json:"usage_cap,omitempty"` UsageUnit string `json:"usage_budget_unit,omitempty"` UsageSpend float64 `json:"usage_spend,omitempty"` + + // VoteUsageNote states, on any run whose panels cast, that the seats' + // measured spend (the report's `vote_usage` section) is EXCLUDED from + // `reported` and `spend` — and why (DKT-584). + // + // It is a note rather than a fold because `Reported` must remain exactly + // the rows enforcement's own snapshot sums (the step usage_ledger): the + // report publishing a "reported" no decision was made against is the + // two-sources-of-truth failure RunFloorTx's export exists to prevent. The + // seats' DECLARED cost reaches the budget through the floor instead — a + // vote step's expected_cost accrues at materialization — so the panel is + // not invisible; its measured spend is simply accounted in its own + // section, and this line says so instead of leaving the omission silent. + VoteUsageNote string `json:"vote_usage_note,omitempty"` } +// VoteUsageExcludedNote is VoteUsageNote's one value, a constant so the JSON +// document and the rendered report cannot say it differently. +const VoteUsageExcludedNote = "vote_usage is excluded from reported and spend: " + + "those sum the step usage_ledger the cap enforcement reads, and seat casts " + + "land in the separate vote_usage ledger; a vote step's declared " + + "expected_cost accrues to the floor instead" + // RunReport is the whole document — R1 through R7, in that order. type RunReport struct { Run *model.Run `json:"run"` @@ -82,6 +103,28 @@ type RunReport struct { Budget RunBudgetReport `json:"budget"` + // PinnedWorkflows is DKT-594's first half: per pinned workflow, how many + // registered versions the corpus has advanced since this run froze. + // + // It rides in the report rather than in `verify-pins` because the question + // is not about DRIFT. A pinned `ui-change@8` whose file is byte-identical to + // what the registry holds is perfectly sound and can still be five versions + // behind, and `verify-pins` — which compares hashes at one ref — is right to + // call it `ok`. What a post-mortem reader needs before trusting a finding is + // the other number, and until this section it existed in no read verb: every + // analyst on RUN-32 recovered it from git by hand. + PinnedWorkflows []PinnedWorkflowStaleness `json:"pinned_workflows,omitempty"` + + // PinEpochs is DKT-594's second half: the run's pin-agreement timeline, + // PRESENT ONLY ON A RUN THAT REPINNED. + // + // A run whose agreement never moved has one epoch, every step ran under it, + // and the `pins` table already says what it was — so the section is absent + // and each step's `pin_epoch` is absent with it, rather than a column of 1s + // on every report in the store. Where it IS present it is what RUN-39's + // analysts assembled by hand from event seqs and step ids. + PinEpochs []PinEpoch `json:"pin_epochs,omitempty"` + // Steps is R3: the count by EFFECTIVE status, computed at read. Steps []model.StatusCount `json:"steps,omitempty"` // Attempts is R3's other half: per-step attempts, ordered by instance. @@ -157,6 +200,17 @@ type RunReport struct { // like a zero. VoteUsageCoverage db.VoteUsageCoverage `json:"vote_usage_coverage"` + // SilentVoteSeats is the identity behind VoteUsageCoverage.Silent + // (DKT-733): each cast that reported no spend — which seat, on which + // proposal, seated via which path. The count alone told an operator that + // seats went silent on a run and nothing said WHICH, so `vote + // backfill-usage` — the verb that exists to close exactly this gap + // (DKT-115) — could not be aimed without spelunking proposals by hand. + // + // `omitempty`: a run whose every seat reported carries no key, because the + // coverage line already says so and an empty list would restate it. + SilentVoteSeats []SilentVoteSeat `json:"silent_vote_seats,omitempty"` + // StepUsage is the ledger row by row — which step, which attempt, which // unit, how much, and who measured it. Budget.Reported is the same rows // summed per unit; this is the detail behind that headline. @@ -222,6 +276,56 @@ type StepAttempt struct { // never opened — which is exactly the never-convened case the reader needs // to tell apart. Vote string `json:"vote,omitempty"` + // PinEpoch is WHICH PIN AGREEMENT this step's recorded work ran under + // (DKT-594), indexing RunReport.PinEpochs. + // + // Absent unless the run actually repinned, and absent on a step that has not + // run — see PinEpochs and stepReportsAnEpoch. On a run whose agreement moved + // mid-flight it is the field that says which bytes a completed step + // consumed: `pins` holds only the CURRENT agreement, completed steps' rows + // are never rewritten, and correlating the two was the hand-join RUN-39's + // post-mortem performed against event seqs (5375/5376 vs STEP-1350/1353). + PinEpoch int `json:"pin_epoch,omitempty"` + + // Metadata is THIS STEP'S WHOLE BAG, verbatim (DKT-868) — the detail behind + // RunReport.Metadata exactly as StepUsage is the detail behind + // Budget.Reported. + // + // The rollup answers "which values did this key take, and how often". It + // cannot answer "which values did two keys take TOGETHER on one step", + // because grouping by key is precisely what discards the pairing. Any bag + // whose keys are a REQUEST and its RESOLUTION — the shape the corpus + // actually writes — is therefore unaggregatable from the report: RUN-51's + // rollup showed one key with no `low` value and its partner with one, a + // mismatch on exactly one step that no reader could name. Recovering it + // meant `step show` per step, and the audit that motivated this ran ~90 of + // them across 19 runs. + // + // It rides on the ATTEMPT ROW rather than in a section of its own so the + // bag arrives already joined to the four facts that make it interpretable: + // effective status, routing, attempt count and issue. That is what closes + // the other half of the gap — a step that FAILED or was reaped carries only + // what its dispatcher recorded at claim (`step claim --metadata`, DKT-592), + // and in a rollup that half-bag is indistinguishable from a completed + // step's, so drift concentrated in failures reads as no drift at all. + // + // CORE READS NO KEY HERE, as everywhere (docs/design/genericity.md, R7). It + // publishes the bag; what a pair of keys MEANS — a tier, a variant, a desk + // — stays the workflow author's business, and the consumer does the + // comparison core must not learn how to make. + Metadata map[string]any `json:"metadata,omitempty"` + + // MetadataUnreadable marks a step whose stored bag exists but does not + // decode, so an absent `metadata` is never silently read as "the dispatcher + // recorded nothing" — the exact ambiguity DKT-868 is about. It mirrors + // model.Vote's field of the same name and the same purpose. + // + // The bag is NOT re-validated here: a read verb that refused because one + // row held odd bytes would be useless during exactly the run an operator + // wants to inspect (R10), which is why db.MetadataRollup skips such a row + // too. This row says so out loud rather than skipping silently, because a + // per-step row IS the row — there are no other rows to carry the fact. + MetadataUnreadable bool `json:"metadata_unreadable,omitempty"` } // DispositionAbandoned is the one issue-level terminal ruling core records as @@ -339,12 +443,28 @@ func LoadRunReport(conn *sql.DB, runID int, nowMS int64) (*RunReport, error) { conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID)); err != nil { return nil, err } + // DKT-584: the vote-step key family alone missed every panel the run's + // machinery convened OUTSIDE a vote step — reap-ack ballots, and the + // conversational gates (activation panels and the like) whose only link to + // the run is that their text names it. Their casts appeared in NO run + // section at all. The extra ids widen the usage rollup and its coverage + // line to those proposals; the vote-step sections above are unchanged. + extraProposalIDs, err := conversationalRunProposalIDs(conn, runID) + if err != nil { + return nil, err + } if report.VoteUsage, err = db.VoteUsageRollup( - conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID)); err != nil { + conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID), + extraProposalIDs...); err != nil { return nil, err } if report.VoteUsageCoverage, err = db.VoteUsageCoverageFor( - conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID)); err != nil { + conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID), + extraProposalIDs...); err != nil { + return nil, err + } + if report.SilentVoteSeats, err = silentVoteSeats( + conn, runID, extraProposalIDs); err != nil { return nil, err } if report.StepUsage, err = db.UsageByStep(conn, runID); err != nil { @@ -353,6 +473,14 @@ func LoadRunReport(conn *sql.DB, runID int, nowMS int64) (*RunReport, error) { if report.Artifacts, err = artifactIndex(conn, runID); err != nil { return nil, err } + // DKT-594's staleness diff: the run's workflow pins against the registry's + // current head. Up here with the other pool reads for the reason stated + // above — it needs no snapshot, and reading it from inside the transaction + // below would deadlock the one-connection pool. + if report.PinnedWorkflows, err = pinnedWorkflowStaleness( + conn, run.ProjectID, runID); err != nil { + return nil, err + } // ---- R1/R2/R3, over ONE consistent snapshot. --------------------------- // @@ -412,6 +540,11 @@ func LoadRunReport(conn *sql.DB, runID int, nowMS int64) (*RunReport, error) { if err := annotateVoteOutcomes(tx, runID, sched, attempts); err != nil { return nil, err } + // DKT-594: which agreement each step's recorded work ran under, in the SAME + // snapshot as the statuses that decide whether a step ran at all. + if report.PinEpochs, err = annotatePinEpochs(tx, runID, attempts); err != nil { + return nil, err + } report.Steps, report.Attempts = counts, attempts // The issue-level rulings, read in the SAME snapshot as the steps they @@ -444,6 +577,12 @@ func LoadRunReport(conn *sql.DB, runID int, nowMS int64) (*RunReport, error) { BurnRate: burnRate(floor, report.WallClockMS), BreachReason: facts.BreachReason, } + // The exclusion is stated whenever there is anything to exclude: a cast + // happened, whether or not its seat reported spend (DKT-584). A run with + // no panels carries no note — there is nothing being left out. + if report.VoteUsageCoverage.Casts > 0 || len(report.VoteUsage) > 0 { + report.Budget.VoteUsageNote = VoteUsageExcludedNote + } // The transaction is rolled back by the deferred call and never committed: // there is nothing to commit, and R8's zero-write property is that @@ -507,6 +646,20 @@ func effectiveStepFacts(sched *Scheduler) ([]model.StatusCount, []StepAttempt) { if step.IssueID != 0 { row.Issue = model.FormatID(step.IssueID) } + // DKT-868: the step's own bag rides with its status. + // + // TOLERANT, NOT SILENT. A stored bag that does not decode leaves + // `Metadata` nil and sets the flag beside it — the R10 tolerance + // db.MetadataRollup already applies to the same column (a read verb must + // not refuse because one row holds odd bytes), without the rollup's + // freedom to drop the row and let the other rows carry the answer. + // + // The decode names no key: it hands over whatever object was stored. + if bag, err := decodeMetadata(step.Metadata); err != nil { + row.MetadataUnreadable = true + } else { + row.Metadata = bag + } attempts = append(attempts, row) } @@ -769,6 +922,120 @@ func artifactIndex(conn *sql.DB, runID int) ([]ArtifactIndexEntry, error) { return out, nil } +// The two seating paths a run's vote seats are minted through (DKT-733). +// These are the values SilentVoteSeat.Path carries, and they are core's own +// closed vocabulary — derived from HOW the proposal joined the run's +// membership, never from anything a caster asserted. +const ( + // SeatPathVoteStep: the proposal is keyed under the run's vote-step + // family (voteIdempotencyPrefix) — an engine-minted `type = "vote"` step + // row, whose panel is seated in-wave. + SeatPathVoteStep = "vote-step" + // SeatPathConversationalGate: everything the run's machinery convened + // OUTSIDE a vote step — a reap-ack ballot keyed under + // ReapAckProposalKey's family, or a proposal whose text names the run (an + // activation panel opened with `vote create`). These panels are seated + // conductor-side. + SeatPathConversationalGate = "conversational-gate" +) + +// SilentVoteSeat is one cast that reported no spend, with the seating path +// that minted its proposal (DKT-733). The proposal id is the argument `vote +// backfill-usage` takes, so each row is an aimable backfill, not just a name. +type SilentVoteSeat struct { + Proposal string `json:"proposal"` + Voter string `json:"voter"` + Role string `json:"role,omitempty"` + Path string `json:"path"` +} + +// silentVoteSeats enumerates the casts the coverage line counts as silent and +// labels each with its seating path. The rows come through the SAME +// membership the coverage count uses; the label is resolved HERE because only +// the engine owns the key-family spellings: a proposal in the run's vote-step +// family is a vote-step seat, and anything else in the membership — reap-ack +// keyed or run-named — is a conversational gate (the extraIDs' two halves, +// per conversationalRunProposalIDs). +func silentVoteSeats(conn *sql.DB, runID int, extraIDs []int) ([]SilentVoteSeat, error) { + rows, err := db.SilentVoteSeatsFor( + conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID), extraIDs...) + if err != nil || len(rows) == 0 { + return nil, err + } + + keyed, err := db.LookupIdempotencyKeys( + conn, db.ScopeVoteCreate, voteIdempotencyPrefix(runID)) + if err != nil { + return nil, err + } + voteStep := make(map[int]bool, len(keyed)) + for _, id := range keyed { + voteStep[id] = true + } + + out := make([]SilentVoteSeat, 0, len(rows)) + for _, r := range rows { + path := SeatPathConversationalGate + if voteStep[r.ProposalID] { + path = SeatPathVoteStep + } + out = append(out, SilentVoteSeat{ + Proposal: model.FormatProposalID(r.ProposalID), + Voter: r.Voter, + Role: r.Role, + Path: path, + }) + } + return out, nil +} + +// conversationalRunProposalIDs resolves the run's CONVERSATIONAL-GATE +// proposals — the ballots the run's machinery convened outside any vote step +// (DKT-584), whose casts otherwise appear in no run section at all: +// +// - reap-ack ballots, keyed under ReapAckProposalKey's family — positively +// attributed through the same idempotency table the vote-step family uses; +// - proposals that NAME the run in their description or rationale (an +// activation panel opened with `vote create` carries no key and no step, +// and its text is its only link to the run it gates). +// +// Vote-step proposals are deliberately NOT re-resolved here: the rollups +// already select their key family by prefix, and the membership test is a set +// test, so an overlap would be harmless but a second spelling of that family +// would not. +func conversationalRunProposalIDs(conn *sql.DB, runID int) ([]int, error) { + keyed, err := db.LookupIdempotencyKeys( + conn, db.ScopeVoteCreate, reapAckRunPrefix(runID)) + if err != nil { + return nil, err + } + seen := make(map[int]bool, len(keyed)) + ids := make([]int, 0, len(keyed)) + for _, id := range keyed { + if !seen[id] { + seen[id] = true + ids = append(ids, id) + } + } + + named, err := db.ProposalIDsNaming(conn, model.FormatRunID(runID)) + if err != nil { + return nil, err + } + for _, id := range named { + if !seen[id] { + seen[id] = true + ids = append(ids, id) + } + } + + // A TOTAL order (R9): the ids feed a parameterized IN whose bound values + // participate in query text equality for no engine, but a deterministic + // argument list keeps two reports byte-identical in any future trace. + sort.Ints(ids) + return ids, nil +} + // configuredBudgetDefault reads `budget.default` for R6's source derivation. func configuredBudgetDefault(conn *sql.DB, projectID int) (float64, error) { entry, err := db.GetConfig(conn, projectID, db.KeyBudgetDefault) diff --git a/internal/engine/saga.go b/internal/engine/saga.go index 7c7acba8..3d2c14d5 100644 --- a/internal/engine/saga.go +++ b/internal/engine/saga.go @@ -125,6 +125,26 @@ type Engine struct { // warning it did not disprove, which is the opposite direction from // IsAncestorFn's own fail-open and deliberately so. TreeMatchFn func(execRoot, sha string) (match, known bool) + // ObjectExistsFn reports whether `sha` resolves as a COMMIT OBJECT from + // execRoot at all — `git cat-file -e ^{commit}` — and whether that + // question could be answered (DKT-742). A field for DiffFn's reason: the + // real one shells out to git. + // + // It exists because IsAncestorFn's `known = false` conflates two states + // that read very differently to a packet consumer: "git could not answer" + // (git absent, not a repository — nothing to warn about) and "the object + // is not in the shared store at all" (a separate-clone worktree whose + // objects never reached the shared store, or a pruned+GC'd one). RUN-52's + // DKT-V253 was the second: all three vote seats ran `git cat-file -t` on + // the packet's target, found no object anywhere, and burned an + // investigation each — while the engine, which had asked git about that + // exact sha at dispatch open, had silently skipped it as unanswerable. + // + // staleTargets consults it only where ancestry was UNANSWERABLE, and only + // a definitive `exists = false, known = true` produces a warning — so the + // "absence of evidence is not staleness" posture holds for every state + // where git genuinely could not answer. + ObjectExistsFn func(execRoot, sha string) (exists, known bool) } // NewEngine builds the S5 engine: the REAL gate runner, the REAL action runner, @@ -149,12 +169,13 @@ type Engine struct { func NewEngine() *Engine { paths := repoPathsFrom(resolvePaths()) return &Engine{ - Gates: NewExecRunner(paths), - Actions: NewActionRunner(paths), - DiffFn: GitDiff, - HeadFn: sharedCheckoutHead, - IsAncestorFn: gitAncestorOfHead, - TreeMatchFn: gitTreeMatchesHead, + Gates: NewExecRunner(paths), + Actions: NewActionRunner(paths), + DiffFn: GitDiff, + HeadFn: sharedCheckoutHead, + IsAncestorFn: gitAncestorOfHead, + TreeMatchFn: gitTreeMatchesHead, + ObjectExistsFn: gitCommitResolvable, } } @@ -738,10 +759,15 @@ func (e *Engine) runGateStage( } // The step's RECORDED worktree rides along (DKT-9), the same resolution // the diff stage applies: a completion gate measures the tree the work - // happened in, not the shared checkout the saga was resumed from. + // happened in, not the shared checkout the saga was resumed from. So does + // the worktree's base commit (DKT-992), resolved per gate rather than once + // per step because the saga advances stage by stage and may resume from a + // different invocation — and the fork point never moves for the worktree's + // lifetime, so every gate of one step resolves the same sha. sc := StepContext{ Instance: step.Instance, RunID: step.RunID, IssueID: step.IssueID, Scope: scope, WorkRoot: step.WorkRoot, + Base: gateBaseSHA(conn, step.RunID, step.WorkRoot), } rows, err := runGate(e.Gates, spec, sc) @@ -910,6 +936,39 @@ func (e *Engine) runRoutingStage( } } + // THE HAND-BACK COMPARISON (DKT-588), for a LOOP BODY at ordinal > 0: the + // commit this completion is about to hand back (the `head` of the round + // record above) against the hand-back the SAME loop body recorded at its + // newest earlier ordinal. Identical shas mean the round moved nothing — + // RUN-34's fix@2 handed back the unchanged HEAD 64d3336b3d71 and a full + // 5-judge + synthesize + verify round (~2.9 budget units) still ran over + // the zero-byte delta, because DKT-340's non-convergence guard fires only + // at the NEXT loop's entry, after the wasted round has already been paid + // for. The comparison is scoped to this step's OWN records rather than + // latestIssueDiffHead's issue-wide newest, because another producer + // (`implement` at ordinal 0, a downstream committer) may have written the + // issue's newest head, and the question here is whether THIS body moved + // the tree since ITS last round. + // + // Read here on the pooled connection, before the transaction opens, for + // loadHoldTally's reason exactly: inside the transaction it would deadlock + // against the one-connection pool rather than fail. + // + // THE MEASUREMENT MUST BE REAL, failing in roundMovedNothing's direction: + // an unresolvable head on either side ("" — HeadFn could not name a + // commit, or the prior round recorded none) is not evidence of an + // unchanged tree, so every degenerate case answers "changed" and the round + // proceeds. Parking a run on a broken head resolution would turn a diff + // setup problem into a stalled run, a worse failure than the wasted round + // this exists to prevent. + unchangedHandBack := "" + if spec.Loop && step.Ordinal > 0 { + if head := handBackHead(diffPayload); head != "" && + head == priorRoundHandBack(conn, step) { + unchangedHandBack = head + } + } + // §5: the order the threshold evaluates under, and the validator an action's // output is checked against, both come from the schema the run PINNED. They // are resolved ONCE here and threaded into both consumers, so a step cannot @@ -987,6 +1046,7 @@ func (e *Engine) runRoutingStage( var ( routing string reason string + cover *batchCover ) switch { case gapOnly: @@ -1027,7 +1087,27 @@ func (e *Engine) runRoutingStage( // A failed gate routes per `on_fail`, not through the threshold: the // threshold asks a question about a RESULT, and a step whose gate // failed has no result to ask about. + // + // UNLESS a run-scoped batch override grant covers EVERY failed gate's + // signature (DKT-546): the operator already ruled this exact failure + // environmental for this run, and re-parking it would re-ask a settled + // question. The pass it records is the same generic RoutingPass the + // operator's own override-pass records, attributed to the grant(s) in + // the routing record and in its own event — never silently. A cover + // blocked by an interposed threshold target parks as usual, with the + // block named as the reason. routing = spec.EffectiveOnFail() + if cover, err = batchOverrideCover(conn, step, spec); err != nil { + return err + } + switch { + case cover == nil: + case cover.blocked != "": + reason = cover.blocked + cover = nil + default: + routing, reason = RoutingPass, cover.reason() + } case action != nil && action.Failed: // B3: a builtin's bad params or an unorderable value, and a trusted // command's non-zero exit or unmatched name, are STEP failures routed @@ -1084,14 +1164,23 @@ func (e *Engine) runRoutingStage( return err } + // The action's artifact and per-attempt records land FIRST, before any + // routing effect in this transaction — H4's "one transaction, or a crash + // leaves a held payload with nobody able to resolve it" for the holding + // path, and, for the routing path, what makes the CURRENT round visible to + // the loop-entry evidence reads (DKT-870): an action step's artifact used + // to land after applyFixLoop, so routingVerdictUnchanged (DKT-589) and the + // flat-volume signal read an action-routed round's evidence one round + // late — an executor's artifact is recorded at `complete`, stages earlier, + // and the two step classes must measure alike. + if action != nil { + if err := recordActionResult(tx, step, action, nowMS); err != nil { + return err + } + } + if holding { if !stale { - // The artifact and the per-attempt records land FIRST, in this same - // transaction, so H4's "one transaction, or a crash leaves a held - // payload with nobody able to resolve it" holds exactly. - if err := recordActionResult(tx, step, action, nowMS); err != nil { - return err - } return enterHeld(tx, step, action.Held, tally, nowMS) } // H20: a STALE-LINEAGE hold is inert exactly as a stale-lineage routing @@ -1106,6 +1195,70 @@ func (e *Engine) runRoutingStage( status := statusForRouting(routing) + // AN UNCHANGED HAND-BACK PARKS THE ROUND AT ITS SOURCE (DKT-588), before + // the review chain downstream spends anything. The park lands on the loop + // body's OWN row — the gap-only and measured-nothing parks above are the + // precedent, and "waiting-human stops the lineage at its source" is their + // exact reasoning — because the downstream chain this ordinal instantiated + // at loop entry already waits on this body: readiness withholds the + // `after_loop` closure from every offer while its same-ordinal loop body + // is non-terminal (blockingLoopBodyAbsent, DKT-48/DKT-61), and the run + // rollup parks the run with the step. Nothing is superseded, deliberately: + // `superseded` is terminal, so sweeping the chain would let the issue + // COMPLETE without its judges ever running the moment an operator resolved + // the park, and would leave `--as override-pass` — the way out the reason + // names — releasing a round with nothing left in it to run. + // + // Only a routing that would have handed the round downstream is overridden + // — `pass`, the status-table row that makes the chain claimable. A failed + // gate's `on_fail` and a threshold's own routing already decided something + // about this completion and are left alone. + if !stale && routing == RoutingPass && unchangedHandBack != "" { + routing = workflow.OnFailWaitingHuman + reason = fmt.Sprintf( + "round %d of %s handed back the same commit %.12s its previous "+ + "round recorded: the loop body changed nothing, so the review "+ + "chain would read the identical tree and reach the identical "+ + "verdict; `docket step resolve --as override-pass` runs the "+ + "chain anyway, `--as retry` redoes the round", + step.Ordinal, model.FormatID(step.IssueID), unchangedHandBack) + status = statusForRouting(routing) + } + + // A PASS THAT WOULD LEAVE DECLARED-FLOOR WORK STANDING PARKS INSTEAD + // (DKT-870). RUN-58's reconcile routed `pass` with all 16 clusters open — + // six at the order's high position, none held, none operator-resolved — + // and the loop exited clean: the threshold had read the field its author + // pointed it at, and nothing read the evidence recorded beside it. When + // the step declares a `pass_floor`, the pass is measured against the + // step's OWN recorded payload under the same pinned order the threshold + // evaluates under, and a pass with standing floor-or-above elements + // becomes `waiting-human` — see passFloorStanding for the exemptions and + // the fail-toward-pass discipline. + // + // The override takes the unchanged-hand-back park's exact shape and + // placement, for its reasons: only a `pass` — the one routing that hands + // the issue onward as settled — is overridden, a failed gate's `on_fail` + // and a threshold's own `fix-loop` already decided something and are left + // alone, and the park lands before applyFixLoop so a parked pass never + // touches the loop counter. `--as override-pass` (below) is the recorded + // operator exit; `--as fix-round` buys a round instead. + if !stale && routing == RoutingPass && spec.PassFloor != nil { + if floor := passFloorStanding(spec.PassFloor, payloads, order); floor.Standing > 0 { + routing = workflow.OnFailWaitingHuman + reason = fmt.Sprintf( + "%d element(s) of %s's recorded payload sit at or above the "+ + "declared pass_floor (%s >= %s), none held and none "+ + "operator-resolved, so a `pass` would exit with that work "+ + "still standing; `docket step resolve --as override-pass` "+ + "exits anyway, `--as fix-round` authorizes another round "+ + "instead", + floor.Standing, step.Instance, spec.PassFloor.Field, + spec.PassFloor.At) + status = statusForRouting(routing) + } + } + // A `fix-loop` routing from a LIVE lineage is the loop entry (§11.3). It // runs before the step's own status is written so its outcome — entered, or // bounded to `waiting-human` — decides that status. applyFixLoop re-reads @@ -1123,12 +1276,6 @@ func (e *Engine) runRoutingStage( } } - if action != nil { - if err := recordActionResult(tx, step, action, nowMS); err != nil { - return err - } - } - if wantsDiff { if _, err := db.InsertArtifactTx(tx, db.Artifact{ RunID: step.RunID, StepID: step.ID, Kind: ArtifactKindIssueDiff, @@ -1152,6 +1299,24 @@ func (e *Engine) runRoutingStage( }); err != nil { return err } + // A batch-covered pass spends operator authority, so it is attributed like + // the resolution it stands in for (DKT-546): the covering grants' counters + // move and the feed records which grants decided this step — in this same + // transaction, and BEFORE the stale guard, because §7.3 (3) records a stale + // lineage's routing for the ledger too, and an override the ledger carries + // must name its authority either way. + if cover != nil { + if err := db.CoverGateOverrideGrantsTx(tx, cover.grantIDs); err != nil { + return err + } + if err := recordEvent(tx, eventRecord{ + Kind: EventStepBatchOverridden, RunID: step.RunID, + Instance: step.Instance, IssueID: step.IssueID, + Data: cover.eventData(), AtMS: nowMS, + }); err != nil { + return err + } + } // ---- The inert guard (§7.3 (3), §11.3 (2)). --------------------------- // @@ -1263,21 +1428,7 @@ func gateVerdict(conn *sql.DB, stepID int) (string, []string, error) { // verdictOverRows reduces a step's recorded rows to a routing verdict, and // names the gates that measured nothing. func verdictOverRows(rows []db.GateResultRow) (string, []string) { - // Ordinals carry flaky re-runs (§5.6 F3): every attempt is its own row, and - // F4 makes the LAST attempt's verdict the one that routes. So the decision - // is made per (gate, ordinal-max), never over every row — otherwise a gate - // that failed twice and passed on the third try would route as a failure, - // which is exactly what declaring it flaky was meant to prevent. - last := make(map[string]db.GateResultRow) - for _, r := range rows { - if r.Pre { - continue // PG4 - } - prev, seen := last[r.Gate] - if !seen || r.Ordinal >= prev.Ordinal { - last[r.Gate] = r - } - } + last := lastGateAttempts(rows) verdict := VerdictPass var unmeasured []string @@ -1295,6 +1446,90 @@ func verdictOverRows(rows []db.GateResultRow) (string, []string) { return verdict, unmeasured } +// lastGateAttempts reduces a step's recorded rows to the one attempt per gate +// that ROUTES. +// +// Ordinals carry flaky re-runs (§5.6 F3): every attempt is its own row, and F4 +// makes the LAST attempt's verdict the one that routes. So the decision is made +// per (gate, ordinal-max), never over every row — otherwise a gate that failed +// twice and passed on the third try would route as a failure, which is exactly +// what declaring it flaky was meant to prevent. Pre-gate rows are excluded per +// PG4: they are inputs to the step, not judgments of it. +// +// It is shared by the routing verdict and by FailedGates so the gates a caller +// REPORTS are, by construction, the rows the saga ROUTED on. Two reductions +// over the same table is how a report comes to contradict the routing beside +// it — which is the class of defect DKT-982 is. +func lastGateAttempts(rows []db.GateResultRow) map[string]db.GateResultRow { + last := make(map[string]db.GateResultRow) + for _, r := range rows { + if r.Pre { + continue // PG4 + } + prev, seen := last[r.Gate] + if !seen || r.Ordinal >= prev.Ordinal { + last[r.Gate] = r + } + } + return last +} + +// FailedGate is one completion gate whose routing attempt did not pass, in the +// shape a caller needs to SAY SO: the name, the verdict word, and the exit code +// the process left behind. +// +// Exit is a pointer for db.GateResultRow's reason exactly — an `unmatched` gate +// never ran, and rendering `exit 0` for a process that does not exist reads as +// a pass (T11). +type FailedGate struct { + Gate string `json:"gate"` + Verdict string `json:"verdict"` + Exit *int `json:"exit"` + // Reason explains an `unmatched` verdict or a timeout, verbatim from the + // recorded row. + Reason string `json:"reason,omitempty"` +} + +// FailedGates returns the completion gates that did not pass, sorted by name. +// +// It is the READ SIDE of the routing decision (DKT-982): a step whose gate +// failed parks, and until this existed the verb that parked it could name +// nothing — `step record` printed a success-shaped line with no gate on it, and +// the executor reading that line reported the run green. The engine knew at +// print time; it had no accessor to say it with. +// +// Empty means every completion gate passed, or none ran. +func FailedGates(conn *sql.DB, stepID int) ([]FailedGate, error) { + rows, err := db.GateResultsForStep(conn, stepID) + if err != nil { + return nil, err + } + return failedGatesOverRows(rows), nil +} + +// failedGatesOverRows is FailedGates over rows already read — pure, so a test +// can weigh the reduction without a store. +func failedGatesOverRows(rows []db.GateResultRow) []FailedGate { + last := lastGateAttempts(rows) + + failed := make([]FailedGate, 0, len(last)) + for _, r := range last { + // Not-pass, in the same sense verdictOverRows uses: `fail`, `unmatched`, + // and `skipped` all route as a failure, and a report that named only the + // first would be silent about the two the routing acted on. + if r.Verdict == db.GateVerdictPass { + continue + } + failed = append(failed, FailedGate{ + Gate: r.Gate, Verdict: r.Verdict, Exit: r.Exit, Reason: r.Reason, + }) + } + // Sorted for verdictOverRows' reason: the sentence a step parks with must be + // the same on every run over the same rows, and map range order is not. + sort.Slice(failed, func(i, j int) bool { return failed[i].Gate < failed[j].Gate }) + return failed +} + // runGate invokes the runner and normalizes its output to rows. // // A runner that implements the richer GateExecution shape (the real one) @@ -1408,7 +1643,7 @@ func recordGateEvents( // announced a minute later landed in the feed stamped a minute before // the event preceding it, and the two events closing one gate disagreed // about when the gate happened. - data, err := gateEventData(gate, r.Verdict, r.Exit) + data, err := gateEventData(gate, r.Verdict, r.Exit, r.Pre, r.StubEntry) if err != nil { return err } @@ -1431,8 +1666,8 @@ func recordGateEvents( // The verdict is the LAST row's: attempts are recorded in order and a flaky // re-run supersedes the attempt before it, so the final row is the outcome // the saga itself routes on. - verdict, exit := gateOutcome(rows) - data, err := gateEventData(gate, verdict, exit) + verdict, exit, pre, stub := gateOutcome(rows) + data, err := gateEventData(gate, verdict, exit, pre, stub) if err != nil { return err } @@ -1443,13 +1678,20 @@ func recordGateEvents( } // gateOutcome reduces a gate's attempts to the one that counts: the LAST, -// because a re-run supersedes the attempt before it. -func gateOutcome(rows []GateResultRow) (verdict string, exit *int) { +// because a re-run supersedes the attempt before it. `pre` and `stub` ride +// along from the same row: every attempt of one gate is produced by one +// phase, so the last row's markers are the gate's. +// +// `stub` reads StubEntry — the matched trust entry's own `stub` declaration +// (DKT-265) — not the legacy S3-migration `Stub` field. StubEntry is what +// `gate_results` (`step gates --json`) and `run report` already render as +// `stub`, and DKT-983 asks this event to say what those surfaces already say. +func gateOutcome(rows []GateResultRow) (verdict string, exit *int, pre, stub bool) { if len(rows) == 0 { - return "", nil + return "", nil, false, false } last := rows[len(rows)-1] - return last.Verdict, last.Exit + return last.Verdict, last.Exit, last.Pre, last.StubEntry } // gateEventData renders a gate event's payload: the gate name in `detail`, @@ -1458,7 +1700,35 @@ func gateOutcome(rows []GateResultRow) (verdict string, exit *int) { // A missing exit stays ABSENT rather than rendering as 0 — an unmatched gate // never ran, and `exit=0` on a gate that was refused execution would read as a // pass. -func gateEventData(gate, verdict string, exit *int) (string, error) { +// +// `pre` MARKS THE VERDICTS THAT ROUTED NOTHING (DKT-862). A §11.1 pre-gate runs +// at claim as an input to the step, and PG4 keeps its result out of the saga's +// verdict — so `gate-recorded ... verdict=fail` on a pre-gate was reporting a +// failure that never blocked anything, in bytes identical to one that did. On +// RUN-61 three such rows appeared in the feed beside the `step-routed` that +// contradicted them, and nothing on the line said which was which. +// +// It rides as its OWN KEY rather than as an adjective on the verdict, for two +// reasons. `verdict` is a closed vocabulary a program reads, and "fail +// (advisory)" is not in it. And the human line comes from eventDetail, which +// renders `data` as sorted `key=value` pairs and INTERPRETS NOTHING — a +// renderer that special-cased this pair would be the first key core's event +// feed had an opinion about. +// +// It is ABSENT on a blocking gate rather than `pre=false`, so the marker's +// presence is the whole signal and the overwhelmingly common line does not +// grow a column that always says the same thing. +// +// `stub` MARKS A GATE WHOSE PASS WAS NEVER MEASURED (DKT-983). A stub-trusted +// command's pass is already marked stub:true in the trust store, in +// `gate_results` rows (`step gates --json`), and in `run report` — but until +// this, the event stream carried none of it, so `gate-recorded ... verdict=pass` +// for a stub was byte-identical to a real measurement. It is ABSENT rather +// than `stub=false` for the same reason `pre` is: the marker's presence is the +// whole signal. The caller passes StubEntry (the trust entry's own `stub` +// declaration), never the legacy S3-migration `Stub` field — StubEntry is +// what every other stub-aware surface already reads. +func gateEventData(gate, verdict string, exit *int, pre, stub bool) (string, error) { fields := map[string]any{"detail": gate} if verdict != "" { fields["verdict"] = verdict @@ -1466,6 +1736,12 @@ func gateEventData(gate, verdict string, exit *int) (string, error) { if exit != nil { fields["exit"] = *exit } + if pre { + fields["pre"] = true + } + if stub { + fields["stub"] = true + } out, err := json.Marshal(fields) if err != nil { return "", fmt.Errorf("encoding the gate event for %s: %w", gate, err) @@ -2042,6 +2318,39 @@ func runDiffBase(conn *sql.DB, runID int, dir, execRoot string) string { return sharedCheckoutHead(execRoot) } +// gateBaseSHA resolves the base commit a completion gate's child is told +// about via DOCKET_GATE_BASE (DKT-992): for a worktree-recorded step, the +// worktree's fork point — the commit the worktree was created from, the SAME +// resolution the diff stage's runDiffBase applies to `dir` — so a gate can +// scan exactly the committed range the recorded issue.diff describes, instead +// of guessing (`git diff HEAD~1`, wrong for multi-commit steps) or scanning +// the working tree an executor already committed to (always clean, so +// RUN-66's secret-scan passed 8/8 write steps having scanned zero lines). +// +// "" — the var UNSET — everywhere a step's committed range is not knowable, +// and deliberately NOT runDiffBase's pinned-commit fallback: +// +// - a shared-checkout step ("" or workRoot == the run's exec root) has no +// fork point, and the pinned run commit is not this STEP's base — sibling +// issues' work lands between it and the step's own commits, so exporting +// it would attribute their range to this step. The acceptance choice here +// is UNSET for non-worktree steps, not "equal to HEAD": a live HEAD read +// is a value docket cannot vouch for as a range endpoint, and absence is +// the honest encoding (the same convention as DOCKET_SCOPE). +// - a worktree whose fork point cannot be resolved exports nothing rather +// than a guess; a range-shaped gate finding the var absent over a clean +// tree fails closed, which is the correct direction for a control. +func gateBaseSHA(conn *sql.DB, runID int, workRoot string) string { + if workRoot == "" { + return "" + } + execRoot := runExecRoot(conn, runID) + if workRoot == execRoot { + return "" + } + return worktreeForkPoint(workRoot, execRoot) +} + // worktreeForkPoint resolves the merge-base of a worktree's HEAD and the // shared checkout's — DKT-42's base for worktree-recorded steps. "" when // either side cannot be resolved, which sends runDiffBase to its pinned @@ -2143,6 +2452,47 @@ func gitAncestorOfHead(execRoot, sha string) (ancestor, known bool) { return false, false } +// gitCommitResolvable is ObjectExistsFn's real implementation (DKT-742): does +// this sha resolve as a commit object from execRoot at all? It is exactly the +// probe a packet consumer runs by hand — `git cat-file -e ^{commit}` — +// asked once at dispatch time instead of once per seat mid-wave. +// +// The three-valued mapping needs TWO probes, because cat-file's peel form +// exits 128 for both "no such object" and "not a repository" — one a +// definitive absence worth warning on, the other an unanswerable question +// that must stay silent: +// +// - `cat-file -e ^{commit}` exit 0: the object exists and peels to a +// commit — (true, true). +// - otherwise `cat-file -e ` (no peel) exit 1: git ran, looked, and +// found NO OBJECT — a definitive (false, true). Exit 0 here means the +// object exists but is not a commit, which is equally definitive: a +// recorded target sha naming a blob or tree does not resolve for any +// consumer either. +// - anything else — git absent, not a repository — (false, false), which no +// caller may treat as absence. +func gitCommitResolvable(execRoot, sha string) (exists, known bool) { + if execRoot == "" || sha == "" { + return false, false + } + if exec.Command("git", + gitDirArgs(execRoot, "cat-file", "-e", sha+"^{commit}")...).Run() == nil { + return true, true + } + err := exec.Command("git", + gitDirArgs(execRoot, "cat-file", "-e", sha)...).Run() + if err == nil { + // Present but not peelable to a commit: definitively unresolvable as + // the commit the packet records. + return false, true + } + var exitErr *exec.ExitError + if errors.As(err, &exitErr) && exitErr.ExitCode() == 1 { + return false, true + } + return false, false +} + // treeMatchBatch caps how many pathspecs ride on one `git diff` argv, so a // commit touching thousands of files cannot overflow the platform's argument // limit. The comparison is split across batches and every batch must come back @@ -2402,13 +2752,53 @@ func latestIssueDiffHead(conn *sql.DB, runID, issueID int) string { WHERE a.run_id = ? AND s.issue_id = ? AND a.kind = ? ORDER BY a.id DESC LIMIT 1`, runID, issueID, ArtifactKindIssueDiff).Scan(&payload) - if err != nil || payload.String == "" { + if err != nil { + return "" + } + return handBackHead(payload.String) +} + +// priorRoundHandBack is the hand-back head THIS step's own name recorded at +// its newest earlier ordinal — the sha the same loop body handed back last +// round — or "" when it never recorded one (DKT-588). +// +// BELOW, not AT, the step's ordinal, for newestIssueDiffUpTo's exact reason: +// the engine suppresses a byte-identical or empty re-record (DKT-258/DKT-259), +// so a round may leave no artifact of its own, and the newest earlier record +// is still the last commit this body actually handed back. Excluding the +// step's OWN ordinal is what keeps a re-completion at the same ordinal — an +// operator `--as retry` — from comparing against its own first record. +// +// Filtered to the step's NAME, unlike latestIssueDiffHead's issue-wide read: +// the issue's newest head may belong to a different producer entirely, and +// this comparison is only meaningful between two hand-backs of the same body. +func priorRoundHandBack(conn *sql.DB, step *db.Step) string { + var payload sql.NullString + err := conn.QueryRow( + `SELECT a.payload FROM artifacts a JOIN steps s ON s.id = a.step_id + WHERE a.run_id = ? AND s.issue_id = ? AND s.step_name = ? + AND s.ordinal < ? AND a.kind = ? + ORDER BY a.id DESC LIMIT 1`, + step.RunID, step.IssueID, step.StepName, step.Ordinal, + ArtifactKindIssueDiff).Scan(&payload) + if err != nil { + return "" + } + return handBackHead(payload.String) +} + +// handBackHead decodes the `head` out of a round record payload — +// appendRoundDelta's `{"head": "..."}` — "" when the payload carries none or +// cannot be read. "" is the degenerate answer every consumer must treat as +// "no measurement", never as a comparable value. +func handBackHead(payload string) string { + if payload == "" { return "" } var record struct { Head string `json:"head"` } - if json.Unmarshal([]byte(payload.String), &record) != nil { + if json.Unmarshal([]byte(payload), &record) != nil { return "" } return record.Head diff --git a/internal/engine/saga_resume.go b/internal/engine/saga_resume.go index 65c2da79..4495b1a8 100644 --- a/internal/engine/saga_resume.go +++ b/internal/engine/saga_resume.go @@ -211,7 +211,13 @@ func (e *Engine) parkInterruptedGate( // The VERDICT rides along (DKT-63) so this reads as a refusal in the feed // rather than as a gate that merely happened. There is no exit code: an // unmatched gate never ran, and `exit=0` here would read as a pass. - unmatched, err := gateEventData(gate.Name, VerdictUnmatched, nil) + // + // `gate.Pre` is read rather than hardcoded false (DKT-862). It IS false on + // every reachable call today — this path resolves a completion gate, and + // completionGates drops the `pre` ones — but the marker has exactly one + // definition, the declaration itself, and a second spelling here is how the + // two surfaces would start to disagree again. + unmatched, err := gateEventData(gate.Name, VerdictUnmatched, nil, gate.Pre, false) if err != nil { return err } diff --git a/internal/engine/scope_refresh.go b/internal/engine/scope_refresh.go new file mode 100644 index 00000000..b54b160b --- /dev/null +++ b/internal/engine/scope_refresh.go @@ -0,0 +1,375 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "errors" + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// The mid-run scope refresh (DKT-869), and why it exists after DKT-741 said it +// would not. +// +// DKT-741 established the freeze and it STANDS: `run_issues.issue_snapshot` is +// written once at activation stage 4, the packet's `context.issue.scope` and +// the recorded `issue.diff` scope both read it (§5.1.1, §6.6, §6.7.1 D1), and +// nothing in the engine rewrites it. What DKT-741 also decided — that no verb +// would ever refresh it, and that an authorized widen must be spent by +// abandoning the issue and re-planning it — is what RUN-52 (VPL-434) then +// charged for, twice. +// +// The RUN-52 shape, from the run's own terminal routing note: the panel +// rejected the work 3/3 on an out-of-scope migration blocker, the operator +// AGREED and widened the issue's scope, the conductor ran `issue edit --scope` +// — and the already-minted `fix@2` step still rendered the old two-path scope. +// "Scope snapshots per-step at creation and does not reach the already-created +// fix@2 step — no engine verb refreshes an existing step's scope." The +// sanctioned remedy (`run abandon --issue` plus a full re-plan) is priced for a +// run whose PREMISE changed; here the premise was intact and one declaration +// had been corrected by the operator, mid-loop, on purpose. The issue was +// abandoned instead. +// +// So the freeze keeps its default and gains an explicit, refusable, recorded +// exception. Four properties are what make it not a hole in §9 item 5: +// +// 1. IT CARRIES NO SCOPE OF ITS OWN. The refresh reads `issues.scope_globs` +// and copies it verbatim; there is no `--scope` here and there must never +// be one. `issue create|edit --scope` stays the SOLE writer of that column +// (issue_scope.go), so this verb cannot make real any scope that was not +// already declared through the one gate scope widening has always had. The +// authorization for what lands is, exactly, the authorization for the +// widen — and a refresh with no widen behind it has nothing to copy and is +// refused (D4 below). +// 2. NO STEP STRADDLES IT. It refuses while any of the issue's steps is +// `claimed`, `running`, or `gated`, and while a dispatch is open — the +// repin quiescence rule (repin.go), applied to the other frozen premise. A +// step that already holds a packet rendered under the old scope must not +// record its diff under the new one. +// 3. IT REWRITES NO HISTORY. Terminal steps keep their artifacts, their +// recorded diffs, and the scope those were computed over. What changes is +// what the REMAINING steps will render — the same division of labour +// `run repin` draws between an agreement and the history under it. +// 4. THE DISCONTINUITY IS IN THE LEDGER. One `issue-scope-refreshed` event +// carries the old scope, the new scope, the steps it reaches and the +// operator's reason, so two steps of one run rendering two different +// scopes is a fact a reader can date and attribute rather than a drift +// they must infer. This is what keeps "a packet is reproducible from the +// ledger" true: the ledger now says when the input changed. +// +// Everything the freeze protects that is NOT scope is untouched — title, kind, +// labels, `linked`, and the description snapshot are re-encoded byte-for-byte +// from what activation wrote. A mid-run relabel still cannot change how a step +// routes, and a mid-run description edit still cannot reach a packet. + +// RefreshedScope reports what a refresh did, as the verb answers with it. +type RefreshedScope struct { + Run string `json:"run"` + Issue string `json:"issue"` + // From and To are the frozen scope and the live one it was replaced with, + // in DECLARED ORDER — the author's order, which the snapshot echoes back + // verbatim (issue_scope.go) and which this must not sort. + From []string `json:"from"` + To []string `json:"to"` + // Steps names the non-terminal instances the refresh reaches, in id order: + // exactly the steps that will render the new scope and record their diffs + // over it. + Steps []string `json:"steps"` +} + +// refreshBlockingStepStatuses are the statuses that make a refresh a straddle. +// +// `claimed` and `running` are an executor mid-flight: its packet was rendered +// under the frozen scope and its diff will be recorded after the change, which +// is the one combination that would falsify a step's own provenance. +// +// `gated` is here for a reason the scheduler's own scope predicate does NOT +// share (stepExcludesScope deliberately omits it): a gated step's artifact has +// recorded and its token has retired, but its saga is still running, and a +// diff-shaped gate reads snapshotScope to decide what it measures (saga.go, +// DKT-63). Refreshing under one would run a gate over paths the artifact it is +// gating never covered. +// +// `pending` and `waiting-human` are the refreshable states, and both on +// purpose. `pending` is the RUN-52 case exactly. `waiting-human` is the case +// DKT-741's own advisory already treats as live (TestScopeEditWarningFiresFor +// AParkedStep): a parked step will be resolved and will render again, and a +// `retry` re-executes it and records a fresh diff — over the refreshed scope, +// which is the entire point of authorizing the widen. +var refreshBlockingStepStatuses = []string{ + db.StepClaimed, db.StepRunning, db.StepGated, +} + +// RefreshIssueScopeInRun re-reads `issues.scope_globs` and overwrites the +// `scope` key of ONE run-issue's activation snapshot (DKT-869). +// +// The unit is (run, issue) rather than a single step because that is where the +// snapshot LIVES: one blob per bound issue per run, read by every step of that +// issue. A `step refresh-scope STEP-N` would name one step and silently move +// its siblings' scope too, which is a blast radius the argument no longer +// matches — and it would let two steps of one issue disagree about scope +// inside one dispatch, which is the drift D2 exists to prevent. +func RefreshIssueScopeInRun( + conn *sql.DB, runID, issueID int, reason string, nowMS int64, +) (*RefreshedScope, error) { + if strings.TrimSpace(reason) == "" { + return nil, validationErr( + "a reason is required to refresh a snapshotted scope; the event " + + "trail must say why a live run's packets changed what they declare") + } + + tx, err := conn.Begin() + if err != nil { + return nil, fmt.Errorf("refreshing the snapshotted scope: %w", err) + } + defer tx.Rollback() + + run, err := db.GetRunTx(tx, runID) + if errors.Is(err, db.ErrRunNotFound) { + return nil, notFoundErr(err, "run %s not found", model.FormatRunID(runID)) + } + if err != nil { + return nil, err + } + // The same two statuses repin accepts, for the same reason: a `planning` + // run has frozen nothing yet (its next activation snapshots the widened + // scope by itself, which is why the DKT-741 advisory stays silent there), + // and a terminal run's snapshot is referenced only by completed steps' + // history. + if run.Status != model.RunActive && run.Status != model.RunWaitingHuman { + return nil, conflictErr( + "run %s is %s; a scope refresh applies to a run that is %s — a "+ + "planning run snapshots the current scope at its next "+ + "activation, and a terminal run's snapshot is history, not a "+ + "premise to move", + run.Ref(), run.Status, + orStatusList([]model.RunStatus{model.RunActive, model.RunWaitingHuman})) + } + + ri, err := runIssueTx(tx, runID, issueID) + if err != nil { + return nil, err + } + if ri.IssueSnapshot == "" { + return nil, conflictErr( + "issue %s is attached to %s but not yet bound: activation has "+ + "frozen no snapshot for it, so the next `docket run activate` "+ + "will snapshot the scope declared then. There is nothing to "+ + "refresh", + model.FormatID(issueID), run.Ref()) + } + + frozen, err := decodeSnapshotScope(ri.IssueSnapshot) + if err != nil { + return nil, err + } + + // D1: the value comes from the column `--scope` writes, and from nowhere + // else. This verb takes no globs. + live, err := liveIssueScopeTx(tx, issueID) + if err != nil { + return nil, err + } + + // D4, the gate: without a widen recorded through `issue edit --scope` + // there is nothing this verb is authorized to make real. Refusing rather + // than no-opping keeps the ledger honest — an `issue-scope-refreshed` + // event that refreshed nothing is a ruling that ruled nothing, and a + // reader auditing the discontinuities would have to open each one to find + // out which were real. + if sameScope(frozen, live) { + return nil, conflictErr( + "%s already renders %s in %s; a refresh copies the scope declared "+ + "on the issue into the run's snapshot, and nothing has been "+ + "declared since it was frozen. Widen it first — `docket issue "+ + "edit %s --scope GLOB --scope GLOB` — then refresh", + model.FormatID(issueID), renderScope(frozen), run.Ref(), + model.FormatID(issueID)) + } + + // D2: quiescence, per issue for the steps and per run for the dispatch. + steps, err := refreshableSteps(tx, runID, issueID, run.Ref()) + if err != nil { + return nil, err + } + if open, err := db.OpenDispatchTx(tx, runID); err == nil { + return nil, conflictErr( + "a dispatch is open for %s (%s, expiring at %d); close or abandon "+ + "it before refreshing — its manifest was offered under the "+ + "frozen scope, and a relay spawning from it would claim rows "+ + "whose packets no longer say what the manifest's reader saw", + run.Ref(), FormatDispatchID(open.ID), open.ExpiresMS) + } else if !errors.Is(err, db.ErrNoOpenDispatch) { + return nil, err + } + + refreshed, err := reScopedSnapshot(ri.IssueSnapshot, live) + if err != nil { + return nil, err + } + if err := db.SetRunIssueSnapshotTx(tx, runID, issueID, refreshed); err != nil { + return nil, err + } + + // D4: one event, carrying both scopes, the steps it reaches, and why. + data, err := json.Marshal(map[string]any{ + "issue": model.FormatID(issueID), "reason": reason, + "from": frozen, "to": live, "steps": steps, + }) + if err != nil { + return nil, fmt.Errorf("recording the scope refresh: %w", err) + } + if err := recordEvent(tx, eventRecord{ + Kind: EventIssueScopeRefreshed, RunID: runID, IssueID: issueID, + Data: string(data), AtMS: nowMS, + }); err != nil { + return nil, err + } + + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("refreshing the snapshotted scope: %w", err) + } + return &RefreshedScope{ + Run: run.Ref(), Issue: model.FormatID(issueID), + From: frozen, To: live, Steps: steps, + }, nil +} + +// refreshableSteps returns the non-terminal instances of one issue in one run, +// in id order, refusing when any of them would straddle the change or when +// there are none left to reach. +func refreshableSteps(tx *sql.Tx, runID, issueID int, runRef string) ([]string, error) { + rows, err := tx.Query( + `SELECT instance, status FROM steps + WHERE run_id = ? AND issue_id = ? AND status NOT IN (?, ?, ?, ?) + ORDER BY id`, + runID, issueID, + db.StepDone, db.StepSkipped, db.StepSuperseded, db.StepFailedRouted) + if err != nil { + return nil, fmt.Errorf("collecting %s's remaining steps: %w", runRef, err) + } + type liveStep struct{ instance, status string } + remaining, err := scanTxRows(rows, func(r *sql.Rows) (liveStep, error) { + var s liveStep + return s, r.Scan(&s.instance, &s.status) + }) + if err != nil { + return nil, err + } + + blocking := make(map[string]bool, len(refreshBlockingStepStatuses)) + for _, s := range refreshBlockingStepStatuses { + blocking[s] = true + } + var ( + instances []string + straddled []string + ) + for _, s := range remaining { + if blocking[s.status] { + straddled = append(straddled, fmt.Sprintf("%s (%s)", s.instance, s.status)) + continue + } + instances = append(instances, s.instance) + } + if len(straddled) > 0 { + return nil, conflictErr( + "%d step(s) of %s in %s are mid-flight (%s); a refresh under one "+ + "would change what an executor's packet means mid-execution, "+ + "or run a gate over paths the artifact it gates never covered "+ + "— wait for them to record (or for their leases to be reaped), "+ + "then retry", + len(straddled), model.FormatID(issueID), runRef, + strings.Join(straddled, ", ")) + } + if len(instances) == 0 { + return nil, conflictErr( + "every step of %s in %s is terminal; their diffs are already "+ + "recorded over the scope they ran under and no packet will "+ + "render again, so a refresh could only rewrite that record. "+ + "The widened scope reaches the issue's NEXT run", + model.FormatID(issueID), runRef) + } + return instances, nil +} + +// runIssueTx loads one `run_issues` row, distinguishing "not in this run" from +// a read failure — the membership check `run abandon --issue` makes, returning +// the row because the caller needs its snapshot. +func runIssueTx(tx *sql.Tx, runID, issueID int) (*db.RunIssue, error) { + all, err := db.ListRunIssuesTx(tx, runID) + if err != nil { + return nil, err + } + for _, ri := range all { + if ri.IssueID == issueID { + return ri, nil + } + } + return nil, notFoundErr(db.ErrNotFound, "issue %s is not part of run %s", + model.FormatID(issueID), model.FormatRunID(runID)) +} + +// liveIssueScopeTx is liveIssueScope inside the refresh's transaction, so the +// value written is the value that was read — a widen landing between a +// standalone read and the UPDATE would otherwise be reported as the `to` of an +// event that wrote something else. +func liveIssueScopeTx(tx *sql.Tx, issueID int) ([]string, error) { + stored, err := db.IssueScopeGlobsTx(tx, issueID) + if err != nil { + return nil, fmt.Errorf("reading the live scope for %s: %w", + model.FormatID(issueID), err) + } + return decodeScope(stored) +} + +// decodeSnapshotScope pulls the `scope` key out of a snapshot blob. It is +// snapshotScope's transaction-free half, spelled here because the refresh reads +// the blob it is about to rewrite rather than the column. +func decodeSnapshotScope(snapshot string) ([]string, error) { + var frozen struct { + Scope []string `json:"scope"` + } + if err := json.Unmarshal([]byte(snapshot), &frozen); err != nil { + return nil, fmt.Errorf("reading the snapshotted scope: %w", err) + } + return frozen.Scope, nil +} + +// reScopedSnapshot re-encodes a snapshot with a new `scope` and EVERY OTHER +// FIELD byte-identical. +// +// It decodes into `issueSnapshotFields` — the same type activation encodes +// with — so the canonical key order is the one type's declaration order in both +// directions, and a field added to §11.4's issue shape lands in both paths at +// once. TestRefreshedSnapshotIsByteIdenticalApartFromScope pins the round trip. +// +// An unknown key in the stored blob would be DROPPED by this round trip, which +// is why the pinning test exists rather than a tolerant merge: a snapshot +// carrying a key no Go field names means the two writers have already diverged, +// and the right time to find that out is in the test suite. +func reScopedSnapshot(snapshot string, scope []string) (string, error) { + var fields issueSnapshotFields + if err := json.Unmarshal([]byte(snapshot), &fields); err != nil { + return "", fmt.Errorf("reading the issue snapshot: %w", err) + } + if fields.Labels == nil { + fields.Labels = []string{} + } + // The undeclared case survives as `[]`, exactly as activation writes it for + // an issue with no `--scope`: the blob's own vocabulary has no NULL, and + // snapshotScope reads a missing or empty list as "no declared scope". + if scope == nil { + scope = []string{} + } + fields.Scope = scope + + out, err := json.Marshal(fields) + if err != nil { + return "", fmt.Errorf("serializing the refreshed snapshot: %w", err) + } + return string(out), nil +} diff --git a/internal/engine/scope_snapshot.go b/internal/engine/scope_snapshot.go new file mode 100644 index 00000000..716bedeb --- /dev/null +++ b/internal/engine/scope_snapshot.go @@ -0,0 +1,214 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// The mid-run scope widen, and why there is no verb that performs it (DKT-741). +// +// `docket issue edit --scope` writes `issues.scope_globs` and nothing else. +// That column is the LIVE scope, and exactly one consumer reads it live: the +// scheduler's mutual-exclusion check (§6.3 R4, loadIssueFacts), which asks +// "what does this issue touch NOW" so an operator's correction takes effect +// against collisions immediately. +// +// EVERY OTHER CONSUMER READS THE ACTIVATION SNAPSHOT, and does so by design: +// +// - the context bundle's `issue.scope` — the scope a rendered packet's brief +// carries — comes from `run_issues.issue_snapshot` (§5.1.1, §6.6's +// five-source rule). Assembly reads NO live issue field at all, and +// TestContextAssemblyReadsNoLiveState enforces that at the code level. +// - the recorded `issue.diff` is computed over snapshotScope (§6.7.1 D1), +// the same frozen blob, so a step's recorded diff cannot depend on an edit +// made after the run was activated. +// +// This is §9 item 5's mid-run edit immunity — the same invariant that keeps a +// `description` edit out of an already-activated packet (DKT-725) — and it is +// what makes a packet reproducible from the ledger. So `issue edit --scope` +// still re-snapshots NOTHING: an edit that reached a live step by itself would +// let two steps of one run render two different issue scopes and record two +// diffs over two different path sets, which is precisely the drift the +// snapshot exists to prevent. +// +// The consequence an operator meets is that an AUTHORIZED widen — the panel +// rejected the work as out of scope and the operator agreed to widen it — is +// not executable against the running run BY THE EDIT ALONE. DKT-741 left the +// sanctioned path at abandon + re-plan; RUN-52 (VPL-434) then paid for that +// twice on an intact premise, and DKT-869 added the narrower disposition: +// +// docket run refresh-scope RUN-N --issue DKT-M --reason "scope widened" +// +// which copies the declared column into ONE run-issue's snapshot as a second, +// explicit, refusable, event-logged act (scope_refresh.go states the four +// properties that keep it from being a hole in the freeze). Abandon + re-plan +// remains right where the run's PREMISE changed rather than one declaration: +// +// docket run abandon RUN-N --issue DKT-M --reason "scope widened" +// # then re-plan the issue into a new run +// +// ScopeEditFrozenForActiveRuns is what names both at the moment it matters. + +// ScopeEditFrozenForActiveRuns names the live runs an `issue edit --scope` on +// issueID did NOT reach, and the abandon + re-plan path that would (DKT-741). +// +// It fires only when the widen is actually invisible somewhere: the issue is +// bound into a NON-TERMINAL run that has already been activated (so a snapshot +// exists), that run still holds at least one NON-TERMINAL step for the issue +// (so a packet will still be rendered from the stale snapshot), and the live +// scope genuinely DIFFERS from what that run froze. A re-declaration of the +// same globs, an edit to an unactivated `planning` run, and an issue whose +// steps have all recorded are all silent — none of them has anything to +// discover. +// +// It reports rather than refuses. The write to `issues.scope_globs` is real +// and does take effect for scheduling, so the operator is told what landed and +// what did not, not stopped. Advisory-only also means it is safe to call after +// the edit commits, which is where `issue edit` calls it: the answer is a +// property of the run's frozen snapshot, which the edit did not touch. +// +// Every error is swallowed to nil, matching OverridePassSkipsInterposedTargets: +// an advisory that cannot be computed must never fail the verb that already +// succeeded. +func ScopeEditFrozenForActiveRuns(conn *sql.DB, issueID int) []string { + // The live column FIRST, before the cursor below is open. The engine's + // pool is single-connection (§4.8's SQLite writer discipline), so a second + // query issued while rows are still being walked waits forever for a + // connection that the walk itself holds. + live, err := liveIssueScope(conn, issueID) + if err != nil { + return nil + } + + rows, err := conn.Query( + `SELECT ri.run_id, ri.issue_snapshot, COUNT(s.id) + FROM run_issues ri + JOIN runs r ON r.id = ri.run_id + LEFT JOIN steps s + ON s.run_id = ri.run_id AND s.issue_id = ri.issue_id + AND s.status NOT IN (`+placeholders(len(terminalStepStatuses))+`) + WHERE ri.issue_id = ? + AND ri.issue_snapshot IS NOT NULL AND ri.issue_snapshot != '' + AND r.status NOT IN (?, ?) + GROUP BY ri.run_id + HAVING COUNT(s.id) > 0 + ORDER BY ri.run_id`, + append( + terminalStepStatusArgs(), + issueID, string(model.RunDone), string(model.RunAbandoned), + )..., + ) + if err != nil { + return nil + } + defer func() { _ = rows.Close() }() + + var warnings []string + for rows.Next() { + var ( + runID int + snapshot string + liveSteps int + ) + if err := rows.Scan(&runID, &snapshot, &liveSteps); err != nil { + return nil + } + var frozen struct { + Scope []string `json:"scope"` + } + if err := json.Unmarshal([]byte(snapshot), &frozen); err != nil { + continue + } + if sameScope(frozen.Scope, live) { + continue + } + warnings = append(warnings, fmt.Sprintf( + "%s is bound into %s, which froze its scope at activation as %s "+ + "(§5.1.1) — the %d step(s) still live there will render that "+ + "frozen scope in their packets and record their diffs over it, "+ + "NOT the %s you just wrote. The live column does take effect "+ + "for scheduling's mutual-exclusion check, and for nothing else. "+ + "To make the widened scope real for this run's remaining work, "+ + "refresh its snapshot — "+ + "`docket run refresh-scope %s --issue %s --reason \"scope widened\"` "+ + "(DKT-869; refuses while any of the issue's steps is claimed, "+ + "running or gated, or while a dispatch is open). If the run's "+ + "PREMISE changed rather than one declaration, the older "+ + "disposition still applies: take the issue out of the run — "+ + "`docket run abandon %s --issue %s --reason \"scope widened\"` "+ + "— and re-plan it into a new run, whose activation snapshots "+ + "the scope afresh", + model.FormatID(issueID), model.FormatRunID(runID), + renderScope(frozen.Scope), liveSteps, renderScope(live), + model.FormatRunID(runID), model.FormatID(issueID), + model.FormatRunID(runID), model.FormatID(issueID), + )) + } + if rows.Err() != nil { + return nil + } + return warnings +} + +// terminalStepStatuses is db.StepTerminal's membership, as the SQL above needs +// it. It is derived from the constants rather than spelled as literals so a +// tenth status added to the machine cannot go stale here — TestStepTerminal +// StatusesMatchesStepTerminal pins the two together. +var terminalStepStatuses = []string{ + db.StepDone, db.StepSkipped, db.StepSuperseded, db.StepFailedRouted, +} + +func terminalStepStatusArgs() []any { + args := make([]any, 0, len(terminalStepStatuses)+3) + for _, s := range terminalStepStatuses { + args = append(args, s) + } + return args +} + +// placeholders renders `?, ?, …` for an IN clause of n values. +func placeholders(n int) string { + return strings.TrimSuffix(strings.Repeat("?, ", n), ", ") +} + +// liveIssueScope reads `issues.scope_globs` — the column `--scope` writes. +func liveIssueScope(conn *sql.DB, issueID int) ([]string, error) { + var stored sql.NullString + if err := conn.QueryRow( + `SELECT scope_globs FROM issues WHERE id = ?`, issueID, + ).Scan(&stored); err != nil { + return nil, err + } + return decodeScope(stored.String) +} + +// sameScope compares two scope declarations IN ORDER, because the order is the +// author's and the snapshot echoes it back verbatim (issue_scope.go). Two +// declarations of the same globs in a different order are still a real edit to +// what a re-activation would freeze, so treating them as equal would hide one. +func sameScope(a, b []string) bool { + if len(a) != len(b) { + return false + } + for i := range a { + if a[i] != b[i] { + return false + } + } + return true +} + +// renderScope names a scope declaration for a human. It distinguishes the +// undeclared case from the declared-empty one, which issue_scope.go keeps +// apart on purpose and which a bare `[]` would collapse. +func renderScope(globs []string) string { + if len(globs) == 0 { + return "no declared scope" + } + return "[" + strings.Join(globs, ", ") + "]" +} diff --git a/internal/engine/source_drift.go b/internal/engine/source_drift.go new file mode 100644 index 00000000..eef5946f --- /dev/null +++ b/internal/engine/source_drift.go @@ -0,0 +1,276 @@ +package engine + +import ( + "errors" + "fmt" + "io/fs" + "os" + "path/filepath" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// SOURCE DRIFT (DKT-590): a registered workflow's `source_path` and +// `source_sha256` are two columns nothing ever compared against each other. +// +// The measured state that produced this: `workflow show investigation --json` +// reported version 4 at sha 4cb066e3, while the file at the recorded +// `source_path` was version 8 at sha 6ed74d17 — and every other registered +// workflow on the machine was the same. RUN-40 then bound investigation@4 and +// ran it while the operator read v8 on disk. Nothing anywhere said so, because +// the path was documented as "provenance only … never re-read". +// +// Provenance nobody checks is provenance nobody can trust. This file reads it +// once, at the two moments it matters — describing a registered workflow, and +// binding one to a run — and reports the verdict. IT NEVER REPAIRS ANYTHING: +// registration and de-registration of installed files belong to the install +// path, so a drifted workflow is surfaced, never re-registered, re-hashed, or +// dropped on the operator's behalf. + +// CheckWorkflowSource reads the file at a registered workflow's recorded +// `source_path` and reports whether its bytes still hash to `source_sha256`. +// +// It hashes with workflow.SHA256 — the SAME function registration hashed with — +// so a mismatch is a real difference in bytes and never an artifact of two +// hashers disagreeing. +// +// The verdict is never an error: a missing file is a fact about the repository +// to report, not a failure of the reporting. Callers decide what each state is +// worth (see checkBoundSources for activation's disposition). +func CheckWorkflowSource(sourcePath, registeredSHA string) *model.WorkflowSourceStatus { + status := &model.WorkflowSourceStatus{ + Path: sourcePath, + RegisteredSHA256: registeredSHA, + } + + if strings.TrimSpace(sourcePath) == "" { + status.State = model.WorkflowSourceUnchecked + status.Reason = "no source path was recorded at registration " + + "(a definition registered from stdin has no file to compare against)" + return status + } + + path, err := expandHomePath(sourcePath) + if err != nil { + status.State = model.WorkflowSourceUnchecked + status.Reason = err.Error() + return status + } + + // A RELATIVE recorded path is not resolvable from here, and pretending + // otherwise is worse than declining. `docket workflow register wf.toml` + // stores exactly what was typed; resolving that against the cwd of some + // later invocation would compare the registered bytes against a DIFFERENT + // file that merely shares a name, and report drift that does not exist. + // The auto-registered corpus — the population this check exists for — + // records absolute paths (config.InstanceConfigDirs joins an absolute + // root), so nothing this check is aimed at lands here. + if !filepath.IsAbs(path) { + status.State = model.WorkflowSourceUnchecked + status.Reason = fmt.Sprintf( + "the recorded source path %q is relative, so it names no particular "+ + "file from here; only an absolute path can be compared", sourcePath) + return status + } + + src, err := os.ReadFile(path) + if err != nil { + status.State = model.WorkflowSourceUnreadable + if errors.Is(err, fs.ErrNotExist) { + status.Reason = "the file no longer exists at that path" + } else { + status.Reason = err.Error() + } + return status + } + + status.CurrentSHA256 = workflow.SHA256(src) + if status.CurrentSHA256 == registeredSHA { + status.State = model.WorkflowSourceMatches + return status + } + status.State = model.WorkflowSourceDrifted + return status +} + +// expandHomePath expands a leading `~` to the user's home directory. +// +// The repository records absolute paths everywhere it records one itself, so +// this only reaches a path an operator typed with the tilde QUOTED (`docket +// workflow register '~/wf.toml'`, which the shell hands over unexpanded). It is +// here rather than left to the caller so both call sites resolve a stored path +// identically. +func expandHomePath(path string) (string, error) { + if path != "~" && !strings.HasPrefix(path, "~"+string(os.PathSeparator)) { + return path, nil + } + home, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf( + "the recorded source path %q starts with ~ and the home directory "+ + "cannot be resolved: %v", path, err) + } + if path == "~" { + return home, nil + } + return filepath.Join(home, path[2:]), nil +} + +// DescribeWorkflowSource renders one verdict as a human-readable clause, so +// `workflow show` and `run activate` say the same thing about the same state. +func DescribeWorkflowSource(s *model.WorkflowSourceStatus) string { + if s == nil { + return "" + } + switch s.State { + case model.WorkflowSourceMatches: + return "matches — the file at this path still holds the registered bytes" + case model.WorkflowSourceDrifted: + return fmt.Sprintf( + "DRIFTED — the file at this path now hashes to sha256:%s, not the "+ + "registered sha256:%s; what is registered is NOT what is on disk", + s.CurrentSHA256, s.RegisteredSHA256) + case model.WorkflowSourceUnreadable: + return fmt.Sprintf("UNREADABLE — %s; the registered bytes are intact, "+ + "but their provenance no longer resolves", s.Reason) + default: + return fmt.Sprintf("unchecked — %s", s.Reason) + } +} + +// SourceWarning is one BOUND workflow whose registered source file no longer +// answers for the registered bytes, on the terms activation warns rather than +// refuses about (see checkBoundSources). +// +// It travels on the result for ScopeWarnings' reason: the engine holds no +// output dependency, and the verb picks the channel — stderr in human mode, an +// array in JSON. +type SourceWarning struct { + Workflow string `json:"workflow"` + // State is the model.WorkflowSourceState value, so a consumer branches on + // the verdict rather than on prose. + State string `json:"state"` + SourcePath string `json:"source_path,omitempty"` + RegisteredSHA256 string `json:"registered_sha256"` + CurrentSHA256 string `json:"current_sha256,omitempty"` + Reason string `json:"reason"` +} + +// checkBoundSources is activation's half of DKT-590: every workflow this run's +// issues bind to, checked against the bytes at its own recorded `source_path`. +// +// THE DISPOSITION TURNS ON WHETHER THE BINDING IS NEW, not on whether the +// activation is: +// +// - A binding made HERE with a drifted source REFUSES the whole activation +// (CONFLICT). It is the same fact F9's collision refusal already refuses +// on — registered bytes and file bytes disagreeing at one name@version — +// reached from the other side, and the same argument decides it: a run +// that binds and pins the registered bytes while an operator reads +// something else at that path cannot be reviewed by the person approving +// it. Every comparable engine-integrity condition here refuses (a terminal +// run, an open dispatch, a ref offered by two roots with different bytes, +// a `--pin` path that will not read); the conditions that merely WARN are +// planning omissions and environment facts — an unscoped holder, a missing +// trust entry, a context bundle over the warn cap — where proceeding is +// legitimate and often intended. Drift is not one of those. +// +// - A binding INHERITED from an earlier activation only WARNS. RA2 and F15 +// are explicit that a config file edited mid-run must not reach a run +// already under way — the pin set is inherited and nothing is re-scanned — +// and refusing here would break exactly that guarantee in its most ordinary +// form: bumping a workflow to a new version leaves the OLD version's +// recorded path holding the new version's bytes, which is the retro loop +// working as designed, and it would wedge every in-flight run bound to the +// old version with no remedy but restoring a file. +// +// An UNREADABLE source always warns, however the binding was made. The +// registered bytes are intact and still reproduce — only their provenance is +// gone — and refusing would wedge every activation in a store holding a +// workflow whose file was moved or deleted, a state only the install path can +// resolve. +// +// AND `registration.auto = false` DOWNGRADES DRIFT TO A WARNING TOO. That +// toggle's documented meaning is "I don't want silent version upgrades: bind +// what is REGISTERED, not what the corpus now says" — so a registry lagging the +// files is the state the operator asked for, not a divergence nobody decided, +// and refusing over it would turn the off switch into a wedge the moment the +// corpus moved. With adoption enabled (the default) the same drift means +// something genuinely went unnoticed: the scan would have adopted a bumped +// version, or refused under F9 on changed bytes at an unchanged version, so a +// bound workflow whose file disagrees is a divergence no one has ruled on. +func checkBoundSources( + runIssues []*db.RunIssue, bindings map[int]*boundDefinition, inherited map[int]bool, + autoRegister bool, +) ([]SourceWarning, error) { + // One check per distinct bound workflow, in first-bound order, so the + // report is deterministic and a workflow bound by six issues is named once. + // Freshness is folded across ALL of a workflow's bindings first: a workflow + // bound freshly by one issue and inherited by another is a NEW binding, and + // the strict disposition applies. + type bound struct { + wf *model.Workflow + fresh bool + } + order := make([]int, 0, len(bindings)) + seen := make(map[int]*bound, len(bindings)) + for _, ri := range runIssues { + def := bindings[ri.IssueID] + if def == nil { + continue + } + id := def.workflow.ID + if b, ok := seen[id]; ok { + b.fresh = b.fresh || !inherited[ri.IssueID] + continue + } + seen[id] = &bound{wf: def.workflow, fresh: !inherited[ri.IssueID]} + order = append(order, id) + } + + var warnings []SourceWarning + for _, id := range order { + b := seen[id] + status := CheckWorkflowSource(b.wf.SourcePath, b.wf.SourceSHA256) + switch status.State { + case model.WorkflowSourceMatches, model.WorkflowSourceUnchecked: + continue + case model.WorkflowSourceDrifted: + if b.fresh && autoRegister { + return nil, driftedSourceErr(b.wf.Ref(), status) + } + } + warnings = append(warnings, SourceWarning{ + Workflow: b.wf.Ref(), + State: string(status.State), + SourcePath: status.Path, + RegisteredSHA256: status.RegisteredSHA256, + CurrentSHA256: status.CurrentSHA256, + Reason: DescribeWorkflowSource(status), + }) + } + return warnings, nil +} + +// driftedSourceErr is the refusal, written to be acted on without a second +// command: what disagrees, both hashes, and the two ways out — NEITHER of which +// this engine takes on the operator's behalf, because registering installed +// files is the install path's job. +func driftedSourceErr(ref string, s *model.WorkflowSourceStatus) error { + return conflictErr( + "workflow %s has DRIFTED from its registered source: the file at %s no "+ + "longer holds the bytes registered under that name@version\n\n"+ + " registered sha256:%s\n"+ + " on disk sha256:%s\n\n"+ + "This activation would bind and pin the REGISTERED bytes, so the run "+ + "would execute a definition that is not the one at that path — the "+ + "divergence would be invisible for the life of the run. A registered "+ + "name@version is frozen, so nothing here re-registers or re-hashes it: "+ + "either restore %s to the registered bytes, or register its current "+ + "bytes at a new [pipeline].version through the install path, then "+ + "activate again", + ref, s.Path, s.RegisteredSHA256, s.CurrentSHA256, s.Path) +} diff --git a/internal/engine/stale_waiver_test.go b/internal/engine/stale_waiver_test.go new file mode 100644 index 00000000..5df411c2 --- /dev/null +++ b/internal/engine/stale_waiver_test.go @@ -0,0 +1,344 @@ +package engine + +import ( + "os/exec" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-742, both halves. +// +// HALF ONE — detection completeness. IsAncestorFn's `known = false` conflated +// "git could not answer" with "the object is not in the shared store at all", +// and staleTargets skipped both silently. RUN-52's DKT-V253 was the second +// state: a three-seat vote panel each ran `git cat-file -t` on the packet's +// target, found no object anywhere, and no warning had fired. The absence +// probe (ObjectExistsFn) closes exactly that gap, and ONLY that gap: a +// genuinely unanswerable existence question stays as silent as it ever was. +// +// HALF TWO — waiver memory. A stale-target warning an operator investigated +// and ruled acceptable re-fired unchanged at every subsequent dispatch +// open/verify of the same (step, target) pair — four times in RUN-52 — with +// the standing ruling living only in session memory. A run-scoped waiver +// (`dispatch waive-target`) makes it engine-visible; the signature is the +// pair alone, so a different sha or an unnamed row still warns. + +// TestDispatchWarnsWhenTargetObjectIsAbsent: ancestry unanswerable, existence +// DEFINITIVELY no — the DKT-V253 shape. Every consuming row warns, marked +// `absent`, with the reason naming the cat-file probe rather than a +// divergence nothing measured. +func TestDispatchWarnsWhenTargetObjectIsAbsent(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, false } + e.ObjectExistsFn = func(_, sha string) (exists, known bool) { + if sha != "cafe1234cafe1234" { + t.Errorf("existence asked about %q, want the recorded head", sha) + } + return false, true + } + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 4 { + t.Fatalf("stale targets = %d, want the four review siblings — an absent "+ + "object must warn, not be skipped as unanswerable: %+v", + len(m.StaleTargets), m.StaleTargets) + } + for _, s := range m.StaleTargets { + if !s.Absent { + t.Errorf("%s is not marked absent: %+v", s.Instance, s) + } + if !strings.Contains(s.Reason, "does not resolve as a commit") || + !strings.Contains(s.Reason, "cat-file") { + t.Errorf("%s reason %q does not name the absence or the probe", + s.Instance, s.Reason) + } + // The divergence wording must NOT appear: nothing measured a + // divergence, and the two advisory shapes may not blur (DKT-415's + // discipline applied to the third shape). + if strings.Contains(s.Reason, "not an ancestor") { + t.Errorf("%s reason %q claims an ancestry fact that was unanswerable", + s.Instance, s.Reason) + } + // DKT-415: the claim-time semantics still ride every shape. + if !strings.Contains(s.Reason, "does not re-derive the target from HEAD") || + !strings.Contains(s.Reason, "resolves at claim time") { + t.Errorf("%s reason %q dropped the claim-time semantics", + s.Instance, s.Reason) + } + } +} + +// An object that EXISTS while ancestry is unanswerable stays silent: the +// probe accuses on definitive absence only, never on the ancestry question it +// could not answer. +func TestDispatchStaysQuietWhenObjectExistsButAncestryUnanswerable(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, false } + e.ObjectExistsFn = func(_, _ string) (exists, known bool) { return true, true } + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 0 { + t.Errorf("a present object with unanswerable ancestry was flagged: %+v", + m.StaleTargets) + } +} + +// An engine with no existence probe wired keeps the pre-DKT-742 behavior +// exactly: unanswerable ancestry stays silent. +func TestMissingExistenceProbeKeepsTheUnansweredCaseSilent(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, false } + e.ObjectExistsFn = nil + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 0 { + t.Errorf("an unanswerable question warned with no probe wired: %+v", + m.StaleTargets) + } +} + +// TestGitCommitResolvable drives the real implementation across its +// three-valued contract: present commit, definitively absent object, an +// object that exists but is not a commit, and the two unanswerable shapes. +func TestGitCommitResolvable(t *testing.T) { + if _, err := exec.LookPath("git"); err != nil { + t.Skip("git is not installed") + } + repo := t.TempDir() + gitRun(t, repo, "init", "-q") + writeFile(t, repo, "a.txt", "content\n") + gitRun(t, repo, "add", "-A") + gitRun(t, repo, "commit", "-q", "-m", "base") + commit := gitRun(t, repo, "rev-parse", "HEAD") + blob := gitRun(t, repo, "rev-parse", "HEAD:a.txt") + + cases := []struct { + name string + execRoot, sha string + exists, known bool + }{ + {"a present commit", repo, commit, true, true}, + {"an absent object", repo, "0123456789abcdef0123456789abcdef01234567", false, true}, + {"a blob, not a commit", repo, blob, false, true}, + {"no repository", t.TempDir(), commit, false, false}, + {"empty inputs", "", "", false, false}, + } + for _, c := range cases { + exists, known := gitCommitResolvable(c.execRoot, c.sha) + if exists != c.exists || known != c.known { + t.Errorf("%s: = (%v, %v), want (%v, %v)", + c.name, exists, known, c.exists, c.known) + } + } +} + +// TestDispatchWarnsAbsentTargetRecordedOutsideTheSharedStore is DKT-V253's +// shape end to end with real git: the executor commits in a checkout whose +// object store the shared checkout does NOT share (a separate clone — the +// same absence a pruned-then-GC'd linked worktree leaves), the step records +// that head, and dispatch open must warn that the target resolves from +// nowhere the consumers can reach — the case that previously produced NO +// warning at all. +func TestDispatchWarnsAbsentTargetRecordedOutsideTheSharedStore(t *testing.T) { + if _, err := exec.LookPath("git"); err != nil { + t.Skip("git is not installed") + } + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + shared := t.TempDir() + gitRun(t, shared, "init", "-q") + writeFile(t, shared, "internal/work.txt", "original\n") + gitRun(t, shared, "add", "-A") + gitRun(t, shared, "commit", "-q", "-m", "base") + + // The executor's checkout: a clone, so its new commit's object never + // enters the shared store. + clone := t.TempDir() + gitRun(t, clone, "clone", "-q", shared, ".") + writeFile(t, clone, "internal/work.txt", "the executor's change\n") + gitRun(t, clone, "add", "-A") + gitRun(t, clone, "commit", "-q", "-m", "implement the issue") + target := gitRun(t, clone, "rev-parse", "HEAD") + + execSQL(t, conn, `UPDATE runs SET exec_root = ? WHERE id = ?`, shared, run.ID) + implementID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implementID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, implementID, CompleteOptions{ + Token: claim.Token, + Artifact: []byte("the change summary"), + WorkDir: clone, + NowMS: nowMS, + }) + testsupport.Must(t, err, "complete implement: %v", err) + + // THE PREMISE, ASSERTED: the recorded target really is absent from the + // shared store (the acceptance criterion's own probe), and ancestry really + // is unanswerable — the exact state that used to skip silently. + if exists, known := gitCommitResolvable(shared, target); exists || !known { + t.Fatalf("premise broken: existence = (%v, %v), want a definitive NO — "+ + "the clone's commit must not be in the shared store", exists, known) + } + if _, known := gitAncestorOfHead(shared, target); known { + t.Fatal("premise broken: ancestry answered about an absent object, so " + + "this case no longer covers the silent-skip gap") + } + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 4 { + t.Fatalf("stale targets = %d, want the four review siblings — the "+ + "absent-object case fired no warning: %+v", + len(m.StaleTargets), m.StaleTargets) + } + for _, s := range m.StaleTargets { + if !s.Absent || s.TargetSHA != target { + t.Errorf("%s: absent=%v target=%q, want absent with the recorded head %q", + s.Instance, s.Absent, s.TargetSHA, target) + } + } +} + +// TestWaiverSuppressesAdjudicatedStaleTarget: the AC's companion half. A +// warning acknowledged once for a (step, target) pair does not re-fire +// unchanged on subsequent open/verify of the same pair — and the waiver's +// sha may be the 12-character prefix the warning itself renders. +func TestWaiverSuppressesAdjudicatedStaleTarget(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, true } + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 4 { + t.Fatalf("premise: stale targets = %d, want 4", len(m.StaleTargets)) + } + + // The operator adjudicates three of the four rows, copying the sha at the + // advisory's own 12-character rendering. + waived, err := e.WaiveStaleTargets(conn, run.ID, + []string{"review@0#0", "review@0#1", "review@0#2"}, + "cafe1234cafe", "the divergence is the later format pass", nowMS) + testsupport.Must(t, err, "waive: %v", err) + if len(waived) != 3 { + t.Fatalf("waivers minted = %d, want 3: %+v", len(waived), waived) + } + + result, mismatch, err := e.VerifyDispatch(conn, run.ID, nowMS) + testsupport.Must(t, err, "dispatch verify: %v", err) + if mismatch != nil { + t.Fatalf("verify mismatch: %+v", mismatch) + } + if len(result.StaleTargets) != 1 || result.StaleTargets[0].Instance != "review@0#3" { + t.Fatalf("post-waiver stale targets = %+v, want exactly the unwaived "+ + "review@0#3", result.StaleTargets) + } + + // The waivers are event-logged: standing precedent must be findable in + // the feed, or a warning that stopped appearing is indistinguishable from + // a warning that stopped being true. + var events int + err = conn.QueryRow( + `SELECT COUNT(*) FROM events WHERE run_id = ? AND kind = 'stale-target-waived'`, + run.ID).Scan(&events) + testsupport.Must(t, err, "counting waiver events: %v", err) + if events != 3 { + t.Errorf("stale-target-waived events = %d, want one per waiver", events) + } +} + +// TestWaiverDoesNotCoverADifferentSignature: a different sha on the waived +// row, and the waived sha on an unnamed row, both still warn — a new +// divergence never rides an old ruling. +func TestWaiverDoesNotCoverADifferentSignature(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, true } + + // A waiver for a DIFFERENT sha on every row: nothing may be suppressed. + _, err := e.WaiveStaleTargets(conn, run.ID, + []string{"review@0#0", "review@0#1", "review@0#2", "review@0#3"}, + "beefbeefbeef", "ruled on some other target", nowMS) + testsupport.Must(t, err, "waive: %v", err) + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 4 { + t.Errorf("stale targets = %d, want all 4 — a waiver for another sha "+ + "suppressed a warning it never ruled on: %+v", + len(m.StaleTargets), m.StaleTargets) + } +} + +// A waiver covers the ABSENT advisory shape too: the adjudication is about +// the (step, target) pair, whichever reason the pair warned with. +func TestWaiverSuppressesAbsentTargetWarning(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + staleFixture(t, conn, e) + e.IsAncestorFn = func(_, _ string) (ancestor, known bool) { return false, false } + e.ObjectExistsFn = func(_, _ string) (exists, known bool) { return false, true } + + _, err := e.WaiveStaleTargets(conn, run.ID, + []string{"review@0#0", "review@0#1", "review@0#2", "review@0#3"}, + "cafe1234cafe1234", "seats judge the integrated successor instead", nowMS) + testsupport.Must(t, err, "waive: %v", err) + + m, err := e.OpenDispatch(conn, run.ID, 0, nil, nowMS) + testsupport.Must(t, err, "dispatch open: %v", err) + if len(m.StaleTargets) != 0 { + t.Errorf("a waived absent-target warning re-fired: %+v", m.StaleTargets) + } +} + +// The verb's own refusals: a sha that is not hex (or too short to be an +// unambiguous prefix), an empty instance list, and a run that does not exist. +func TestWaiveStaleTargetsRefusesBadInputs(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + for _, c := range []struct { + name string + instances []string + sha string + }{ + {"no instances", nil, "cafe1234cafe"}, + {"an empty instance", []string{""}, "cafe1234cafe"}, + {"a non-hex sha", []string{"review@0#0"}, "not-a-sha!!"}, + {"a too-short prefix", []string{"review@0#0"}, "cafe12"}, + } { + if _, err := e.WaiveStaleTargets(conn, run.ID, c.instances, c.sha, "", nowMS); err == nil { + t.Errorf("%s: the waiver was recorded", c.name) + } + } + + if _, err := e.WaiveStaleTargets(conn, 999999, + []string{"review@0#0"}, "cafe1234cafe", "", nowMS); err == nil { + t.Error("a waiver was recorded against a run that does not exist") + } +} diff --git a/internal/engine/strand.go b/internal/engine/strand.go index 0bd472c9..67c64b72 100644 --- a/internal/engine/strand.go +++ b/internal/engine/strand.go @@ -39,7 +39,15 @@ import ( // A conductor that does not use it loses nothing it had: its proposal simply is // not auto-closed, exactly as today. func ReapAckProposalKey(runID int, seq int64) string { - return "reap-ack:" + strconv.Itoa(runID) + ":" + strconv.FormatInt(seq, 10) + return reapAckRunPrefix(runID) + strconv.FormatInt(seq, 10) +} + +// reapAckRunPrefix is the key family of ONE RUN's reap-ack ballots — the part +// of the key above that does not vary per reap. It exists so the report's +// vote-usage attribution (DKT-584) and the writer above cannot spell the key +// differently, exactly as voteIdempotencyPrefix guards its own family. +func reapAckRunPrefix(runID int) string { + return "reap-ack:" + strconv.Itoa(runID) + ":" } // closeRunProposalsTx closes every OPEN proposal this run's vote steps opened, diff --git a/internal/engine/union_config_test.go b/internal/engine/union_config_test.go index 7c282c6f..1585d055 100644 --- a/internal/engine/union_config_test.go +++ b/internal/engine/union_config_test.go @@ -2,6 +2,7 @@ package engine import ( "database/sql" + "fmt" "os" "os/exec" "path/filepath" @@ -53,6 +54,18 @@ func unionRepo(t *testing.T) (conn *sql.DB, shared, repo string) { return conn, filepath.Join(home, ".docket", "config"), filepath.Join(work, ".docket", "config") } +// packetWorkflow renders autoWorkflowSrc with the given `packet` entries on +// its one step. Since DKT-581 activation pins only the packet CLOSURE the +// bound workflows reach, a union test whose subject is a pinned contract must +// have the workflow actually declare it — exactly as a real corpus does. +func packetWorkflow(entries ...string) string { + quoted := make([]string, len(entries)) + for i, e := range entries { + quoted[i] = fmt.Sprintf("%q", e) + } + return autoWorkflowSrc + "packet = [" + strings.Join(quoted, ", ") + "]\n" +} + // canonical resolves a path the way the scan does, so an assertion compares the // form the engine records rather than the form the test typed (macOS resolves // /var to /private/var under every temp directory). @@ -158,7 +171,8 @@ func TestUnionRegistersDisjointRootsSchemasFirst(t *testing.T) { // the file. func TestUnionPinsBothRoots(t *testing.T) { conn, shared, repo := unionRepo(t) - writeConfigFile(t, shared, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, shared, "workflows/auto-dev.toml", + packetWorkflow("contracts/house-style.md", "contracts/project.md")) writeConfigFile(t, shared, "contracts/house-style.md", "the corpus's contract\n") writeConfigFile(t, repo, "contracts/project.md", "this repo's contract\n") @@ -190,8 +204,9 @@ func TestUnionPinsBothRoots(t *testing.T) { // pin would double-count the same bytes against the closure caps. func TestUnionIdenticalDuplicateIsANoOp(t *testing.T) { conn, shared, repo := unionRepo(t) - writeConfigFile(t, shared, "workflows/auto-dev.toml", autoWorkflowSrc) - writeConfigFile(t, repo, "workflows/auto-dev.toml", autoWorkflowSrc) + vendored := packetWorkflow("contracts/fix.md") + writeConfigFile(t, shared, "workflows/auto-dev.toml", vendored) + writeConfigFile(t, repo, "workflows/auto-dev.toml", vendored) writeConfigFile(t, shared, "contracts/fix.md", "identical\n") writeConfigFile(t, repo, "contracts/fix.md", "identical\n") @@ -302,8 +317,9 @@ func TestSharedRootSymlinkScansLikeARealDirectory(t *testing.T) { conn, shared, _ := unionRepo(t) // The real corpus, somewhere else entirely, with ONE link pointing at it. + src := packetWorkflow("contracts/fix.md") real := filepath.Join(t.TempDir(), "corpus", "config") - writeConfigFile(t, real, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, real, "workflows/auto-dev.toml", src) writeConfigFile(t, real, "contracts/fix.md", "a contract\n") err := os.MkdirAll(filepath.Dir(shared), 0o755) @@ -321,7 +337,7 @@ func TestSharedRootSymlinkScansLikeARealDirectory(t *testing.T) { t.Fatalf("registered %d files through the link, want 1: %+v", len(result.Registered), result.Registered) } - if got, want := result.Registered[0].SHA256, workflow.SHA256([]byte(autoWorkflowSrc)); got != want { + if got, want := result.Registered[0].SHA256, workflow.SHA256([]byte(src)); got != want { t.Errorf("registered sha256 %s through the link, want %s — the same bytes a "+ "real root registers", got, want) } @@ -381,7 +397,7 @@ func TestDanglingConfigRootSymlinkRefuses(t *testing.T) { // own duplicate and turn a legal install into a self-conflict. func TestOneDirectoryReachedTwiceIsOneRoot(t *testing.T) { conn, shared, repo := unionRepo(t) - writeConfigFile(t, repo, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, repo, "workflows/auto-dev.toml", packetWorkflow("contracts/fix.md")) writeConfigFile(t, repo, "contracts/fix.md", "a contract\n") err := os.MkdirAll(filepath.Dir(shared), 0o755) @@ -448,7 +464,7 @@ func TestPacketResolutionPrefersTheFirstRoot(t *testing.T) { pins := testPinSet(t, first, "checklists/a.md") pins["checklists/a.md"] = pins[filepath.Join(first, "checklists/a.md")] - files, err := resolvePacketFiles(pins, []string{first, second}, + files, err := resolvePacketFiles("RUN-1", pins, []string{first, second}, []string{"checklists/a.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) if len(files) != 1 || strings.TrimSpace(files[0].Body) != "FROM THE SHARED ROOT" { @@ -467,7 +483,7 @@ func TestPacketResolutionFallsThroughToALaterRoot(t *testing.T) { pins := testPinSet(t, second, "checklists/a.md") pins["checklists/a.md"] = pins[filepath.Join(second, "checklists/a.md")] - files, err := resolvePacketFiles(pins, []string{first, second}, + files, err := resolvePacketFiles("RUN-1", pins, []string{first, second}, []string{"checklists/a.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) if len(files) != 1 || strings.TrimSpace(files[0].Body) != "ONLY IN THE SECOND ROOT" { @@ -490,7 +506,7 @@ func TestSharedRootPacketResolvesFromALinkedWorktree(t *testing.T) { } conn, shared, _ := unionRepo(t) - writeConfigFile(t, shared, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, shared, "workflows/auto-dev.toml", packetWorkflow("contracts/fix.md")) writeConfigFile(t, shared, "contracts/fix.md", "the corpus's contract\n") // The main checkout is the cwd unionRepo chdir'd into. @@ -520,7 +536,8 @@ func TestSharedRootPacketResolvesFromALinkedWorktree(t *testing.T) { } files, err := resolvePacketFiles( - packetPinsForRun(pins), instanceConfigRoots(), []string{"contracts/fix.md"}) + "RUN-1", packetPinsForRun(pins), instanceConfigRoots(), + []string{"contracts/fix.md"}) testsupport.Must(t, err, "a packet file in the shared root did not resolve from a "+ "linked worktree — the failure that stranded a claimed step: %v", err) @@ -534,7 +551,7 @@ func TestSharedRootPacketResolvesFromALinkedWorktree(t *testing.T) { // it in the list. func TestRepoRootPacketStillResolves(t *testing.T) { conn, shared, repo := unionRepo(t) - writeConfigFile(t, shared, "workflows/auto-dev.toml", autoWorkflowSrc) + writeConfigFile(t, shared, "workflows/auto-dev.toml", packetWorkflow("contracts/project.md")) writeConfigFile(t, repo, "contracts/project.md", "this repo's contract\n") issue := createIssue(t, conn, "repo packet", "body", "task", nil) @@ -546,7 +563,8 @@ func TestRepoRootPacketStillResolves(t *testing.T) { testsupport.Must(t, err, "listing pins: %v", err) files, err := resolvePacketFiles( - packetPinsForRun(pins), instanceConfigRoots(), []string{"contracts/project.md"}) + "RUN-1", packetPinsForRun(pins), instanceConfigRoots(), + []string{"contracts/project.md"}) testsupport.Must(t, err, "resolvePacketFiles: %v", err) if len(files) != 1 || strings.TrimSpace(files[0].Body) != "this repo's contract" { t.Errorf("resolved %+v, want the repository's own contract", files) diff --git a/internal/engine/verify_pins.go b/internal/engine/verify_pins.go index 489ad129..0dbe07ac 100644 --- a/internal/engine/verify_pins.go +++ b/internal/engine/verify_pins.go @@ -7,6 +7,7 @@ import ( "os" "path/filepath" "sort" + "strings" "github.com/ALT-F4-LLC/docket/internal/db" "github.com/ALT-F4-LLC/docket/internal/model" @@ -24,6 +25,10 @@ const ( PinChanged = "changed" // PinMissing: the run depends on the ref and it is no longer there. PinMissing = "missing" + // PinUnpinnedReference: not a verdict on a pin — there IS no pin. The + // pinned bytes reach a file this run never snapshotted, so every step whose + // packet resolves it refuses at claim (VALIDATION_ERROR). + PinUnpinnedReference = "unpinned-reference" ) // PinVerdict is one pin, checked. @@ -41,20 +46,56 @@ type PinVerdict struct { Path string `json:"path,omitempty"` } +// ReferenceVerdict is one ref the run's own pinned bytes reach that the run +// does not pin (DKT-821). +// +// It is deliberately NOT a PinVerdict. A pin verdict answers "do these bytes +// still match", and there are no pinned bytes here to match; conflating the two +// would also hand every consumer of `Pins` — repin's dispositions, the +// `run status` drift advisory — a row with no pinned hash to reason about. +type ReferenceVerdict struct { + // Status is always PinUnpinnedReference; it is carried per row so a + // consumer branches on one field name across both lists. + Status string `json:"status"` + Ref string `json:"ref"` + // IncludedBy is the closure file(s) whose `packet_includes` name the ref — + // the contract an operator actually edits. Empty when a step's own `packet` + // entry is what reaches it. + IncludedBy []string `json:"included_by,omitempty"` + // RequiredBy is the pending step(s) (and unexpanded phases) that can still + // open the ref — the claims that will refuse. + RequiredBy []string `json:"required_by"` + // Path is where the file sits on disk, when it does. Present-but-unpinned + // is the RUN-59 case and its remedy differs from absent-and-unpinned, the + // same distinction DKT-818 drew in the claim-time refusal. + Path string `json:"path,omitempty"` +} + // PinReport is `docket run verify-pins`. type PinReport struct { Run string `json:"run"` // Pins is every pin the run holds, in a total order (kind, then ref), so // two checks of one unchanged run produce identical output. Pins []PinVerdict `json:"pins"` - // Changed and Missing are the counts a caller branches on without - // re-walking Pins. - Changed int `json:"changed"` - Missing int `json:"missing"` + // References is the CLOSURE check, kept beside the per-pin check rather + // than mixed into it: a conductor reading this needs to know which of the + // two failed, because they have different remedies and only one of them is + // about the filesystem. Empty (never nil), in ref order. + References []ReferenceVerdict `json:"references"` + // Changed, Missing and Unpinned are the counts a caller branches on without + // re-walking the lists. + Changed int `json:"changed"` + Missing int `json:"missing"` + Unpinned int `json:"unpinned"` } -// Sound reports whether every pin still matches. -func (r *PinReport) Sound() bool { return r.Changed == 0 && r.Missing == 0 } +// Sound reports whether every pin still matches AND the pin set is closed — +// the two halves of "is this run's pin story healthy". A run can fail either +// half alone: RUN-59 had all 30 pins matching disk and four judge steps that +// could not be claimed. +func (r *PinReport) Sound() bool { + return r.Changed == 0 && r.Missing == 0 && r.Unpinned == 0 +} // VerifyPins checks a run's WHOLE pin set against what those refs resolve to // now (DKT-297). @@ -77,7 +118,49 @@ func (r *PinReport) Sound() bool { return r.Changed == 0 && r.Missing == 0 } // a bundle depend on the working tree. This verb is the deliberate opposite — // it asks about the tree, on purpose, and writes nothing. func VerifyPins(conn *sql.DB, runID int) (*PinReport, error) { - return verifyPinsIn(conn, runID, instanceConfigRoots()) + return verifyPinsClosedIn(conn, runID, instanceConfigRoots()) +} + +// verifyPinsClosedIn is the whole-run answer: the per-pin check, plus the +// CLOSURE check the per-pin check cannot see (DKT-821). +// +// Matching every pinned ref against disk is not the same question as "is this +// run's pin story healthy", and RUN-59 is the run where the two answers +// diverged. A repin had adopted contract bytes whose `packet_includes` reached +// two fragments the run never pinned; every one of those 30 pins matched its +// file exactly, `verify-pins` said exit 0, and minutes earlier four review +// claims had already died on `packet file ... is not pinned by this run`. The +// verb whose job is the whole-run question reported a structurally wedged run +// as healthy, and a conductor used it as a pre-dispatch health check. +// +// So after checking the pins, resolve what those pinned bytes REFERENCE and +// report anything the pin set does not hold. The walk is unpinnedClosureRefs, +// shared verbatim with the additions repin computes when it adopts bytes +// (DKT-805): one computation, so the verb that detects the hole and the verb +// that closes it can never name different refs. +func verifyPinsClosedIn(conn *sql.DB, runID int, roots []string) (*PinReport, error) { + report, err := verifyPinsIn(conn, runID, roots) + if err != nil { + return nil, err + } + closure, err := pendingPacketClosure(conn, runID, roots) + if err != nil { + return nil, err + } + for _, u := range unpinnedClosureRefs(report.Pins, closure, roots) { + // A ref that is there and unreadable is still an unpinned reference — + // readability is the claim-time question, and this one is about the pin + // set. Naming the path is what lets a reader tell the two apart. + report.References = append(report.References, ReferenceVerdict{ + Status: PinUnpinnedReference, + Ref: u.ref, + IncludedBy: u.includedBy, + RequiredBy: u.requiredBy, + Path: u.path, + }) + report.Unpinned++ + } + return report, nil } // verifyPinsIn is VerifyPins over an explicit root list, which is the same @@ -94,7 +177,12 @@ func verifyPinsIn(conn *sql.DB, runID int, roots []string) (*PinReport, error) { return nil, err } - report := &PinReport{Run: model.FormatRunID(runID), Pins: []PinVerdict{}} + report := &PinReport{ + Run: model.FormatRunID(runID), + // Both lists are empty rather than nil so `--json` emits arrays on a + // clean run — the same wire shape a consumer parses either way. + Pins: []PinVerdict{}, References: []ReferenceVerdict{}, + } for _, p := range pins { v := PinVerdict{Kind: p.Kind, Ref: p.Ref, Pinned: p.SHA256} @@ -226,6 +314,21 @@ func PinReportReason(r *PinReport) string { "%s %s is pinned at %s but does not resolve", v.Kind, v.Ref, v.Pinned)) } } + // The closure clauses come after the per-pin ones and read differently on + // purpose: nothing drifted, so neither hash belongs here — what a reader + // needs is the file that wrote the reference and the ref it names. + for _, v := range r.References { + by := strings.Join(v.IncludedBy, ", ") + if by == "" { + by = strings.Join(v.RequiredBy, ", ") + } + where := "and no instance-config root holds it" + if v.Path != "" { + where = "though " + v.Path + " holds it" + } + out = append(out, fmt.Sprintf( + "%s references %s, which %s does not pin (%s)", by, v.Ref, r.Run, where)) + } return joinClauses(out) } diff --git a/internal/engine/vote.go b/internal/engine/vote.go index 104f3e21..00f154a3 100644 --- a/internal/engine/vote.go +++ b/internal/engine/vote.go @@ -45,6 +45,10 @@ type VoteOutcome struct { // Score is the weighted score the existing tally computed, when there is // one. Score *float64 + // Required is the proposal's `required_voters` — the denominator a ballot + // count is read against (DKT-895). Carried because the proposal read below + // already has it; nothing recomputes quorum from it. + Required int } // voteStepKey identifies one run's vote step uniquely across every issue it @@ -250,7 +254,35 @@ func ReadVoteOutcome(conn *sql.DB, step *db.Step, spec *workflow.Step) (*VoteOut if spec.Type != workflow.TypeVote { return nil, nil } + return readVoteProposalOutcome(conn, step) +} + +// ReadStepVoteOutcome is ReadVoteOutcome asked of the STEP ROW rather than the +// pinned spec: the same single read, keyed off `step.Kind`. +// +// It exists for callers that have a step and no spec, and must not grow a +// second read of proposals to compensate (DKT-726). `step resolve` is the +// motivating one — a resolution is offered on a vote step whatever its status +// (R11), so the refusal it needs to compute has to be decidable before the +// pinned definition is even loaded, and for a MATERIALIZED step the definition +// never declares the minted name at all, so `workflow.StepByName` returns nil +// there and the spec-keyed reader could never be called. +// +// The MINTED KIND is the authority for what a step is — the same fact the +// `resolvable` test and the parked-vote refusal above it already key off. A +// step whose kind is not `vote` has no proposal by construction, and reports +// nothing. +func ReadStepVoteOutcome(conn *sql.DB, step *db.Step) (*VoteOutcome, error) { + if step.Kind != workflow.TypeVote { + return nil, nil + } + return readVoteProposalOutcome(conn, step) +} +// readVoteProposalOutcome is the one read both entry points share: resolve the +// step's proposal through its idempotency key and observe the status the tally +// already wrote. +func readVoteProposalOutcome(conn *sql.DB, step *db.Step) (*VoteOutcome, error) { proposalID, err := findVoteProposal(conn, step) if err != nil || proposalID == 0 { return nil, err @@ -265,6 +297,7 @@ func ReadVoteOutcome(conn *sql.DB, step *db.Step, spec *workflow.Step) (*VoteOut ProposalID: proposalID, Status: proposal.Status, Score: proposal.WeightedScore, + Required: proposal.RequiredVoters, } switch proposal.Status { case model.ProposalStatusApproved, model.ProposalStatusCommitted: @@ -369,6 +402,41 @@ func recordVoteEvent( return tx.Commit() } +// voteTallyDetail renders the `vote-tallied` event's detail: the proposal, the +// status, the WEIGHTED SCORE, and the BALLOT COUNT, each labelled. +// +// DKT-895: the detail used to read `DKT-V289 approved (1)`, where `1` was the +// weighted score printed bare. On RUN-62 a three-ballot unanimous +// approve-with-concerns scored 1.00 and rendered as `(1)` — indistinguishable +// from "one ballot", so a reader watching the feed saw a panel that had lost +// two of its three seats and could only disprove it with `docket vote show`. +// Two numbers with one pair of parentheses between them is the whole defect: +// the fix labels both and prints the score the way every other surface does +// (`%.2f`, matching `vote show`'s "Weighted score" line, so the two agree +// digit for digit). +// +// The ballot count is the ONE extra read, taken here at tally time rather than +// on VoteOutcome, so the per-invocation outcome read (readVoteProposalOutcome, +// which every dispatch does for every vote step) does not grow a second query +// to serve an event detail written once — DKT-726's rule about that reader. +// `RequiredVoters` is free: the proposal that read already loaded carries it. +func voteTallyDetail(conn *sql.DB, outcome *VoteOutcome) (string, error) { + score := "none" + if outcome.Score != nil { + score = strconv.FormatFloat(*outcome.Score, 'f', 2, 64) + } + + votes, err := db.GetProposalVotes(conn, outcome.ProposalID) + if err != nil { + return "", fmt.Errorf("reading the casts of %s for its tally event: %w", + model.FormatProposalID(outcome.ProposalID), err) + } + + return fmt.Sprintf("%s %s score=%s ballots=%d/%d", + model.FormatProposalID(outcome.ProposalID), outcome.Status, + score, len(votes), outcome.Required), nil +} + // routeVoteStep is §8.1 phase 5: the ordinary routing transaction, over the // verdict the tally produced. // @@ -377,6 +445,11 @@ func recordVoteEvent( // `fail` ⇒ routed per the step's EFFECTIVE on_fail — identically to a human // gate's reject. // +// DKT-545 adds one clause between the two: an APPROVED tally on a step that +// declares a `threshold` is evaluated over the recorded casts before it +// routes pass — see evaluateVoteThreshold. A step declaring none behaves +// exactly as the paragraph above describes, byte for byte. +// // On a DECLARED vote step that on_fail should not be `waiting-human`, for the // reason V13 states about human gates: parking would make the step wait on the // resolution of the thing that just rejected it. V13 and V13a are written @@ -395,8 +468,27 @@ func routeVoteStep( outcome *VoteOutcome, nowMS int64, ) error { routing := RoutingPass - if outcome.Verdict == VerdictFail { + concernReason := "" + switch { + case outcome.Verdict == VerdictFail: routing = spec.EffectiveOnFail() + case outcome.Status == model.ProposalStatusApproved && len(spec.Threshold) > 0: + // DKT-545: an APPROVED tally with a declared `threshold` is asked one + // more question — over the CAST SET, not the tally: an approval built + // on approve-with-concerns casts can route into the same revise loop + // a rejection does, instead of the concerns evaporating. No threshold + // declared (every pre-existing workflow) means no evaluation and the + // exact prior behavior; a COMMITTED proposal skips it too, because + // §8.4's manual commit is an operator setting the final outcome by + // hand. Evaluated OUTSIDE the transaction below, like every other + // pooled read in this function. + result, err := evaluateVoteThreshold(conn, step, spec, outcome.ProposalID) + if err != nil { + return err + } + if result.Routing != RoutingPass { + routing, concernReason = result.Routing, result.Reason + } } // A MATERIALIZED held step that PASSED resolves its cluster's payload in @@ -421,13 +513,11 @@ func routeVoteStep( // The tally is announced before the routing commits, carrying the score the // EXISTING computation produced — this stage reads it, never recomputes it. - score := "no score" - if outcome.Score != nil { - score = strconv.FormatFloat(*outcome.Score, 'f', -1, 64) + detail, err := voteTallyDetail(conn, outcome) + if err != nil { + return err } - if err := recordVoteEvent(conn, EventVoteTallied, step, - fmt.Sprintf("%s %s (%s)", model.FormatProposalID(outcome.ProposalID), - outcome.Status, score), nowMS); err != nil { + if err := recordVoteEvent(conn, EventVoteTallied, step, detail, nowMS); err != nil { return err } @@ -444,6 +534,12 @@ func routeVoteStep( // with a reproduced blocker, routed `fix-loop`, and the issue closed done // with no fix step ever created. reason := string(outcome.Status) + if concernReason != "" { + // The concern routing's record names the matched predicate (or the T3 + // park's cause), because "approved" alone would read as a pass to + // anyone auditing why the step did not route pass. + reason = concernReason + } var loop *LoopOutcome routing, loop, err = applyFixLoop(tx, step, def, routing, nowMS) if err != nil { diff --git a/internal/engine/vote_record.go b/internal/engine/vote_record.go new file mode 100644 index 00000000..c2f6bddd --- /dev/null +++ b/internal/engine/vote_record.go @@ -0,0 +1,246 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "fmt" + "sort" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// Concern-aware vote routing and the `.vote-record` input form (DKT-545). +// +// The gap this closes, measured on the corpus: every tribunal tally that ever +// ran APPROVED, and none was clean — DKT-V34 passed 2-1 over a substantive +// security dissent, DKT-V140 and DKT-V160 each passed with two +// approve-with-concerns casts — and a workflow definition had no way to act on +// any of it. Only the binary approved/rejected tally reached routing +// (ReadVoteOutcome), `on_fail` fired only on a rejection, and the concern text +// lived on the proposal record where no `inputs` form could address it. The +// concerns evaporated, or the operator carried them by hand. +// +// Two pieces, mirroring machinery that already exists rather than inventing +// any: +// +// - a vote step's `threshold` table, evaluated over the CAST SET after an +// APPROVED tally — the executor threshold's exact evaluator +// (EvaluateThreshold, §6.14 T1/T4) over engine-built cast payloads, with +// the same first-match-routes order and the same no-match ⇒ pass default. +// A rejected tally still routes per `on_fail`, untouched; a step declaring +// no threshold behaves exactly as before. +// - `.vote-record`, resolving to the proposal record the named vote +// step's tally left — outcome, casts, and rationales — the gate-results +// form's reasoning (DKT-77) applied to the vote machinery, so a revise +// step reads WHAT the panel said without shelling out to `docket vote +// show`. +// +// THE TALLY IS STILL db.CastVote'S. Nothing here recomputes a weighted score +// or consults a quorum; the threshold asks a question the tally does not — +// "how many casts carried concerns" — and reads the casts the existing +// machinery recorded to answer it. + +// evaluateVoteThreshold applies a vote step's `threshold` table to its +// proposal's recorded casts (DKT-545). +// +// Called by routeVoteStep after an APPROVED tally only: +// +// - a REJECTED tally routes per `on_fail`, exactly as before — the threshold +// asks "was the approval clean", which is not a question about a rejection. +// - a COMMITTED proposal skips it too: §8.4's manual commit is an operator +// setting the final outcome by hand, and a threshold overriding that would +// re-open a question a person just closed. +// +// The schema resolver is nil — casts have no registered payload schema — which +// makes equality the whole language (T1 needs no schema) and any ordered +// comparison a T3 park. V36 refuses ordered operators at register, so a park +// here means the definition arrived without passing register-time validation +// (a restored database), and the park is the honest answer. +func evaluateVoteThreshold( + conn *sql.DB, step *db.Step, spec *workflow.Step, proposalID int, +) (ThresholdResult, error) { + votes, err := db.GetProposalVotes(conn, proposalID) + if err != nil { + return ThresholdResult{}, fmt.Errorf( + "reading the casts of %s for its threshold: %w", + model.FormatProposalID(proposalID), err) + } + + result, err := EvaluateThreshold( + step.Instance, spec.Threshold, ThresholdOrder(spec.Threshold), + voteCastPayloads(votes), nil) + if err != nil { + return ThresholdResult{}, err + } + + // A matched routing carries the predicate verbatim in its reason (the + // same courtesy T3 extends): the operator reading the routing record sees + // WHY an approved tally did not pass, not just where it went. + if result.Routing != RoutingPass && !result.Parked { + result.Reason = fmt.Sprintf( + "approved, and threshold %q matched: %s", + result.Routing, spec.Threshold[result.Routing]) + } + return result, nil +} + +// voteCastPayloads renders recorded casts as threshold payloads — one element +// per cast, keyed by EXACTLY workflow.VoteCastFields, which is what V36 +// validates predicates against. A key here that the validator does not admit, +// or vice versa, is the drift both sides importing one list prevents. +func voteCastPayloads(votes []*model.Vote) []map[string]any { + out := make([]map[string]any, 0, len(votes)) + for _, v := range votes { + out = append(out, map[string]any{ + // `vote` and `verdict` are aliases for the same value — see the + // VoteCastField constants for why both spellings are admitted. + workflow.VoteCastFieldVote: string(v.Verdict), + workflow.VoteCastFieldVerdict: string(v.Verdict), + workflow.VoteCastFieldVoter: v.VoterName, + }) + } + return out +} + +// resolveVoteRecords is the `.vote-record` form (DKT-545): one input per +// recorded producer instance of the named vote step, carrying that instance's +// proposal record — tally outcome, score, and every cast with its rationale — +// as JSON. +// +// The instance selection mirrors resolveGateResults exactly — same issue, +// recorded producers only, ordinal-scoped with the per-input fallback, ordered +// by (sibling index, id) — because the question is the same one with a +// different ledger: "what did the producer record". The one departure the +// gate-results form makes (self-admission for pre-gates) is NOT made here: a +// vote step is never claimed, so no step ever reads its own vote-record at +// claim time. +// +// A recorded vote step WITHOUT a proposal — one an operator moved past with +// `docket step resolve` before its ballot opened — contributes no input: +// there is no record, and fabricating an empty one would report "a vote +// happened and nobody cast" about a vote that never opened. +func resolveVoteRecords( + tx *sql.Tx, sched *Scheduler, step *db.Step, stepName string, +) ([]ContextInput, error) { + var candidates []*db.Step + best := -1 + for _, s := range sched.steps { + if s.IssueID != step.IssueID || s.StepName != stepName || + s.Kind != workflow.TypeVote { + continue + } + if s.Ordinal > step.Ordinal || !recordedProducer(s.Status) { + continue + } + if s.Ordinal > best { + best = s.Ordinal + } + candidates = append(candidates, s) + } + if best < 0 { + return nil, nil + } + + producers := make([]*db.Step, 0, len(candidates)) + for _, s := range candidates { + if s.Ordinal == best { + producers = append(producers, s) + } + } + sort.SliceStable(producers, func(i, j int) bool { + si, sj := -1, -1 + if producers[i].SiblingIndex != nil { + si = *producers[i].SiblingIndex + } + if producers[j].SiblingIndex != nil { + sj = *producers[j].SiblingIndex + } + if si != sj { + return si < sj + } + return producers[i].ID < producers[j].ID + }) + + out := make([]ContextInput, 0, len(producers)) + for _, producer := range producers { + proposalID, found, err := db.LookupIdempotencyKeyTx( + tx, db.ScopeVoteCreate, + voteIdempotencyKey(producer.RunID, producer.IssueID, producer.Instance)) + if err != nil { + return nil, fmt.Errorf( + "resolving the proposal of %s: %w", producer.Instance, err) + } + if !found { + continue + } + proposal, err := db.GetProposalTx(tx, proposalID) + if err != nil { + return nil, fmt.Errorf( + "reading the proposal of %s: %w", producer.Instance, err) + } + votes, err := db.GetProposalVotesTx(tx, proposalID) + if err != nil { + return nil, fmt.Errorf( + "reading the casts of %s: %w", producer.Instance, err) + } + body, err := encodeVoteRecord(proposal, votes) + if err != nil { + return nil, fmt.Errorf( + "encoding the vote record of %s: %w", producer.Instance, err) + } + out = append(out, ContextInput{ + Artifact: workflow.VoteRecordKind, + Kind: workflow.VoteRecordKind, + ProducerStep: producer.Instance, + Body: body, + }) + } + return out, nil +} + +// encodeVoteRecord renders one proposal and its casts as the vote-record body. +// +// The vocabulary is the model's where the model has one (`verdict`, +// `weighted_score`) and the packet's where it does not (`rationale` is the +// cast's summary — the prose a seat wrote to explain itself, which is exactly +// what a downstream revise step consumes). Structured findings ride along when +// the seat recorded them; a cast without them omits the key rather than +// carrying an empty object. +func encodeVoteRecord(p *model.Proposal, votes []*model.Vote) (string, error) { + type cast struct { + Voter string `json:"voter"` + Role string `json:"role,omitempty"` + Verdict string `json:"verdict"` + Confidence float64 `json:"confidence"` + Rationale string `json:"rationale,omitempty"` + Findings *model.Findings `json:"findings,omitempty"` + } + type wire struct { + Proposal string `json:"proposal"` + Status string `json:"status"` + WeightedScore *float64 `json:"weighted_score,omitempty"` + Casts []cast `json:"casts"` + } + + encoded := wire{ + Proposal: model.FormatProposalID(p.ID), + Status: string(p.Status), + WeightedScore: p.WeightedScore, + Casts: make([]cast, 0, len(votes)), + } + for _, v := range votes { + encoded.Casts = append(encoded.Casts, cast{ + Voter: v.VoterName, Role: v.VoterRole, Verdict: string(v.Verdict), + Confidence: v.Confidence, Rationale: v.Summary, + Findings: v.FindingsJSON, + }) + } + + body, err := json.Marshal(encoded) + if err != nil { + return "", err + } + return string(body), nil +} diff --git a/internal/engine/waive.go b/internal/engine/waive.go new file mode 100644 index 00000000..547b107a --- /dev/null +++ b/internal/engine/waive.go @@ -0,0 +1,109 @@ +package engine + +import ( + "database/sql" + "errors" + "fmt" + "regexp" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// DKT-742 — the run-scoped stale-target waiver. +// +// The stale-target advisory (DKT-193/424/451) has no memory: an operator who +// investigated a warning and ruled it acceptable saw the IDENTICAL warning +// re-fire at every subsequent `dispatch open`/`verify` of the same pair — +// RUN-52 fired one adjudicated warning four times across three dispatches, +// each firing costing an investigation and (the first time) an operator gate, +// until the operator issued a standing waiver that lived only in session +// memory, where the engine could not see it. `dispatch waive-target` records +// that ruling ONCE, as one waiver per named step instance for one target sha, +// and staleTargets drops matching rows from every later advisory. +// +// THE WARNING MACHINERY STILL RUNS. What a waiver changes is the reporting of +// a (step, target) pair the operator already ruled on — the same pair on a +// DIFFERENT sha, or a different row on the same sha, warns exactly as before, +// which is what keeps a new divergence from sailing through under an old +// ruling. Run-scoped by the row's run_id, like gate_override_grants +// (DKT-546): a new run re-warns. + +// WaivedTarget names one recorded waiver, as the verb reports it. +type WaivedTarget struct { + ID int `json:"id"` + Instance string `json:"instance"` + Target string `json:"target_sha"` +} + +// waiverSHAPattern is what `--target` must look like: a hex sha, or an +// unambiguous prefix of one. The 7-character floor matches git's own +// short-sha convention; the advisory the operator copies from renders 12. +var waiverSHAPattern = regexp.MustCompile(`^[0-9a-fA-F]{7,64}$`) + +// WaiveStaleTargets records one waiver per named step instance, all for one +// target sha, in ONE transaction — the operator's ruling either covers every +// row they listed or none of them. Each waiver logs its own +// `stale-target-waived` event, so the feed shows exactly what standing +// precedent was minted and by which verb. +// +// The instances are the strings the advisory itself printed (`review@1#0`), +// matched verbatim by staleTargets — which is also why nothing here checks +// them against live steps: a fanout sibling minted by a later round is a NEW +// instance with a new signature, and the operator waives what warned, not +// what might. +func (e *Engine) WaiveStaleTargets( + conn *sql.DB, runID int, instances []string, targetSHA, note string, + nowMS int64, +) ([]WaivedTarget, error) { + if len(instances) == 0 { + return nil, validationErr("name at least one step instance to waive") + } + for _, instance := range instances { + if instance == "" { + return nil, validationErr("a step instance cannot be empty") + } + } + if !waiverSHAPattern.MatchString(targetSHA) { + return nil, validationErr( + "target sha %q is not a hex commit sha (or a prefix of at least "+ + "7 characters); pass the `target_sha` the stale-target "+ + "warning named", targetSHA) + } + + if _, err := db.GetRun(conn, runID); err != nil { + if errors.Is(err, db.ErrRunNotFound) { + return nil, notFoundErr(err, "run %s not found", model.FormatRunID(runID)) + } + return nil, err + } + + tx, err := conn.Begin() + if err != nil { + return nil, fmt.Errorf("recording the stale-target waiver(s): %w", err) + } + defer tx.Rollback() + + out := make([]WaivedTarget, 0, len(instances)) + for _, instance := range instances { + id, err := db.InsertStaleTargetWaiverTx(tx, db.StaleTargetWaiver{ + RunID: runID, StepInstance: instance, TargetSHA: targetSHA, + Note: note, CreatedAtMS: nowMS, + }) + if err != nil { + return nil, err + } + if err := recordEvent(tx, eventRecord{ + Kind: EventStaleTargetWaived, RunID: runID, Instance: instance, + Data: fmt.Sprintf("%s#%d", targetSHA, id), AtMS: nowMS, + }); err != nil { + return nil, err + } + out = append(out, WaivedTarget{ID: id, Instance: instance, Target: targetSHA}) + } + + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("recording the stale-target waiver(s): %w", err) + } + return out, nil +} diff --git a/internal/exec/env.go b/internal/exec/env.go index 43a761cf..6b8ca494 100644 --- a/internal/exec/env.go +++ b/internal/exec/env.go @@ -122,6 +122,16 @@ type EnvPolicy struct { // that declared no scope gives the check no narrower answer than the tree, // and inventing one would be docket deciding what the issue touches. Scope []string + // Base is the sha of the step's base commit, exported as DOCKET_GATE_BASE + // (DKT-992) so a gate can scan exactly the step's committed range + // (base..HEAD of the tree it runs in) instead of guessing at `HEAD~1` or + // scanning only the working tree — which, for a worktree-recorded step + // whose executor committed before `step record`, is clean and scans + // nothing. It is set only for worktree-recorded steps, to the worktree's + // fork point; empty means unset, and a range-shaped gate that finds the + // var absent over a clean tree should fail closed rather than pass having + // measured nothing. + Base string } // BuildEnv constructs the child environment (§5.3). @@ -169,6 +179,12 @@ func BuildEnv(p EnvPolicy) ([]string, error) { if len(p.Scope) > 0 { env = append(env, "DOCKET_SCOPE="+strings.Join(p.Scope, "\n")) } + // Absent — not empty — when no base is known (DKT-992): an empty sha is + // not a commit, and a gate given one would build a broken git range. The + // gate's own fail-closed check keys on absence. + if p.Base != "" { + env = append(env, "DOCKET_GATE_BASE="+p.Base) + } // The network half, reached ONLY by a gate whose trust entry // declared a requirement. A gate that declared none sees exactly the diff --git a/internal/exec/env_scope_test.go b/internal/exec/env_scope_test.go index c6247706..5d24d772 100644 --- a/internal/exec/env_scope_test.go +++ b/internal/exec/env_scope_test.go @@ -56,3 +56,34 @@ func TestBuildEnvOmitsScopeWhenUndeclared(t *testing.T) { t.Error("DOCKET_ISSUE is set with no issue in the policy") } } + +// DOCKET_GATE_BASE (DKT-992): a gate learns the step's base commit, so a +// range-shaped check can scan exactly the step's committed change +// (base..HEAD of the tree it runs in) instead of guessing `HEAD~1` or +// scanning a working tree the executor already committed to. + +func TestBuildEnvCarriesGateBase(t *testing.T) { + const base = "3f786850e387550fdab836ed7e6dc881de23001b" + env, err := BuildEnv(EnvPolicy{Gate: "secret-scan", Repo: "/repo", Base: base}) + if err != nil { + t.Fatalf("BuildEnv: %v", err) + } + if got, ok := envValue(env, "DOCKET_GATE_BASE"); !ok || got != base { + t.Errorf("DOCKET_GATE_BASE = %q, %v; want the step's base commit %q", + got, ok, base) + } +} + +// TestBuildEnvOmitsGateBaseWhenUnknown: absence — not emptiness — is the +// encoding, exactly as for DOCKET_SCOPE. An empty sha is not a commit, and a +// gate finding the var absent over a clean tree is meant to fail closed +// rather than build a broken range from "". +func TestBuildEnvOmitsGateBaseWhenUnknown(t *testing.T) { + env, err := BuildEnv(EnvPolicy{Gate: "secret-scan", Repo: "/repo"}) + if err != nil { + t.Fatalf("BuildEnv: %v", err) + } + if _, ok := envValue(env, "DOCKET_GATE_BASE"); ok { + t.Error("DOCKET_GATE_BASE is set for a step with no known base commit") + } +} diff --git a/internal/model/model_test.go b/internal/model/model_test.go index eb378f4d..6895b600 100644 --- a/internal/model/model_test.go +++ b/internal/model/model_test.go @@ -292,6 +292,43 @@ func TestParseRelationType(t *testing.T) { } } +// TestParseRelationDirection pins the two-directional vocabulary DKT-547's +// `issue.linked..` input form addresses: canonical tokens read +// a relation from its source, inverse tokens from its target, and both +// spellings normalize hyphens like ParseRelationType. +func TestParseRelationDirection(t *testing.T) { + tests := []struct { + input string + want RelationType + inverse bool + wantErr bool + }{ + {"blocks", RelationBlocks, false, false}, + {"blocked_by", RelationBlocks, true, false}, + {"blocked-by", RelationBlocks, true, false}, + {"depends_on", RelationDependsOn, false, false}, + {"dependency_of", RelationDependsOn, true, false}, + {"dependency-of", RelationDependsOn, true, false}, + {"relates_to", RelationRelatesTo, false, false}, + {"duplicates", RelationDuplicates, false, false}, + {"duplicate_of", RelationDuplicates, true, false}, + {"specified_by", "", false, true}, + {"", "", false, true}, + } + for _, tt := range tests { + got, inverse, err := ParseRelationDirection(tt.input) + if (err != nil) != tt.wantErr { + t.Errorf("ParseRelationDirection(%q) error = %v, wantErr %v", + tt.input, err, tt.wantErr) + continue + } + if got != tt.want || inverse != tt.inverse { + t.Errorf("ParseRelationDirection(%q) = (%q, %v), want (%q, %v)", + tt.input, got, inverse, tt.want, tt.inverse) + } + } +} + func TestRelationTypeInverse(t *testing.T) { tests := []struct { rt RelationType diff --git a/internal/model/relation.go b/internal/model/relation.go index 649e50ad..e0e0b5bc 100644 --- a/internal/model/relation.go +++ b/internal/model/relation.go @@ -44,6 +44,46 @@ func ParseRelationType(input string) (RelationType, error) { return normalized, nil } +// RelationDirectionTokens lists every token ParseRelationDirection accepts, in +// its canonical underscored form — the four relation types and their inverse +// display names. It exists so a consumer refusing a token can name the whole +// vocabulary rather than restate it (DKT-547). +func RelationDirectionTokens() []string { + out := make([]string, 0, 2*len(validRelationTypes)) + for _, rt := range validRelationTypes { + out = append(out, string(rt)) + if inv := rt.Inverse(); inv != string(rt) { + out = append(out, inv) + } + } + return out +} + +// ParseRelationDirection resolves a relation token that may name EITHER +// direction of a relation: a canonical type ("depends_on") or an inverse +// display name ("dependency_of", "blocked_by", "duplicate_of"). It returns the +// canonical type and whether the token named the inverse direction — for +// "depends_on" the subject is the relation's SOURCE, for "dependency_of" its +// TARGET (DKT-547: the `issue.linked..` input form addresses +// linked issues by exactly these tokens). +// +// Hyphenated spellings normalize like ParseRelationType's ("depends-on", +// "blocked-by"). The symmetric "relates_to" is its own inverse and parses as +// the canonical direction; its consumers treat both directions alike. +func ParseRelationDirection(input string) (rt RelationType, inverse bool, err error) { + normalized := strings.ReplaceAll(strings.TrimSpace(input), "-", "_") + if rt, err := ParseRelationType(normalized); err == nil { + return rt, false, nil + } + for _, rt := range validRelationTypes { + if rt.Inverse() == normalized { + return rt, true, nil + } + } + return "", false, fmt.Errorf( + "invalid relation %q: must be one of %v", input, RelationDirectionTokens()) +} + // Inverse returns the display name for the inverse direction of a relation. // For example, "blocks" returns "blocked_by" and "depends_on" returns "dependency_of". // Symmetric relations ("relates_to") return themselves. diff --git a/internal/model/workflow.go b/internal/model/workflow.go index 05d52539..13bee6c3 100644 --- a/internal/model/workflow.go +++ b/internal/model/workflow.go @@ -36,6 +36,24 @@ type Workflow struct { Parsed string CreatedAtMS int64 RowVersion int + // SourceStatus is the DISK VERDICT on SourcePath: whether the bytes at + // that path still hash to SourceSHA256 (DKT-590). + // + // It is NOT a column. Nothing loading a workflow row populates it; a + // reader that chose to go and LOOK sets it (engine.CheckWorkflowSource), + // and it stays nil everywhere else — so a payload that never checked + // carries no key at all rather than a default that reads as an answer. + // That is the same "a field that is not a fact does not appear" rule the + // optional wire fields follow. + SourceStatus *WorkflowSourceStatus `json:"-"` + // Origin is the DISK VERDICT on this row's NAME: whether any file in the + // instance-config roots still declares it (DKT-609). + // + // Like SourceStatus it is NOT a column and is populated by a reader that + // chose to go and look (engine.WorkflowOriginIndex), staying nil + // everywhere else so a payload that never scanned carries no key rather + // than a default that reads as an answer. + Origin *WorkflowOriginStatus `json:"-"` // DeprecatedAtMS is when this version was RETIRED FROM BINDING, or 0 when // it still binds. // @@ -49,6 +67,112 @@ type Workflow struct { // Deprecated reports whether this version has been retired from binding. func (w *Workflow) Deprecated() bool { return w.DeprecatedAtMS > 0 } +// WorkflowSourceState is the verdict on a registered workflow's recorded source +// file (DKT-590). The four values are DIFFERENT FACTS and are never collapsed: +// an operator acts on each of them differently. +type WorkflowSourceState string + +const ( + // WorkflowSourceMatches: the file at source_path hashes to source_sha256. + // What is registered is what is on disk. + WorkflowSourceMatches WorkflowSourceState = "matches" + // WorkflowSourceDrifted: the file is readable and hashes to something + // ELSE. The definition an operator edits at that path is not the + // definition a run binding this name@version would execute — the silent + // divergence DKT-590 is about. + WorkflowSourceDrifted WorkflowSourceState = "drifted" + // WorkflowSourceUnreadable: the file could not be read at all — deleted, + // replaced by a directory, or permission-denied. NOT the same as drift: + // the registered bytes are intact and still reproduce, only their + // provenance no longer resolves. + WorkflowSourceUnreadable WorkflowSourceState = "unreadable" + // WorkflowSourceUnchecked: no comparison was attempted, because the + // recorded path cannot answer the question from here — nothing was + // recorded (a stdin registration), or what was recorded is RELATIVE and + // would resolve against whatever directory happens to be current. Saying + // "unchecked" is the honest answer; resolving a relative provenance path + // against an unrelated cwd would manufacture drift out of a different + // file that happens to share a name. + WorkflowSourceUnchecked WorkflowSourceState = "unchecked" +) + +// WorkflowSourceStatus is one verdict on one registered workflow's source file. +type WorkflowSourceStatus struct { + State WorkflowSourceState `json:"state"` + // Path is the recorded source_path, verbatim — the value compared + // against, so a reader can tell WHICH file was (or was not) read. + Path string `json:"path,omitempty"` + RegisteredSHA256 string `json:"registered_sha256"` + // CurrentSHA256 is the hash of the bytes found at Path, present only when + // the file was actually read. + CurrentSHA256 string `json:"current_sha256,omitempty"` + // Reason explains an `unreadable` or `unchecked` verdict. The two other + // states are self-explaining and carry none. + Reason string `json:"reason,omitempty"` +} + +// Drifted reports the one state that means the registry and the disk disagree +// about bytes that both exist. +func (s *WorkflowSourceStatus) Drifted() bool { + return s != nil && s.State == WorkflowSourceDrifted +} + +// WorkflowOriginState is the verdict on whether a registered workflow's NAME +// still has a source file in the instance-config roots (DKT-609). +// +// It is a SIBLING of WorkflowSourceState and answers a different question. The +// source check asks "do the bytes at THIS row's recorded path still hash to +// what was registered" — the drift case, where one file changed. This asks "is +// there any file, anywhere in the configured roots, that still declares this +// NAME" — the rename case, where the old file is gone from disk entirely and +// its recorded path answers for nothing. +// +// THE VERDICT IS PER NAME, NOT PER VERSION, and deliberately so. A bumped +// definition leaves every superseded version's recorded path holding the new +// version's bytes, and calling all of those orphans would light up the entire +// version history of every workflow in the corpus. A name whose file was +// renamed away has no file declaring it at ANY version, which is exactly the +// state that wedged RUN-45. +type WorkflowOriginState string + +const ( + // WorkflowOriginPresent: some file in some instance-config root declares + // this workflow's name. A rename did not strand it. + WorkflowOriginPresent WorkflowOriginState = "present" + // WorkflowOriginOrphaned: the roots were scanned and NO file in any of + // them declares this name. The registration outlives its definition — the + // residue of a rename or a deleted file — and it keeps binding, because a + // registration is a row and not a file. + WorkflowOriginOrphaned WorkflowOriginState = "orphaned" + // WorkflowOriginUnchecked: nothing was scanned, so the question was not + // answered. No instance-config root exists on this machine, or the scan + // itself failed. Reporting "orphaned" from an unscanned root would call + // every registration in the store an orphan on the strength of having + // looked nowhere. + WorkflowOriginUnchecked WorkflowOriginState = "unchecked" +) + +// WorkflowOriginStatus is one verdict on one registered workflow NAME. +type WorkflowOriginStatus struct { + State WorkflowOriginState `json:"state"` + // Roots are the instance-config roots that were scanned, in precedence + // order, so a reader can tell WHERE the name was looked for. + Roots []string `json:"roots,omitempty"` + // Path is the file that declares this name, present only on a `present` + // verdict. It is the file found by scanning FRESH, which is why it can + // differ from the row's recorded source_path. + Path string `json:"path,omitempty"` + // Reason explains an `unchecked` verdict. The two others are + // self-explaining and carry none. + Reason string `json:"reason,omitempty"` +} + +// Orphaned reports the one state that means a registered name has outlived +// every file that declared it. +func (s *WorkflowOriginStatus) Orphaned() bool { + return s != nil && s.State == WorkflowOriginOrphaned +} + // Ref renders the `name@version` identity a run pins. func (w *Workflow) Ref() string { return fmt.Sprintf("%s@%d", w.Name, w.Version) @@ -92,6 +216,17 @@ type workflowJSON struct { SourceSHA256 string `json:"source_sha256"` CreatedAtMS int64 `json:"created_at_ms"` Definition json.RawMessage `json:"definition,omitempty"` + // SourceStatus reaches v1 WHEN A READER CHECKED (DKT-590), the same + // omitempty shape `scope` and `resolution` took onto `issue show`: a + // payload from a reader that did not look is byte-identical to the frozen + // v1 output, while `workflow show` stops reporting a source_path and a + // source_sha256 that silently disagree with the bytes at that path. + SourceStatus *WorkflowSourceStatus `json:"source_status,omitempty"` + // Origin reaches v1 on the same terms and for the same reason (DKT-609): + // omitted entirely by a reader that did not scan the instance-config + // roots, so the frozen v1 payload is byte-identical everywhere the + // question was never asked. + Origin *WorkflowOriginStatus `json:"origin,omitempty"` } // MarshalJSON renders the v1 shape: the registration's identity and provenance @@ -110,6 +245,8 @@ func (w *Workflow) wire() workflowJSON { SourcePath: w.SourcePath, SourceSHA256: w.SourceSHA256, CreatedAtMS: w.CreatedAtMS, + SourceStatus: w.SourceStatus, + Origin: w.Origin, } if json.Valid([]byte(w.Parsed)) { out.Definition = json.RawMessage(w.Parsed) @@ -134,7 +271,11 @@ type workflowVersionedJSON struct { SourceSHA256 string `json:"source_sha256"` CreatedAtMS int64 `json:"created_at_ms"` Definition json.RawMessage `json:"definition,omitempty"` - RowVersion int `json:"row_version"` + // SourceStatus is v1's field carried forward unchanged; see workflowJSON. + SourceStatus *WorkflowSourceStatus `json:"source_status,omitempty"` + // Origin is likewise v1's field carried forward unchanged. + Origin *WorkflowOriginStatus `json:"origin,omitempty"` + RowVersion int `json:"row_version"` // DeprecatedAtMS marks a version retired from binding (DKT-82): without // it, binding eligibility was unauditable from list output — all // registered versions rendered alike. Omitted while the version still @@ -174,6 +315,8 @@ func (w *Workflow) VersionedPayload() any { SourceSHA256: base.SourceSHA256, CreatedAtMS: base.CreatedAtMS, Definition: base.Definition, + SourceStatus: base.SourceStatus, + Origin: base.Origin, RowVersion: w.RowVersion, DeprecatedAtMS: w.DeprecatedAtMS, } diff --git a/internal/output/human.go b/internal/output/human.go index 942abcfa..6b62685b 100644 --- a/internal/output/human.go +++ b/internal/output/human.go @@ -29,6 +29,30 @@ func writeHumanSuccess(w io.Writer, message string) { } } +// writeHumanOutcome writes a human-readable adverse outcome to w: the same +// line writeHumanSuccess would write, with the failure glyph in place of the +// checkmark. +// +// It is NOT writeHumanError: there is no "Error:" label and nothing goes to +// stderr, because the command did not fail — its subject did. The glyph is the +// whole of the difference, and the message must therefore say what happened on +// its own, since a NO_COLOR terminal gets no glyph at all. +func writeHumanOutcome(w io.Writer, message string) { + if message == "" { + return + } + if strings.Contains(message, "\n") { + fmt.Fprintln(w, message) + return + } + if render.ColorsEnabled() { + icon := lipgloss.NewStyle().Foreground(lipgloss.Color("1")).Bold(true).Render("\u2718") + fmt.Fprintf(w, "%s %s\n", icon, message) + } else { + fmt.Fprintln(w, message) + } +} + // writeHumanError writes a human-readable error message to w. func writeHumanError(w io.Writer, err error) { if render.ColorsEnabled() { diff --git a/internal/output/output.go b/internal/output/output.go index 669caebb..c5404262 100644 --- a/internal/output/output.go +++ b/internal/output/output.go @@ -68,6 +68,29 @@ func (w *Writer) Success(data any, message string) { writeHumanSuccess(w.Stdout, message) } +// Outcome renders a result whose COMMAND succeeded but whose SUBJECT did not. +// +// The envelope, the payload, and the exit code are Success's exactly — the verb +// did what it was asked, and a caller branching on `$?` or on `.ok` must not be +// told otherwise. What changes is the one thing a human reads first: the line +// carries the failure glyph instead of the checkmark. +// +// It exists because `step record` had no way to say "recorded, and the gate +// failed" (DKT-982). A gate failure parks the step, and the park was printed as +// `✔ Completed STEP-N (waiting-human)` — a success glyph on a failed outcome, +// which an executor read as a pass and reported up as one. +func (w *Writer) Outcome(data any, message string) { + if w.JSONMode { + if w.JSONVersion == JSONV2 { + writeJSONSuccessV2(w.Stdout, data, message) + } else { + writeJSONSuccess(w.Stdout, data, message) + } + return + } + writeHumanOutcome(w.Stdout, message) +} + // Error renders an error. In JSON mode the error is wrapped in an error // envelope written to Stdout. In human mode the error is printed to Stderr // with an "Error: " prefix. The corresponding exit code is returned so the diff --git a/internal/output/output_test.go b/internal/output/output_test.go index 8bead1f8..b8fbf016 100644 --- a/internal/output/output_test.go +++ b/internal/output/output_test.go @@ -4,6 +4,7 @@ import ( "bytes" "encoding/json" "errors" + "os" "testing" "github.com/ALT-F4-LLC/docket/internal/render" @@ -363,6 +364,80 @@ func TestWriterWarnEmitsInHumanMode(t *testing.T) { } } +// enableColors makes render.ColorsEnabled() true for the duration of one test. +// +// t.Setenv cannot express it on its own: ColorsEnabled uses LookupEnv, so +// NO_COLOR="" still disables colors. The variable has to be genuinely absent. +func enableColors(t *testing.T) { + t.Helper() + if prev, ok := os.LookupEnv("NO_COLOR"); ok { + testsupport.Must(t, os.Unsetenv("NO_COLOR"), "unsetting NO_COLOR: %v", nil) + t.Cleanup(func() { _ = os.Setenv("NO_COLOR", prev) }) + } + t.Setenv("TERM", "xterm-256color") + if !render.ColorsEnabled() { + t.Fatal("premise: colors are still disabled, so the glyph under test " + + "would not be emitted either way") + } +} + +// TestWriterOutcomeSwapsTheCheckmarkForTheFailureGlyph is DKT-982's second +// acceptance criterion at the writer: a parked outcome must not be presented +// behind a "✔". +// +// Success is exercised in the same test over the same message, because the +// claim is a DIFFERENCE between the two paths — asserting only that Outcome +// prints ✘ would still pass if Success had quietly stopped printing ✔. +func TestWriterOutcomeSwapsTheCheckmarkForTheFailureGlyph(t *testing.T) { + enableColors(t) + + const message = "gate self-hygiene failed (exit 2); STEP-3107 parked waiting-human" + + var adverse, stderr bytes.Buffer + (&Writer{Stdout: &adverse, Stderr: &stderr}).Outcome(nil, message) + + if bytes.Contains(adverse.Bytes(), []byte("✔")) { + t.Errorf("Outcome = %q, and it carries the success checkmark — a park "+ + "behind a ✔ is the whole of DKT-982", adverse.String()) + } + if !bytes.Contains(adverse.Bytes(), []byte("✘")) { + t.Errorf("Outcome = %q, want the failure glyph", adverse.String()) + } + if !bytes.Contains(adverse.Bytes(), []byte(message)) { + t.Errorf("Outcome = %q, want it to carry the message", adverse.String()) + } + if stderr.Len() != 0 { + t.Errorf("stderr = %q; the command succeeded, so nothing goes there", + stderr.String()) + } + + var ok bytes.Buffer + (&Writer{Stdout: &ok, Stderr: &bytes.Buffer{}}).Success(nil, message) + if !bytes.Contains(ok.Bytes(), []byte("✔")) { + t.Errorf("Success = %q, want the checkmark it has always printed", ok.String()) + } +} + +// TestWriterOutcomeJSONIsASuccessEnvelope pins the half that must NOT change: +// the recording succeeded, so a caller branching on `.ok` or on the exit code +// is told exactly what Success would tell it. Only the human line differs. +func TestWriterOutcomeJSONIsASuccessEnvelope(t *testing.T) { + var stdout, stderr bytes.Buffer + w := &Writer{JSONMode: true, Stdout: &stdout, Stderr: &stderr} + + w.Outcome(map[string]string{"status": "waiting-human"}, "gate build failed (exit 2)") + + var env successEnvelope + err := json.Unmarshal(stdout.Bytes(), &env) + testsupport.Must(t, err, "unmarshal: %v", err) + if !env.OK { + t.Error("ok = false; the recording succeeded — its subject did not") + } + if env.Message != "gate build failed (exit 2)" { + t.Errorf("message = %q, want the outcome line", env.Message) + } +} + func TestColorsEnabledRespectsNoColor(t *testing.T) { t.Setenv("NO_COLOR", "1") diff --git a/internal/render/step_test.go b/internal/render/step_test.go new file mode 100644 index 00000000..6ff62352 --- /dev/null +++ b/internal/render/step_test.go @@ -0,0 +1,39 @@ +package render + +import "testing" + +// DKT-862 — `step show` was ALREADY RIGHT, and had to stay byte-for-byte right. +// +// It was the only one of the three gate surfaces carrying the pre marker: `run +// report`'s Gates tally and `events list`'s gate-recorded lines rendered an +// advisory pre-gate failure identically to a blocking one, and the remedy was +// to teach those two what this function already knew. The risk in that remedy +// is a "shared helper" refactor that quietly restyles the surface that was +// working, so this pins the exact bytes. +// +// A GOLDEN STRING, not a set of Contains checks. The complaint DKT-862 fixes +// was about how two surfaces LOOKED beside a third; a test that accepted any +// output containing "[pre]" would not notice this one drifting away from the +// spelling the other two were just aligned to. +func TestStepShowGateSummaryBytesAreUnchanged(t *testing.T) { + exitTwo := 2 + rows := []StepGateRow{ + // RUN-61 STEP-2745: the advisory failure that routed nothing. + {Gate: "ac-commands", Verdict: "fail", Exit: &exitTwo, Pre: true}, + // A blocking gate failing the same way, and a re-run beside it, so the + // ordinal and the marker are pinned together — they compose into one + // name and a refactor could reorder them. + {Gate: "build", Verdict: "fail", Exit: &exitTwo}, + {Gate: "build", Ordinal: 1, Verdict: "pass", Exit: new(int)}, + } + + const want = " gates:\n" + + " fail ac-commands [pre] exit 2\n" + + " fail build exit 2\n" + + " pass build (re-run 1) exit 0\n" + + " 2 gate(s) did not pass; reasons and output: docket step gates STEP-2745\n" + + if got := RenderStepGateSummary("STEP-2745", rows); got != want { + t.Errorf("`step show`'s gate summary changed.\ngot:\n%s\nwant:\n%s", got, want) + } +} diff --git a/internal/trust/add.go b/internal/trust/add.go index ca7523f8..5b061122 100644 --- a/internal/trust/add.go +++ b/internal/trust/add.go @@ -29,8 +29,12 @@ type AddRequest struct { // Stub declares the argv a placeholder rather than the check its name // implies (DKT-265). It changes no execution behavior; it travels with the // verdict so a hollow pass reads as hollow. - Stub bool - Timeout string + Stub bool + // StubReason says why the entry is a stub and which issue tracks replacing + // it (DKT-607). Only meaningful alongside Stub; an add that supplies a + // reason without the declaration is refused. + StubReason string + Timeout string // Network is the host list this command must reach. It declares // a requirement; it grants nothing. Network []string @@ -230,6 +234,12 @@ func buildEntry(req AddRequest) (Entry, []string, error) { if len(req.Argv) == 0 { return Entry{}, nil, fmt.Errorf("%w: a trust entry needs a command; pass it after `--`", ErrParse) } + // The same contradiction parse refuses in a hand-edited file (DKT-607), + // refused at the door: a reason describes a stub, and an add carrying one + // without the declaration meant one of the two flags is a mistake. + if req.StubReason != "" && !req.Stub { + return Entry{}, nil, fmt.Errorf("%w: --stub-reason describes a stub entry; pass --stub with it or drop the reason", ErrParse) + } e := Entry{ Name: req.Name, @@ -241,6 +251,7 @@ func buildEntry(req AddRequest) (Entry, []string, error) { Tree: req.Tree, Flaky: req.Flaky, Stub: req.Stub, + StubReason: req.StubReason, Timeout: req.Timeout, Network: req.Network, AddedAtMS: req.NowMS, @@ -346,6 +357,13 @@ func entryChanges(existing, proposed Entry) []string { changes = append(changes, fmt.Sprintf("%s %t to %t", f.name, f.old, f.new)) } } + // `stub_reason` joins for the same reason `stub` did: the reason is the + // documented decision (DKT-607) — why this assurance is hollow and which + // issue tracks fixing it — and a re-add that silently rewrote or erased it + // would swap one documented decision for another with no trace. + if existing.StubReason != proposed.StubReason { + changes = append(changes, fmt.Sprintf("stub_reason %q to %q", existing.StubReason, proposed.StubReason)) + } if !slices.Equal(existing.Network, proposed.Network) { // Order-sensitive, because the stored list is what an operator reads and // a reorder is still an edit to the file they audit. The remedy the diff --git a/internal/trust/parse.go b/internal/trust/parse.go index fd179ba5..ade45f08 100644 --- a/internal/trust/parse.go +++ b/internal/trust/parse.go @@ -84,6 +84,14 @@ func validateEntry(e Entry, idx int, path string) error { return fmt.Errorf("%w: %s in %s sets both global = true and repo = %q; an entry binds to one repo or to all, never both", ErrParse, where, path, e.Repo) } + // A stub_reason on a non-stub entry is contradictory (DKT-607). Refusing is + // the closed direction, same as global+repo above: honoring the reason + // would imply the entry is a stub the flag denies, and dropping it would + // silently discard a key the operator wrote. + if e.StubReason != "" && !e.Stub { + return fmt.Errorf("%w: %s in %s has a stub_reason but stub is not true; a reason describes a stub, so set stub = true or remove stub_reason", ErrParse, where, path) + } + // The stored hash must describe the stored argv. This catches a corrupted // or hand-edited file (M3): an operator who edits `argv` without recomputing // the hash gets a refusal rather than an entry whose two halves disagree. diff --git a/internal/trust/store.go b/internal/trust/store.go index 0b0fb8dd..43ad1ffc 100644 --- a/internal/trust/store.go +++ b/internal/trust/store.go @@ -78,6 +78,23 @@ type Entry struct { // recorded exactly as any other; the flag only travels with the verdict so // that hollow green stays visibly hollow. Stub bool `toml:"stub"` + // StubReason says WHY the entry is a stub and which issue tracks replacing + // it with the real check (DKT-607). Free text, but the convention the + // stub-gate policy asks for is a reason plus a tracking reference, e.g. + // "no scanner selected yet; removal tracked by DKT-607". + // + // WHY IT EXISTS. A `stub = true` alone says the assurance is hollow; it + // does not say whether that is a deliberate, tracked decision or a + // placeholder somebody forgot. Tribunal panels kept rediscovering the same + // stubs run after run because the decision lived only in transcripts. The + // reason travels with the entry so gate_preflight and `trust list` can + // surface it where the stub itself is surfaced. + // + // It only makes sense alongside Stub: a reason on a non-stub entry is a + // malformed file and is refused at parse (and at add). It is OPTIONAL on a + // stub — every pre-DKT-607 stub entry has none, and an empty reason simply + // renders as unexplained. + StubReason string `toml:"stub_reason,omitempty"` // Timeout is a per-entry override of the default, as a duration string. Timeout string `toml:"timeout"` // Network declares the hosts this command must reach, as bare hostnames. diff --git a/internal/trust/store_test.go b/internal/trust/store_test.go index 6ee0b0a1..f9d14baf 100644 --- a/internal/trust/store_test.go +++ b/internal/trust/store_test.go @@ -389,6 +389,21 @@ repo = "/src/example" `, wantErr: "empty argv", }, + { + // DKT-607: a stub_reason on a non-stub entry is contradictory. + // Refusing is the closed direction: honoring the reason would imply + // the entry is a stub the flag denies, and dropping it would + // silently discard a key the operator wrote. + name: "stub_reason without stub", + content: `version = 1 +[[entry]] +name = "secret-scan" +argv = ["/usr/bin/true"] +repo = "/src/example" +stub_reason = "tracked by DKT-607" +`, + wantErr: "stub_reason", + }, { // A hand-edited argv whose stored hash no longer describes it is // refused rather than obeyed: the two halves disagree, and @@ -446,13 +461,20 @@ name = "fmt" argv = ["gofmt", "-l", "."] global = true prefix = true + +[[entry]] +name = "secret-scan" +argv = ["/usr/bin/true"] +global = true +stub = true +stub_reason = "no scanner selected yet; removal tracked by DKT-607" ` writeStoreFile(t, path, content, storeFileMode) st, err := loadAt(path) testsupport.Must(t, err, "a well-formed store must load: %v", err) - if len(st.Entries) != 2 { - t.Fatalf("expected 2 entries, got %d", len(st.Entries)) + if len(st.Entries) != 3 { + t.Fatalf("expected 3 entries, got %d", len(st.Entries)) } first := st.Entries[0] @@ -463,6 +485,30 @@ prefix = true if !second.Global || !second.Prefix || second.Repo != "" { t.Errorf("global entry did not round-trip: %+v", second) } + third := st.Entries[2] + if !third.Stub || third.StubReason != "no scanner selected yet; removal tracked by DKT-607" { + t.Errorf("stub entry did not round-trip its reason (DKT-607): %+v", third) + } +} + +// TestStubEntryWithoutAReasonStillLoads pins DKT-607's back-compat: every +// pre-DKT-607 stub entry has no stub_reason, and such a file keeps loading with +// an empty reason — the field is optional on a stub, mandatory-absent off one. +func TestStubEntryWithoutAReasonStillLoads(t *testing.T) { + path := sandbox(t) + writeStoreFile(t, path, `version = 1 +[[entry]] +name = "secret-scan" +argv = ["/usr/bin/true"] +global = true +stub = true +`, storeFileMode) + + st, err := loadAt(path) + testsupport.Must(t, err, "a reasonless stub entry must keep loading: %v", err) + if len(st.Entries) != 1 || !st.Entries[0].Stub || st.Entries[0].StubReason != "" { + t.Errorf("got %+v, want one stub entry with an empty reason", st.Entries) + } } // --- The canonical-argv hash (§3.3) ----------------------------------------- diff --git a/internal/workflow/aggregate.go b/internal/workflow/aggregate.go index 68ea927d..8a5fbf16 100644 --- a/internal/workflow/aggregate.go +++ b/internal/workflow/aggregate.go @@ -34,9 +34,9 @@ var BuiltinActions = []string{ActionAggregate} // operator's command under a name core intends to take. var ReservedActions = []string{ActionAggregate} -// AggregateParamKeys are the four keys §7.1's table declares, in its own order. +// AggregateParamKeys are the five keys §7.1's table declares, in its own order. // V28's "no other keys" is checked against exactly this list. -var AggregateParamKeys = []string{"field", "method", "hold_spread", "output"} +var AggregateParamKeys = []string{"field", "method", "hold_spread", "output", "route_at"} // AggregateMethods is the closed reduction vocabulary, in §2's order. var AggregateMethods = []string{"median", "max", "min"} @@ -104,7 +104,8 @@ func validateAggregateInputs(step *Step) error { } // validateAggregateParams is V28: `aggregate` requires `field`, `method`, and -// `output`; `hold_spread` is an integer >= 0 when present; and NO OTHER KEYS. +// `output`; `hold_spread` is an integer >= 0 when present; `route_at` is a +// non-empty string when present; and NO OTHER KEYS. // // The discipline is every V-rule's. A typo'd `method = "medain"` is otherwise // discovered hours into a run, on a step whose inputs are already spent — and an @@ -146,7 +147,7 @@ func validateAggregateParams(step *Step) error { } // `output` is already V11's — an action step must declare it — and is - // restated here so an `aggregate` step's four params are checked as one + // restated here so an `aggregate` step's params are checked as one // table rather than two halves an author has to assemble. if out, _ := step.Params["output"].(string); out == "" { return fail("params", @@ -160,6 +161,19 @@ func validateAggregateParams(step *Step) error { "`params.hold_spread` must be an integer >= 0, got %v", raw) } } + + // `route_at` is optional; when present it must be a non-empty string. WHICH + // strings it may name is the declared order's business, and the order lives + // in the step's registered schema — so the membership check is V28a's, in + // ValidateSchemas, beside V29's order check. This clause is pure bytes and + // stays here with V27, V28, and V31. + if raw, present := step.Params["route_at"]; present { + if s, ok := raw.(string); !ok || s == "" { + return fail("params", + "`params.route_at` must be a non-empty string naming a value of "+ + "`params.field`'s declared order, got %v", raw) + } + } return nil } diff --git a/internal/workflow/expand.go b/internal/workflow/expand.go index 95e821bb..9e58606c 100644 --- a/internal/workflow/expand.go +++ b/internal/workflow/expand.go @@ -97,27 +97,36 @@ func Expand(def *Definition, s Subject, ordinal int) []StepInstance { // // Two sets of steps instantiate at ordinal k, and only these two: // -// - clause (3): `loop = true` steps, which ordinary expansion excludes -// ("excluded from ordinary expansion ... instantiate at ordinal k"); -// - clause (4): the `after_loop` step AND ITS DOWNSTREAM CHAIN +// - clause (3): the SERVING `loop = true` steps, supplied as `bodies` — +// the triggering step's cluster (LoopBodiesFor), which is every body when +// no `serves` is declared anywhere; +// - clause (4): the cluster's `after_loop` roots AND THEIR DOWNSTREAM CHAIN // ("`after_loop` and its downstream chain re-instantiate at ordinal k"), // supplied as `downstream` because the chain is a property of the // definition's shape that the caller already computed for the sweep. // +// Both sets are the CALLER's, computed for the entry's triggering step, so +// the instantiation and the supersede sweep read one answer to "what does +// this cluster re-run" rather than two that can drift. +// // EVERYTHING ELSE IS LEFT AT ITS EXISTING ORDINAL. `implement` is upstream of // `after_loop`: it does not re-run, its artifact is not reproduced, and §7.4's // per-input fallback exists precisely so a step at ordinal k can still bind it. +// A body OUTSIDE the triggering cluster is likewise left alone — its rounds are +// its own triggers' to mint (§11.3 cluster scoping, DKT-544). // // Sharing Expand's per-step body is the point. The status/class/fanout/metadata // rules are applied by one function for both ordinals, so ordinal 1's topology // cannot drift from ordinal 0's — a second implementation of "one row per hint, // in declared order" is exactly how it would. -func ExpandOrdinal(def *Definition, s Subject, ordinal int, downstream map[string]bool) []StepInstance { +func ExpandOrdinal( + def *Definition, s Subject, ordinal int, bodies, downstream map[string]bool, +) []StepInstance { out := make([]StepInstance, 0, len(def.Steps)) for _, step := range def.Steps { // Clause (3) or clause (4); a step in neither set does not re-instantiate. - if !step.Loop && !downstream[step.Name] { + if !(step.Loop && bodies[step.Name]) && !downstream[step.Name] { continue } out = append(out, expandStep(step, s, ordinal)...) @@ -126,6 +135,26 @@ func ExpandOrdinal(def *Definition, s Subject, ordinal int, downstream map[strin return out } +// ServesTrigger reports whether a step's loop-cluster declaration covers a +// triggering step name. An empty `serves` covers EVERY trigger — the +// backward-compatible reading under which a workflow with no `serves` +// anywhere has one cluster spanning every body and root (§11.3). +func ServesTrigger(step *Step, trigger string) bool { + return len(step.Serves) == 0 || slices.Contains(step.Serves, trigger) +} + +// LoopBodiesFor is the set of `loop = true` step names serving one trigger — +// §11.3 clause (3) scoped to the step whose routing entered the loop. +func LoopBodiesFor(def *Definition, trigger string) map[string]bool { + out := make(map[string]bool) + for _, step := range def.Steps { + if step.Loop && ServesTrigger(step, trigger) { + out[step.Name] = true + } + } + return out +} + // ExpandStepAt renders ONE named step's instance rows at an ordinal — the same // rows ExpandOrdinal would have produced for it, and nothing else. // diff --git a/internal/workflow/held.go b/internal/workflow/held.go index 923acec2..4e1b4d07 100644 --- a/internal/workflow/held.go +++ b/internal/workflow/held.go @@ -30,6 +30,15 @@ const HeldSuffix = "-held" // shadow it — which V11a refuses at register. const GateResultsKind = "gate-results" +// VoteRecordKind is the reserved `.vote-record` input suffix (DKT-545): +// the named vote step's RECORDED proposal — tally outcome, casts, and +// rationales — engine-served from the existing proposal machinery. +// GateResultsKind's reasoning applied to the vote record: a definition may not +// emit an artifact of this kind (V11b), and the input form resolves only +// against `type="vote"` steps — any other step opens no proposal, so the +// input could never resolve to anything on any run (V11). +const VoteRecordKind = "vote-record" + // HeldStepName renders the materialized step's name for a routing step. func HeldStepName(step string) string { return step + HeldSuffix } diff --git a/internal/workflow/lint.go b/internal/workflow/lint.go index b89c3cce..fda358d6 100644 --- a/internal/workflow/lint.go +++ b/internal/workflow/lint.go @@ -246,6 +246,13 @@ func lintInputOrdering(def *Definition, g *stepGraph) error { if _, ok := LatestKind(input); ok { continue } + // `issue.linked..` (DKT-547) likewise names no + // producer step: its artifact was recorded under ANOTHER issue and + // pinned at activation, so it exists before any step of this + // workflow runs — there is no ordering for L4 to enforce. + if _, _, ok := LinkedInput(input); ok { + continue + } m := inputShape.FindStringSubmatch(input) if m == nil { continue // V11 already rejected it. diff --git a/internal/workflow/match.go b/internal/workflow/match.go index 8dba1db2..b2111c47 100644 --- a/internal/workflow/match.go +++ b/internal/workflow/match.go @@ -67,9 +67,20 @@ func (m *Match) Matches(s Subject) bool { // // An empty `when` holds: a step that declares no condition is unconditional. // The grammar is §11.1's, validated at register time by V22 and re-parsed here -// against the SAME regexp, so a predicate that registered cannot fail to -// evaluate. Conjuncts are joined by `and` — the only connective (splitWhen) — -// and every one must hold. +// against the SAME regexps — whenShape for a clause, whenConnective for the +// joins — so a predicate that registered cannot fail to evaluate. +// +// Clauses are joined by `and` (every one must hold) or by `or` (at least one +// must), never both in one predicate: V22 refuses a mix, because the grammar +// has no parentheses and therefore no way to say which reading of `a and b or c` +// the author meant (DKT-548). A mixed predicate reaching here predates the +// current grammar, and it evaluates FALSE for the same reason an unparseable +// clause does. +// +// "any of these labels" is a CLAUSE, not a connective: `labels contains-any +// (a, b)` (DKT-550), equivalently `labels contains_any [a, b]` (DKT-1000). +// That is what lets "kind X and any of these labels" stay a single +// homogeneous-`and` predicate instead of needing the mix V22 refuses. // // A false `when` does not omit the step: expansion creates it with status // `skipped` (§5.3.1), which is what keeps a downstream `after` resolvable and @@ -78,7 +89,22 @@ func WhenHolds(when string, s Subject) bool { if strings.TrimSpace(when) == "" { return true } - for _, clause := range splitWhen(when) { + + clauses, connective, mixed := splitWhen(when) + if mixed { + return false + } + + if connective == WhenOr { + for _, clause := range clauses { + if whenClauseHolds(clause, s) { + return true + } + } + return false + } + + for _, clause := range clauses { if !whenClauseHolds(clause, s) { return false } @@ -86,9 +112,12 @@ func WhenHolds(when string, s Subject) bool { return true } -// whenClauseHolds evaluates one ` <==|!=|contains> ` -// clause. An unparseable clause is FALSE rather than an error: V22 rejected it -// at register time, so reaching this branch means a stored definition predates +// whenClauseHolds evaluates one clause — either +// ` <==|!=|contains> ` or the set form +// `labels contains-any (a, b, c)` / `labels contains_any [a, b, c]`. +// +// An unparseable clause is FALSE rather than an error: V22 rejected it at +// register time, so reaching this branch means a stored definition predates // the current grammar, and skipping the step is the conservative reading — // creating it ready would run work whose condition nobody could evaluate. func whenClauseHolds(clause string, s Subject) bool { @@ -96,7 +125,24 @@ func whenClauseHolds(clause string, s Subject) bool { if m == nil { return false } - field, op, value := m[1], m[2], unquote(m[3]) + + // The set form — the same intersection test the workflow-level `labels_any` + // [match] clause runs (Matches), so the spellings of "any of these labels" + // cannot answer a subject differently. + // + // whenShape gives the list body two capture groups because RE2 cannot + // backreference a delimiter: m[3] is the `(…)` body, m[4] the `[…]` one, + // and exactly one of them is non-empty whenever m[1] matched — the grammar + // admits no empty list, so "non-empty" is the whole test. + if m[1] != "" { + body := m[3] + if body == "" { + body = m[4] + } + return intersects(whenList(body), s.Labels) + } + + field, op, value := m[5], m[6], unquote(m[7]) switch field { case "kind": @@ -123,6 +169,21 @@ func whenClauseHolds(clause string, s Subject) bool { return false } +// whenList splits the body of a `contains-any (…)` / `contains_any […]` list +// into its values. +// +// whenShape has already established the shape — at least one element, elements +// free of whitespace, parens and commas — so this only has to cut on the +// separator and unquote, exactly as the one-value form does. +func whenList(body string) []string { + parts := strings.Split(body, ",") + out := make([]string, 0, len(parts)) + for _, p := range parts { + out = append(out, unquote(p)) + } + return out +} + // unquote strips the optional quotes around a predicate literal, so // `kind == "bug"` and `kind == bug` mean the same thing. §11.1's examples use // both spellings. diff --git a/internal/workflow/match_test.go b/internal/workflow/match_test.go index 2d690234..4a307090 100644 --- a/internal/workflow/match_test.go +++ b/internal/workflow/match_test.go @@ -2,6 +2,7 @@ package workflow import ( "os" + "slices" "strings" "testing" @@ -160,10 +161,93 @@ func TestWhenHolds(t *testing.T) { {"kind == bug and labels contains backend", true}, {"kind == bug and labels contains docs", false}, {"kind == task and labels contains backend", false}, + // `contains-any` (DKT-550): the clause holds when the list and the + // issue's labels intersect. The empty-intersection row is the one that + // matters — a membership test that held vacuously would route every + // issue through the specialized lane it was written to select. + {"labels contains-any (backend)", true}, + {"labels contains-any (docs)", false}, + {"labels contains-any (docs, backend)", true}, + {"labels contains-any (backend, docs)", true}, + {"labels contains-any (docs, frontend)", false}, + {"labels contains-any (backend, urgent)", true}, + {`labels contains-any ("docs", "urgent")`, true}, + {"labels contains-any(backend,docs)", true}, + {"labels contains-any ( docs , urgent )", true}, + // One homogeneous-`and` predicate carrying a set test — DKT-550's whole + // point: this needs no `or`, so the mixing rule never applies to it. + {"kind == bug and labels contains-any (docs, backend)", true}, + {"kind == bug and labels contains-any (docs, frontend)", false}, + {"kind == task and labels contains-any (docs, backend)", false}, + {"labels contains-any (docs, backend) and labels != docs-only", true}, + // The set form composes with `or` too — it is a clause, not a + // connective, so it carries no opinion about which one joins it. + {"kind == task or labels contains-any (docs, urgent)", true}, + {"kind == task or labels contains-any (docs, frontend)", false}, + // `contains_any [...]` (DKT-1000): the same clause under the spelling + // and delimiters authors reach for, so every row above has a twin here. + // The two spellings are ONE operator — a subject they answered + // differently would mean the grammar has two set tests, not two ways to + // write one. + {"labels contains_any [backend]", true}, + {"labels contains_any [docs]", false}, + {"labels contains_any [docs, backend]", true}, + {"labels contains_any [backend, docs]", true}, + {"labels contains_any [docs, frontend]", false}, + {`labels contains_any ["docs", "urgent"]`, true}, + {`labels contains_any ["docs", "frontend"]`, false}, + {"labels contains_any[backend,docs]", true}, + {"labels contains_any [ docs , urgent ]", true}, + // The spelling and the delimiter are independent choices: all four + // combinations are the same clause. + {"labels contains_any (docs, backend)", true}, + {"labels contains-any [docs, backend]", true}, + {"labels contains_any (docs, frontend)", false}, + {"labels contains-any [docs, frontend]", false}, + // A list must be closed by the delimiter that opened it. RE2 cannot + // backreference, so this is the assertion that the two branches were + // written out rather than a delimiter class that pairs anything. + {"labels contains_any [docs, backend)", false}, + {"labels contains_any (docs, backend]", false}, + // DKT-1000's own example, verbatim. + {"labels contains_any [security-change, security-load-bearing, security]", false}, + // Composition with `and` and `or` is the point of the clause form — + // it must carry no opinion about which connective joins it. + {"kind == bug and labels contains_any [docs, backend]", true}, + {"kind == bug and labels contains_any [docs, frontend]", false}, + {"kind == task and labels contains_any [docs, backend]", false}, + {"labels contains_any [docs, backend] and labels != docs-only", true}, + {"labels contains_any [docs, backend] and labels contains urgent", true}, + {"kind == task or labels contains_any [docs, urgent]", true}, + {"kind == task or labels contains_any [docs, frontend]", false}, + // `kind` has no set form: it is one value, so membership is `==`. + // Unregisterable, therefore false rather than an error. + {"kind contains-any (bug, task)", false}, + {"kind contains_any [bug, task]", false}, + // `or` (DKT-548): at least one clause holds. The false-false row is the + // one that matters — a disjunction that held vacuously would run every + // lane it was written to select between. + {"kind == bug or labels contains docs", true}, + {"kind == task or labels contains backend", true}, + {"kind == task or labels contains docs", false}, + {"kind == bug or labels contains urgent", true}, + {"kind == task or kind == chore or labels contains urgent", true}, + {"kind == task or kind == chore or labels contains docs", false}, // A clause V22 would have rejected evaluates FALSE rather than // erroring: reaching it means a stored definition predates the current // grammar, and skipping the step is the conservative reading. {"status == done", false}, + // The same conservatism over the disjunction: an unparseable clause is + // false, so it cannot carry the `or`, and it cannot poison a sibling + // clause that genuinely holds. + {"status == done or labels contains backend", true}, + {"status == done or labels contains docs", false}, + // A predicate mixing connectives is not in the grammar (V22 refuses it), + // so it is false whatever its clauses say — including when every clause + // holds, which is the row that would pass under any accidental + // precedence default. + {"kind == bug and labels contains backend or labels contains urgent", false}, + {"labels contains urgent or kind == bug and labels contains backend", false}, } for _, tc := range cases { @@ -331,3 +415,344 @@ func TestHarvestFencesInfoStringTag(t *testing.T) { t.Errorf("HarvestFences over an info string with attributes = %+v", got) } } + +// TestSplitWhenReportsItsConnective covers the parser V22 and WhenHolds share. +// Both read the connective from here, so a predicate that splits one way for +// the validator and another for the evaluator is impossible by construction — +// which is the property DKT-548 required be preserved, not just the new `or`. +func TestSplitWhenReportsItsConnective(t *testing.T) { + cases := []struct { + expr string + clauses []string + connective string + mixed bool + }{ + // Nothing to join: `and` is the identity the pre-DKT-548 grammar had. + {expr: "kind == bug", clauses: []string{"kind == bug"}, connective: WhenAnd}, + { + expr: "kind == bug and labels contains x", + clauses: []string{"kind == bug", "labels contains x"}, + connective: WhenAnd, + }, + { + expr: "labels contains security-load-bearing or labels contains security", + clauses: []string{"labels contains security-load-bearing", "labels contains security"}, + connective: WhenOr, + }, + { + expr: "kind == bug or labels contains x", + clauses: []string{"kind == bug", "labels contains x"}, + connective: WhenOr, + }, + { + expr: "a and b or c", + clauses: []string{"a", "b", "c"}, + connective: WhenAnd, // the FIRST connective; `mixed` is what refuses it + mixed: true, + }, + // The connective is a whitespace-delimited token, so a literal that + // merely embeds one is not a split point. + { + expr: "labels contains and-then", + clauses: []string{"labels contains and-then"}, + connective: WhenAnd, + }, + { + expr: "labels contains x-or-y", + clauses: []string{"labels contains x-or-y"}, + connective: WhenAnd, + }, + } + + for _, tc := range cases { + t.Run(tc.expr, func(t *testing.T) { + clauses, connective, mixed := splitWhen(tc.expr) + if !slices.Equal(clauses, tc.clauses) { + t.Errorf("clauses = %q, want %q", clauses, tc.clauses) + } + if connective != tc.connective { + t.Errorf("connective = %q, want %q", connective, tc.connective) + } + if mixed != tc.mixed { + t.Errorf("mixed = %v, want %v", mixed, tc.mixed) + } + }) + } +} + +// TestDisjunctionCollapsesTwoIdenticalLanes is DKT-548's motivating case as an +// assertion: the workflow that routed a TDD-security author lane on +// "security-load-bearing OR security" carried two byte-identical steps whose +// only difference was the predicate, because the grammar had no `or`. +// +// The test asserts the collapse is EXACT — the single predicate holds on +// precisely the subjects on which at least one of the two old lanes' predicates +// held, and on no others. An `or` that merely covered the motivating labels +// while also holding on unrelated issues would collapse the duplication and +// widen the routing, which is a worse defect than the duplication. +func TestDisjunctionCollapsesTwoIdenticalLanes(t *testing.T) { + const ( + lane = "labels contains security-load-bearing" + laneAlt = "labels contains security" + merged = "labels contains security-load-bearing or labels contains security" + ) + + subjects := []Subject{ + {Kind: "bug"}, + {Kind: "bug", Labels: []string{"security-load-bearing"}}, + {Kind: "bug", Labels: []string{"security"}}, + {Kind: "bug", Labels: []string{"security-load-bearing", "security"}}, + {Kind: "task", Labels: []string{"security"}}, + {Kind: "chore", Labels: []string{"backend", "urgent"}}, + {Kind: "feature", Labels: []string{"securityish"}}, + } + + for _, s := range subjects { + want := WhenHolds(lane, s) || WhenHolds(laneAlt, s) + if got := WhenHolds(merged, s); got != want { + t.Errorf("WhenHolds(%q, %+v) = %v; the two lanes it replaces give %v", + merged, s, got, want) + } + } +} + +// TestContainsAnyMirrorsLabelsAny is DKT-550's semantic requirement as an +// assertion: the step-level `labels contains-any (…)` clause must answer a +// subject exactly as the workflow-level `labels_any` [match] clause does. +// +// Two spellings of "any of these labels" that disagreed on any subject would +// mean a step's condition and its workflow's binding rule read the same issue +// differently — which is the failure the mirror exists to rule out, not a +// stylistic preference. +func TestContainsAnyMirrorsLabelsAny(t *testing.T) { + list := []string{"security-change", "security"} + m := &Match{LabelsAny: list} + + subjects := []Subject{ + {Kind: "doc"}, + {Kind: "doc", Labels: []string{"security"}}, + {Kind: "doc", Labels: []string{"security-change"}}, + {Kind: "doc", Labels: []string{"security-change", "security"}}, + {Kind: "doc", Labels: []string{"security-load-bearing"}}, + {Kind: "doc", Labels: []string{"doc:tdd"}}, + {Kind: "doc", Labels: []string{"doc:tdd", "security"}}, + {Kind: "bug", Labels: []string{"securityish"}}, + } + + // Every accepted spelling of the clause is held to the same mirror + // (DKT-1000): a delimiter or an underscore must not be able to change what + // the predicate asks. + spellings := []string{ + "labels contains-any (security-change, security)", + "labels contains_any [security-change, security]", + "labels contains_any (security-change, security)", + "labels contains-any [security-change, security]", + `labels contains_any ["security-change", "security"]`, + } + for _, when := range spellings { + for _, s := range subjects { + want := m.Matches(s) + if got := WhenHolds(when, s); got != want { + t.Errorf("WhenHolds(%q, %+v) = %v; labels_any %q gives %v", + when, s, got, list, want) + } + } + } +} + +// TestContainsAnySpellingsAgree is DKT-1000's compatibility requirement: adding +// `contains_any` and the bracketed list changed no existing operator's meaning. +// +// The registered corpus carries `contains-any (…)`. If the new spelling parsed +// through a second code path — or if adding the bracket branch shifted a +// capture group the evaluator reads by index — a stored definition would keep +// registering and start evaluating differently, which is the one failure a new +// operator must not be able to cause. The sweep runs the four spellings over +// every labels subset so a divergence anywhere is caught, not just on the +// intersecting cases. +func TestContainsAnySpellingsAgree(t *testing.T) { + spellings := []string{ + "labels contains-any (a, b)", + "labels contains_any (a, b)", + "labels contains-any [a, b]", + "labels contains_any [a, b]", + } + + subjects := []Subject{ + {Kind: "bug"}, + {Kind: "bug", Labels: []string{"a"}}, + {Kind: "bug", Labels: []string{"b"}}, + {Kind: "bug", Labels: []string{"a", "b"}}, + {Kind: "bug", Labels: []string{"c"}}, + {Kind: "bug", Labels: []string{"c", "b"}}, + {Kind: "bug", Labels: []string{"ab"}}, + {Kind: "bug", Labels: []string{""}}, + } + + for _, s := range subjects { + want := WhenHolds(spellings[0], s) + for _, when := range spellings[1:] { + if got := WhenHolds(when, s); got != want { + t.Errorf("WhenHolds(%q, %+v) = %v; %q gives %v", + when, s, got, spellings[0], want) + } + } + } +} + +// TestContainsAnyNeedsNoConnectiveMix is DKT-550's motivating case: routing TDD +// authoring to a security-specialized executor when the issue carries ANY of +// the security label spellings. +// +// The natural predicate — `labels contains doc:tdd and (labels contains +// security-change or labels contains security)` — MIXES connectives, which V22 +// refuses and WhenHolds therefore reads as false. The test asserts the set form +// expresses the same routing as a single homogeneous-`and` predicate, and that +// it agrees on every subject with the mixed reading the author meant. The +// mixing rule itself is untouched: the row asserting the mixed spelling is +// still false is the guard on that. +func TestContainsAnyNeedsNoConnectiveMix(t *testing.T) { + const ( + mixed = "labels contains doc:tdd and (labels contains security-change or labels contains security)" + single = "kind == doc:tdd and labels contains-any (security-change, security)" + ) + + subjects := []Subject{ + {Kind: "doc:tdd"}, + {Kind: "doc:tdd", Labels: []string{"security"}}, + {Kind: "doc:tdd", Labels: []string{"security-change"}}, + {Kind: "doc:tdd", Labels: []string{"security-change", "security"}}, + {Kind: "doc:tdd", Labels: []string{"security-load-bearing"}}, + {Kind: "doc:tdd", Labels: []string{"backend"}}, + {Kind: "bug", Labels: []string{"security"}}, + {Kind: "bug"}, + } + + for _, s := range subjects { + // The intent the author could not spell: kind is doc:tdd AND at least + // one security spelling is present. + want := s.Kind == "doc:tdd" && + (slices.Contains(s.Labels, "security-change") || slices.Contains(s.Labels, "security")) + if got := WhenHolds(single, s); got != want { + t.Errorf("WhenHolds(%q, %+v) = %v, want %v", single, s, got, want) + } + // The mixed spelling remains outside the grammar, on every subject. + if WhenHolds(mixed, s) { + t.Errorf("WhenHolds(%q, %+v) = true; a mixed predicate must not hold", mixed, s) + } + } +} + +// TestContainsAnyRoutesSecurityTDD is DKT-1000's requested predicate, verbatim: +// spec-doc routes a `doc:tdd` issue to the security TDD author when it carries +// ANY of policy.toml's three [security].labels. +// +// It asserts the whole ask in one string — a labels clause and a set clause +// joined by the single `and` connective — rather than the one duplicated +// [[step]] per label the grammar used to force. The false rows are the load- +// bearing ones: a `doc:tdd` issue with an unrelated label, and a security issue +// that is not a TDD, must both stay out of the specialized lane. +func TestContainsAnyRoutesSecurityTDD(t *testing.T) { + const when = "labels contains doc:tdd and labels contains_any " + + "[security-change, security-load-bearing, security]" + + cases := []struct { + labels []string + want bool + }{ + {nil, false}, + {[]string{"doc:tdd"}, false}, + {[]string{"security"}, false}, + {[]string{"security-change"}, false}, + {[]string{"doc:tdd", "security"}, true}, + {[]string{"doc:tdd", "security-change"}, true}, + {[]string{"doc:tdd", "security-load-bearing"}, true}, + {[]string{"doc:tdd", "security-change", "security"}, true}, + {[]string{"doc:tdd", "backend"}, false}, + {[]string{"doc:tdd", "securityish"}, false}, + {[]string{"doc:adr", "security"}, false}, + } + + for _, tc := range cases { + s := Subject{Kind: "doc", Labels: tc.labels} + if got := WhenHolds(when, s); got != tc.want { + t.Errorf("WhenHolds(%q, labels=%v) = %v, want %v", when, tc.labels, got, tc.want) + } + } +} + +// TestContainsAnySplitsAsOneClause pins the interaction DKT-550 has with the +// DKT-548 splitter: a `contains-any` list is ONE clause, and the commas inside +// it are not connectives. If splitWhen ever cut inside the parens, V22 and +// WhenHolds would still agree (both use it), but the clause would stop being +// evaluable — so this is the assertion that the new form needed no change to +// the connective logic at all. +func TestContainsAnySplitsAsOneClause(t *testing.T) { + cases := []struct { + expr string + clauses []string + connective string + }{ + { + expr: "labels contains-any (a, b, c)", + clauses: []string{"labels contains-any (a, b, c)"}, + connective: WhenAnd, + }, + { + expr: "kind == doc:tdd and labels contains-any (a, b)", + clauses: []string{ + "kind == doc:tdd", "labels contains-any (a, b)", + }, + connective: WhenAnd, + }, + { + expr: "labels contains-any (a, b) or kind == bug", + clauses: []string{ + "labels contains-any (a, b)", "kind == bug", + }, + connective: WhenOr, + }, + // The bracketed spelling is subject to the identical claim (DKT-1000): + // the commas inside `[…]` are not connectives either. + { + expr: "labels contains_any [a, b, c]", + clauses: []string{"labels contains_any [a, b, c]"}, + connective: WhenAnd, + }, + { + expr: "kind == doc:tdd and labels contains_any [security-change, security-load-bearing, security]", + clauses: []string{ + "kind == doc:tdd", + "labels contains_any [security-change, security-load-bearing, security]", + }, + connective: WhenAnd, + }, + { + expr: "labels contains_any [a, b] or kind == bug", + clauses: []string{ + "labels contains_any [a, b]", "kind == bug", + }, + connective: WhenOr, + }, + } + + for _, tc := range cases { + t.Run(tc.expr, func(t *testing.T) { + clauses, connective, mixed := splitWhen(tc.expr) + if mixed { + t.Errorf("splitWhen(%q) reports mixed", tc.expr) + } + if !slices.Equal(clauses, tc.clauses) { + t.Errorf("clauses = %q, want %q", clauses, tc.clauses) + } + if connective != tc.connective { + t.Errorf("connective = %q, want %q", connective, tc.connective) + } + for _, clause := range clauses { + if !whenShape.MatchString(clause) { + t.Errorf("clause %q does not match whenShape", clause) + } + } + }) + } +} diff --git a/internal/workflow/parse.go b/internal/workflow/parse.go index 1eb7af9a..ca166519 100644 --- a/internal/workflow/parse.go +++ b/internal/workflow/parse.go @@ -62,6 +62,15 @@ type Gate struct { Pre bool `json:"pre"` } +// PassFloor is a step's `pass_floor = { field, at }` table (DKT-870): the +// author's exit bar on a `pass` routing. Both values are OPAQUE TOKENS — a +// payload property name and a value of that property's declared order — read +// by position exactly as `route_at`'s value is, and by nothing else. +type PassFloor struct { + Field string `toml:"field" json:"field"` + At string `toml:"at" json:"at"` +} + // Step is one §11.1 `[[step]]`, carrying every row of that table. type Step struct { Name string `toml:"name" json:"name"` @@ -80,17 +89,108 @@ type Step struct { Params map[string]any `toml:"params" json:"params,omitempty"` MinSiblings *int `toml:"min_siblings" json:"min_siblings,omitempty"` Threshold map[string]string `toml:"threshold" json:"threshold,omitempty"` - OnFail string `toml:"on_fail" json:"on_fail,omitempty"` - Loop bool `toml:"loop" json:"loop"` - AfterLoop string `toml:"after_loop" json:"after_loop,omitempty"` - MaxAttempts *int `toml:"max_attempts" json:"max_attempts,omitempty"` + // PassFloor refuses a `pass` routing that would exit with declared-floor + // work still standing (DKT-870): when this step's routing resolves to + // `pass` but its recorded payload holds an element whose `field` value + // sits AT OR ABOVE `at`'s position in the step's pinned schema order — + // and that element is neither `held` nor `operator_resolved` — the step + // parks `waiting-human` instead of exiting. + // + // It exists because a threshold reads whatever field its author chose, + // and a routing step's self-reported disposition can contradict its own + // recorded evidence: RUN-58's reconcile routed `pass` and the loop exited + // with all 16 clusters open, SIX at the order's high position, none held + // and none operator-resolved — "converged" in the ledger meaning + // "dispositioned". The floor is the author's declaration of the exit bar, + // exactly as `route_at` is the author's declaration of the routing floor: + // `field` and `at` are opaque tokens compared only by position in the + // declared order (genericity.md), so core still holds no opinion about + // severities. + // + // `held` and `operator_resolved` elements are exempt because both already + // carry a decision channel: a held cluster gates the step for an operator, + // and a resolved one records the operator's acceptance. The park names + // `--as override-pass` (exit as recorded) and `--as fix-round` (buy a + // round instead) as the ways out. Requires `payload` (V37) — without a + // pinned order there is no position to compare — and the floor's own + // coherence against that schema is V37a's, at register time. + PassFloor *PassFloor `toml:"pass_floor" json:"pass_floor,omitempty"` + OnFail string `toml:"on_fail" json:"on_fail,omitempty"` + Loop bool `toml:"loop" json:"loop"` + AfterLoop string `toml:"after_loop" json:"after_loop,omitempty"` + // Serves scopes a `loop = true` body to the steps whose `fix-loop` + // routings it answers (§11.3 cluster scoping, DKT-544): on a `fix-loop` + // routed by a step named here, this body instantiates and its + // `after_loop` root joins the sweep/re-instantiation set; on one routed + // by any other step, this body is not part of the entry at all. + // + // OMITTED MEANS SERVES EVERY TRIGGER — the backward-compatible reading. + // A workflow with no `serves` anywhere therefore has exactly one loop + // cluster spanning every body and every `after_loop` root, which is + // §11.3's original single-construct behavior, byte for byte. + // + // Valid only on `loop = true` steps (V35); every entry must name a step + // that can actually route `fix-loop` (V35), and every step that can must + // be served by at least one body (V17c). + Serves []string `toml:"serves" json:"serves,omitempty"` + MaxAttempts *int `toml:"max_attempts" json:"max_attempts,omitempty"` // MaxFixLoops bounds EVERY `fix-loop` routing on the issue -- threshold, - // rejected vote, or rejected human gate alike (see engine.EnterLoop). The - // only way past it is a tracked operator grant (`fix-round`), never a skip. - MaxFixLoops *int `toml:"max_fix_loops" json:"max_fix_loops,omitempty"` - ExpectedCost *float64 `toml:"expected_cost" json:"expected_cost,omitempty"` - When string `toml:"when" json:"when,omitempty"` - Metadata map[string]any `toml:"metadata" json:"metadata,omitempty"` + // gate-failure or attempt-exhaustion `on_fail`, rejected vote, rejected + // human gate, and quorum miss alike (see engine.EnterLoop; DKT-587). It is + // ONE workflow-wide bound over ONE issue-level counter: whichever + // non-cluster step declares it, every routing source moves the same + // counter, so the bound cannot depend on which step happened to route. + // + // HOW THE COUNT IS COMPUTED (§11.3 (1)): each admitted entry increments + // the issue's counter FIRST and the new value is that entry's 1-indexed + // loop ordinal; an entry whose new count EXCEEDS the bound is refused -- + // the counter is put back, nothing is instantiated, and the routing step + // parks `waiting-human`. `max_fix_loops = N` therefore admits exactly N + // loop entries (ordinals 1..N) and refuses the N+1th, from any mix of + // routing sources. Zero or absent means unbounded. + // + // Declared on a `serves`-scoped loop body it is instead that CLUSTER's + // round bound (§11.3 cluster scoping): it bounds entries triggered by the + // steps the body serves — counted as the distinct ordinals holding the + // cluster's scoped bodies, checked INDEPENDENTLY of and ADDITIVELY under + // the issue-level ceiling, which stays whatever a non-cluster step + // declares. A cluster bound never raises or lowers the issue-level one; + // each refuses on its own arithmetic, and both admit exactly their + // declared number of rounds. + // + // The only way past either bound is a tracked operator grant — `docket + // step resolve --as fix-round` (DKT-237) — never a skip. Each grant + // records `loop_grants + 1` on the issue's run row and enters the round in + // the same transaction: the effective issue-level bound becomes declared + + // grants, and the authorized entry also skips the cluster bound and the + // non-convergence refusal, because the operator who granted it has + // answered the question those refusals ask. A third completed ordinal + // under `max_fix_loops = 2` is therefore the signature of a recorded + // grant, not of the counter differing by entry path. + MaxFixLoops *int `toml:"max_fix_loops" json:"max_fix_loops,omitempty"` + // MaxStalledRounds parks a `fix-loop` entry when this step — a routing + // step, i.e. one that can route `fix-loop` — has recorded that many + // CONSECUTIVE rounds without its routed payload's element count ever + // falling below the smallest count any earlier round recorded (DKT-870). + // + // It is the author's declaration of the non-convergence signal the corpus + // already reads by hand: a fix loop that is converging shrinks the standing + // set it routes each round, and one whose volume stays flat is churning. + // RUN-51 held 8-12 clusters across TEN rounds (~271k + ~251k output tokens + // spent on the last two alone) and RUN-50 held 7-10 across six; both ended + // only by operator action, because nothing engine-visible read the plateau. + // The count is of PAYLOAD ELEMENTS — packaging vocabulary, like `held` — + // so core never learns what a cluster or a severity is. + // + // The refusal takes the non-convergence park's exact shape (engine.EnterLoop, + // DKT-340/DKT-589): nothing superseded, nothing instantiated, the counter + // restored, `waiting-human` naming `--as fix-round` as the way out, and an + // authorized entry skips it. Zero or absent means the check never fires — + // the same opt-in reading `hold_spread = 0` has. + MaxStalledRounds *int `toml:"max_stalled_rounds" json:"max_stalled_rounds,omitempty"` + ExpectedCost *float64 `toml:"expected_cost" json:"expected_cost,omitempty"` + When string `toml:"when" json:"when,omitempty"` + Metadata map[string]any `toml:"metadata" json:"metadata,omitempty"` // HoldsTree reports whether this step OCCUPIES its issue's scope while it // runs — the question scope exclusion (R4) is actually asking. // diff --git a/internal/workflow/schemas.go b/internal/workflow/schemas.go index 9a628986..ee83ceac 100644 --- a/internal/workflow/schemas.go +++ b/internal/workflow/schemas.go @@ -4,6 +4,7 @@ import ( "encoding/json" "errors" "fmt" + "slices" "sort" "strconv" "strings" @@ -105,6 +106,57 @@ func ValidateSchemas(def *Definition, resolver SchemaResolver) error { if err := validateAggregateSchema(step, registered); err != nil { return withWorkflow(err, def.Pipeline.Name) } + if err := validatePassFloorSchema(step, registered); err != nil { + return withWorkflow(err, def.Pipeline.Name) + } + } + return nil +} + +// validatePassFloorSchema is V37a (DKT-870): a declared `pass_floor` must be +// answerable by the step's pinned schema — the field declared and ordered, and +// `at` a value of that order. It is V28a's discipline applied to the exit bar: +// a floor is a POSITION in the declared order, a value with no position has no +// floor to name, and the question is asked at register time rather than at the +// end of a run, on the routing whose `pass` the floor exists to gate. V37 has +// already required `field` and `at` to be non-empty and `payload` to be +// declared; this is the half only the schema can answer. +func validatePassFloorSchema(step *Step, registered *Registered) error { + if step.PassFloor == nil { + return nil + } + declared, ok := registered.Field(step.PassFloor.Field) + if !ok { + return &Error{ + Rule: "V37a", Step: step.Name, Field: "pass_floor", + Message: fmt.Sprintf( + "step %q: `pass_floor.field` names %q, which %s does not "+ + "declare. Declared fields: %s", + step.Name, step.PassFloor.Field, registered.Ref(), + describeFields(registered)), + } + } + if !declared.Ordered { + return &Error{ + Rule: "V37a", Step: step.Name, Field: "pass_floor", + Message: fmt.Sprintf( + "step %q: `pass_floor.field` names %q, but %s declares no "+ + "`ordered_enum` on it; an exit bar is a position in a "+ + "declared order, and core does not invent one", + step.Name, step.PassFloor.Field, registered.Ref()), + } + } + if !slices.Contains(declared.Enum, step.PassFloor.At) { + return &Error{ + Rule: "V37a", Step: step.Name, Field: "pass_floor", + Message: fmt.Sprintf( + "step %q: `pass_floor.at` names %q, which is not in the order "+ + "%s declares for %q (%s); an exit bar is a position in the "+ + "declared order, and core does not guess one for an unknown "+ + "value", + step.Name, step.PassFloor.At, registered.Ref(), + step.PassFloor.Field, quotedList(declared.Enum)), + } } return nil } @@ -149,6 +201,28 @@ func validateAggregateSchema(step *Step, registered *Registered) error { } } + // V28a: `route_at`, when declared, must name a value of the field's + // declared order — the routing floor is a POSITION in that order, and a + // value with no position has no floor to name (G4's discipline, asked at + // register time rather than hours into a run). V28 has already required it + // to be a non-empty string; this is the half only the schema can answer, + // which is why it lives here beside V29 rather than in the pure-bytes + // param check. + if routeAt, ok := step.Params["route_at"].(string); ok && routeAt != "" { + if !slices.Contains(declared.Enum, routeAt) { + return &Error{ + Rule: "V28a", Step: step.Name, Field: "params", + Message: fmt.Sprintf( + "step %q: `params.route_at` names %q, which is not in the "+ + "order %s declares for %q (%s); a routing floor is a "+ + "position in the declared order, and core does not guess "+ + "one for an unknown value", + step.Name, routeAt, registered.Ref(), field, + quotedList(declared.Enum)), + } + } + } + // V30: the conjunction must be SATISFIABLE. return validateAggregateProbe(step, registered, declared) } diff --git a/internal/workflow/validate.go b/internal/workflow/validate.go index df0c83d4..a9d0adcb 100644 --- a/internal/workflow/validate.go +++ b/internal/workflow/validate.go @@ -8,6 +8,8 @@ import ( "strconv" "strings" "time" + + "github.com/ALT-F4-LLC/docket/internal/model" ) // RuleIDs is the register-time validation table, by ID, in documented order — @@ -17,17 +19,17 @@ import ( // (V13/V13a are one check and two author-facing errors; V21 is now one grammar // rule and four cross-validation rules). // -// V21a-V21d, V25a, V29, and V30 are the SCHEMA-AWARE half and live in -// ValidateSchemas rather than in Validate: they are the only rules that ask a -// question about the environment, and Validate stays a pure function of bytes +// V21a-V21d, V25a, V28a, V29, V30, and V37a are the SCHEMA-AWARE half and live +// in ValidateSchemas rather than in Validate: they are the only rules that ask +// a question about the environment, and Validate stays a pure function of bytes // (§4.9.2). V27, V28, and V31 are decisions about bytes and stay here. var RuleIDs = []string{ "V1", "V2", "V3", "V4", "V5", "V6", "V7", "V8", "V9", "V10", - "V11", "V12", "V13", "V13a", "V14", "V15", "V16", "V17", "V17b", "V18", "V19", + "V11", "V11a", "V11b", "V12", "V13", "V13a", "V14", "V15", "V16", "V17", "V17b", "V17c", "V18", "V19", "V20", "V21", "V21a", "V21b", "V21c", "V21d", "V22", "V23", "V24", "V25", "V25a", "V26", - "V27", "V28", "V29", "V30", "V31", - "V32", "V33", "V34", + "V27", "V28", "V28a", "V29", "V30", "V31", + "V32", "V33", "V34", "V35", "V36", "V37", "V37a", "V38", } // VoteRuleResolver reports whether a named vote rule is registered, and lists @@ -227,6 +229,19 @@ func validateSteps(def *Definition) error { step.Name), } } + // The same reservation for `issue.linked` (DKT-547): the + // cross-issue linked-artifact form is resolved before any step + // lookup, so a step under that name could never be addressed. + if step.Name == "issue.linked" || strings.HasPrefix(step.Name, InputIssueLinkedPrefix) { + return &Error{ + Rule: "V34", Step: step.Name, Field: "name", + Message: fmt.Sprintf( + "step name %q is reserved: `issue.linked..` is "+ + "the engine-served cross-issue input form, and a step under "+ + "that name could never be addressed as an input", + step.Name), + } + } if _, dup := byName[step.Name]; dup { return &Error{ Rule: "V4", Step: step.Name, Field: "name", @@ -313,6 +328,22 @@ func validateStep(def *Definition, step *Step, index int, byName map[string]*Ste } } + // V11b: `vote-record` is a RESERVED kind (DKT-545) — V11a's reasoning for + // the vote step's record: `.vote-record` resolves to the proposal + // the engine recorded for the named vote step, so a step emitting an + // artifact of that kind could never be addressed as an input. + if produced, _ := producedKind(step); produced == VoteRecordKind { + return &Error{ + Rule: "V11b", Step: step.Name, Field: "emits", + Message: fmt.Sprintf( + "step %q produces kind %q, which is reserved: "+ + "`.%s` is the engine-served input form for a vote "+ + "step's recorded proposal, and an artifact of that kind "+ + "could never be addressed", step.Name, VoteRecordKind, + VoteRecordKind), + } + } + // V8: after required except on the first step and on loop = true steps. // V10: `after = []` is legal and means root; a MISSING `after` on a // non-exempt step is the error. The two are one check over hasAfter, and @@ -550,6 +581,74 @@ func validateStep(def *Definition, step *Step, index int, byName map[string]*Ste } } + // V35: `serves` declares a loop CLUSTER (§11.3 cluster scoping, DKT-544) + // and every part of the declaration must be coherent at register time: + // only a `loop = true` body has fix-loop routings to serve, an entry must + // name a step of this workflow, and the named step must be able to route + // `fix-loop` at all — a body serving a step that never routes there is a + // cluster that can never be entered, which is a misdeclaration and not a + // choice. + if len(step.Serves) > 0 && !step.Loop { + return &Error{ + Rule: "V35", Step: step.Name, Field: "serves", + Message: fmt.Sprintf( + "step %q: `serves` is only valid on `loop = true` steps — it names "+ + "the steps whose `fix-loop` routings this loop body answers", + step.Name), + } + } + for _, target := range step.Serves { + if strings.TrimSpace(target) == "" { + return &Error{ + Rule: "V35", Step: step.Name, Field: "serves", + Message: fmt.Sprintf("step %q: `serves` contains an empty entry", step.Name), + } + } + named, ok := byName[target] + if !ok { + return &Error{ + Rule: "V35", Step: step.Name, Field: "serves", + Message: fmt.Sprintf( + "step %q: `serves` names %q, which is not a step in this workflow", + step.Name, target), + } + } + if !canRouteFixLoop(named) { + return &Error{ + Rule: "V35", Step: step.Name, Field: "serves", + Message: fmt.Sprintf( + "step %q: `serves` names %q, which never routes `fix-loop` — "+ + "neither its `on_fail` nor any `threshold` key routes there, "+ + "so this loop body would serve a trigger that cannot fire", + step.Name, target), + } + } + } + + // V17c — V17b scoped per trigger (DKT-544): once any body declares + // `serves`, a step that can route `fix-loop` must still be served by AT + // LEAST ONE `loop = true` body (a body with no `serves` serves every + // trigger, so this only bites when every body is scoped and one trigger is + // in none of their lists). An unserved trigger's loop entry would bump the + // counter and instantiate nothing — V17b's exact silent-no-op shape, + // reintroduced per cluster. + if hasLoopStep(def) && canRouteFixLoop(step) && !anyBodyServes(def, step.Name) { + field := "on_fail" + if step.OnFail != OnFailFixLoop { + field = "threshold" + } + return &Error{ + Rule: "V17c", Step: step.Name, Field: field, + Message: fmt.Sprintf( + "step %q can route `fix-loop`, but no `loop = true` step serves it — "+ + "every loop body's `serves` names other steps, so this trigger's "+ + "loop entry would supersede downstream work and instantiate "+ + "nothing in its place; add %q to a body's `serves` (or declare "+ + "a body without `serves`, which serves every trigger)", + step.Name, step.Name), + } + } + // V19: max_attempts >= 1; max_fix_loops >= 0; expected_cost >= 0. if step.MaxAttempts != nil && *step.MaxAttempts < 1 { return &Error{ @@ -565,6 +664,14 @@ func validateStep(def *Definition, step *Step, index int, byName map[string]*Ste "step %q: `max_fix_loops` must be >= 0, got %d", step.Name, *step.MaxFixLoops), } } + if step.MaxStalledRounds != nil && *step.MaxStalledRounds < 0 { + return &Error{ + Rule: "V19", Step: step.Name, Field: "max_stalled_rounds", + Message: fmt.Sprintf( + "step %q: `max_stalled_rounds` must be >= 0, got %d", + step.Name, *step.MaxStalledRounds), + } + } if step.ExpectedCost != nil && *step.ExpectedCost < 0 { return &Error{ Rule: "V19", Step: step.Name, Field: "expected_cost", @@ -573,6 +680,63 @@ func validateStep(def *Definition, step *Step, index int, byName map[string]*Ste } } + // V38: `max_stalled_rounds` is a non-convergence bound over THIS step's + // per-round routed volume (DKT-870), so it is coherent only on a step + // that (a) can actually route `fix-loop` — the check runs at loop entry, + // triggered by this step's own routing — and (b) records an artifact + // whose payload the volume can be counted from. Declared anywhere else it + // is inert, and an inert declaration is a misdeclaration, not a choice + // (V17/V35's discipline). + if step.MaxStalledRounds != nil && *step.MaxStalledRounds > 0 { + if !canRouteFixLoop(step) { + return &Error{ + Rule: "V38", Step: step.Name, Field: "max_stalled_rounds", + Message: fmt.Sprintf( + "step %q: `max_stalled_rounds` bounds this step's own "+ + "`fix-loop` rounds, but neither its `on_fail` nor any "+ + "`threshold` key routes there, so the bound could never "+ + "fire", step.Name), + } + } + if ArtifactKind(step) == "" { + return &Error{ + Rule: "V38", Step: step.Name, Field: "max_stalled_rounds", + Message: fmt.Sprintf( + "step %q: `max_stalled_rounds` counts the elements of this "+ + "step's recorded payload per round, and a `type` step "+ + "records no artifact to count, so the bound could never "+ + "fire", step.Name), + } + } + } + + // V37: `pass_floor` compares a payload value's POSITION against the + // declared `at`'s (DKT-870), and a position exists only in the order a + // pinned `payload` schema declares — a floor with no schema can never + // position anything and is an inert declaration. Whether the field IS + // ordered and `at` IS in its order is V37a's, beside the schema (§4.9.1). + if step.PassFloor != nil { + if step.PassFloor.Field == "" || step.PassFloor.At == "" { + return &Error{ + Rule: "V37", Step: step.Name, Field: "pass_floor", + Message: fmt.Sprintf( + "step %q: `pass_floor` requires both `field` (the payload "+ + "property to compare) and `at` (a value of that "+ + "property's declared order)", step.Name), + } + } + if step.Payload == "" { + return &Error{ + Rule: "V37", Step: step.Name, Field: "pass_floor", + Message: fmt.Sprintf( + "step %q: `pass_floor` compares positions in a declared "+ + "order, so the step must declare `payload = "+ + "\"name@version\"` naming a schema that orders %q", + step.Name, step.PassFloor.Field), + } + } + } + // V20/V21: threshold routings and predicates. if err := validateThreshold(step, byName); err != nil { return err @@ -642,6 +806,43 @@ func LatestKind(input string) (kind string, ok bool) { return rest, true } +// InputIssueLinkedPrefix is the `issue.linked..` engine form's +// prefix (DKT-547): the latest recorded artifact of one kind held by the +// issue(s) this issue is LINKED to by a relation, resolved and pinned at +// activation. Exported because the engine's resolver consumes the same form +// the validator admits. +const InputIssueLinkedPrefix = "issue.linked." + +// linkedRelationShape is the `` half of the form — a relation token +// (canonical or inverse, hyphenated or underscored); model.ParseRelationDirection +// is the authority on which tokens mean anything, this only bounds the shape. +var linkedRelationShape = regexp.MustCompile(`^[A-Za-z0-9_-]+$`) + +// LinkedInput reports whether a declared input is the +// `issue.linked..` form, returning its two halves. ONE parser +// for the form, shared by V11, L4, and the engine's activation-time resolver — +// LatestKind's reasoning again: two readings of one grammar in two packages +// are exactly how they would drift. +// +// The split is on the FIRST dot after the prefix: relation tokens contain no +// dot, and the kind half reuses latestKindShape (one kind, never `*` — the +// wildcard answers no question a cross-issue consumer can ask). +func LinkedInput(input string) (relation, kind string, ok bool) { + rest, found := strings.CutPrefix(input, InputIssueLinkedPrefix) + if !found { + return "", "", false + } + i := strings.Index(rest, ".") + if i <= 0 || i == len(rest)-1 { + return "", "", false + } + relation, kind = rest[:i], rest[i+1:] + if !linkedRelationShape.MatchString(relation) || !latestKindShape.MatchString(kind) { + return "", "", false + } + return relation, kind, true +} + // PayloadShape matches §11.1's `schema@ver`. // // It is EXPORTED and shared rather than restated, because `docket schema @@ -730,12 +931,58 @@ func validateInputs(step *Step, byName map[string]*Step) error { continue } + // `issue.linked..` (DKT-547) is engine-produced like + // the forms above, but its PRODUCER IS ANOTHER ISSUE'S RUN: activation + // resolves this issue's linked issue(s) by the relation, pins each + // one's latest recorded artifact of the kind, and fails loudly when + // the relation or the artifact is missing. V11's produced-kind table + // therefore deliberately does NOT apply — no step of THIS workflow + // need produce the kind, because none does; the binding it enforces + // elsewhere is enforced here by activation instead. What register CAN + // refuse it does: a malformed shape, a relation outside the + // vocabulary, and the two engine-reserved kinds, which no step + // anywhere may emit (V11a/V11b) and which the form could therefore + // never resolve. + if input == "issue.linked" || strings.HasPrefix(input, InputIssueLinkedPrefix) { + relation, kind, ok := LinkedInput(input) + if !ok { + return &Error{ + Rule: "V11", Step: step.Name, Field: "inputs", + Message: fmt.Sprintf( + "step %q: `inputs` entry %q must be "+ + "`issue.linked..`, with a relation token "+ + "and one kind of letters, digits, `_` or `-` (never `*`)", + step.Name, input), + } + } + if _, _, err := model.ParseRelationDirection(relation); err != nil { + return &Error{ + Rule: "V11", Step: step.Name, Field: "inputs", + Message: fmt.Sprintf( + "step %q: `inputs` entry %q names relation %q, which is not "+ + "a relation type; must be one of %v", + step.Name, input, relation, model.RelationDirectionTokens()), + } + } + if kind == GateResultsKind || kind == VoteRecordKind { + return &Error{ + Rule: "V11", Step: step.Name, Field: "inputs", + Message: fmt.Sprintf( + "step %q: `inputs` entry %q names kind %q, which is "+ + "engine-reserved — no step may emit it, so no linked "+ + "issue could ever hold an artifact of it", + step.Name, input, kind), + } + } + continue + } + m := inputShape.FindStringSubmatch(input) if m == nil { return &Error{ Rule: "V11", Step: step.Name, Field: "inputs", Message: fmt.Sprintf( - "step %q: `inputs` entry %q must be `.`, `.*`, `issue.body`, `issue.diff`, or `issue.latest.`", + "step %q: `inputs` entry %q must be `.`, `.*`, `issue.body`, `issue.diff`, `issue.latest.`, or `issue.linked..`", step.Name, input), } } @@ -761,6 +1008,26 @@ func validateInputs(step *Step, byName map[string]*Step) error { continue } + // `.vote-record` (DKT-545) is ENGINE-PRODUCED like gate-results: + // it resolves to the proposal record the named vote step's tally left + // — outcome, casts, and rationales — so it needs the producer to exist + // and to BE a vote step. Any other step opens no proposal, so the + // input could never resolve to anything on any run: a typo caught now + // rather than an input silently absent forever. + if kind == VoteRecordKind { + if producer.Type != TypeVote { + return &Error{ + Rule: "V11", Step: step.Name, Field: "inputs", + Message: fmt.Sprintf( + "step %q: `inputs` entry %q names step %q, which is not a "+ + "`type=\"vote\"` step — `%s` resolves only against vote "+ + "steps, whose tally leaves the record it serves", + step.Name, input, producerName, VoteRecordKind), + } + } + continue + } + produced, produces := producedKind(producer) if !produces { return &Error{ @@ -801,6 +1068,28 @@ func anyStepProduces(byName map[string]*Step, kind string) bool { // thresholdRoutings are the §11.2 non-step routings. var thresholdRoutings = []string{OnFailFixLoop, OnFailWaitingHuman, "pass"} +// VoteCastField* are the addressable fields of one CAST in a vote step's +// threshold evaluation (DKT-545). The engine builds each cast's payload from +// exactly these keys, so the vocabulary is engine-defined the same way `when`'s +// kind/labels vocabulary is (V22) — a field outside it would silently never +// match, which is why V36 refuses it at register instead. +// +// `vote` and `verdict` are ALIASES for the same value: the model's word is +// `verdict` (model.Verdict, `vote cast --verdict`), and `vote` is the word a +// threshold author reaches for (`count>=2(vote == approve-with-concerns)`). +// Admitting both costs one map key; refusing one of them would refuse the +// spelling half of authors would try first. +const ( + VoteCastFieldVote = "vote" + VoteCastFieldVerdict = "verdict" + VoteCastFieldVoter = "voter" +) + +// VoteCastFields is the closed field vocabulary of a vote step's threshold +// predicates — V36's authority, exported because the engine's cast-payload +// builder must produce exactly these keys and no reader may drift from it. +var VoteCastFields = []string{VoteCastFieldVote, VoteCastFieldVerdict, VoteCastFieldVoter} + // predicateShape matches the §11.2 grammar `agg(field op literal)`. var predicateShape = regexp.MustCompile( `^\s*(any|all|count>=\d+)\s*\(\s*([A-Za-z0-9_.-]+)\s*(==|!=|>=|>|<=|<)\s*(\S+?)\s*\)\s*$`) @@ -843,6 +1132,68 @@ func validateThreshold(step *Step, byName map[string]*Step) error { step.Name, step.Threshold[routing]), } } + + // V36 (DKT-545): a `type="vote"` step's threshold is evaluated over + // the tally's CAST SET after an APPROVED tally, not over recorded + // payloads, and each constraint below refuses a declaration that + // could never route: + // + // - the routing vocabulary is the three non-step routings only. + // Step-name interposition is the saga's machinery (the same- + // transaction skip of the unrouted gate, DKT-38's latch), and the + // vote routing path has none of it — a step-name key would record + // a routing nothing downstream consumes, RUN-25's exact shape. + // - the field vocabulary is the cast's (VoteCastFields). Casts are + // engine-produced like `when`'s kind/labels (V22), so any other + // field would silently never match. + // - ordered operators are refused. Casts have no registered schema, + // and §11.2 defines ordered comparisons only over `ordered_enum` + // fields — the comparison would be a guaranteed T3 park on every + // evaluation, which is a misdeclaration and not a choice. + if step.Type == TypeVote { + if !slices.Contains(thresholdRoutings, routing) { + return &Error{ + Rule: "V36", Step: step.Name, Field: "threshold", + Message: fmt.Sprintf( + "step %q: a `type=\"vote\"` step's `threshold` routing %q "+ + "must be one of %s — step-name interposition is not "+ + "available on vote steps", + step.Name, routing, quotedList(thresholdRoutings)), + } + } + pred, err := ParsePredicate(step.Threshold[routing]) + if err != nil { + // Unreachable: V21's shape check above admits exactly what + // ParsePredicate parses. Kept so a grammar drift fails loudly. + return &Error{ + Rule: "V36", Step: step.Name, Field: "threshold", + Message: fmt.Sprintf("step %q: %v", step.Name, err), + } + } + if pred.Ordered() { + return &Error{ + Rule: "V36", Step: step.Name, Field: "threshold", + Message: fmt.Sprintf( + "step %q: `threshold` predicate %q uses the ordered "+ + "operator %q, but a vote step's threshold evaluates "+ + "over casts, which have no registered schema — "+ + "ordered comparisons are defined only over "+ + "`ordered_enum` fields (engine-spec §11.2); use == or !=", + step.Name, step.Threshold[routing], pred.Op), + } + } + if !slices.Contains(VoteCastFields, pred.Field) { + return &Error{ + Rule: "V36", Step: step.Name, Field: "threshold", + Message: fmt.Sprintf( + "step %q: `threshold` predicate %q addresses field %q, "+ + "but a vote step's threshold evaluates over the cast "+ + "set, whose addressable fields are %s", + step.Name, step.Threshold[routing], pred.Field, + quotedList(VoteCastFields)), + } + } + } } return nil } @@ -851,17 +1202,90 @@ func validateThreshold(step *Step, byName map[string]*Step) error { // `kind` or `labels` only (engine-core §4: "conditions (predicates over issue // kind/labels only)"). Nothing else is addressable — a `when` over a status or // an assignee is a validation error, not a silently-false predicate. +// +// Two clause forms: +// +// - ` <==|!=|contains> ` — one value +// - `labels contains-any (a, b, c)` — the lists intersect (DKT-550), also +// spelled `labels contains_any [a, b, c]` (DKT-1000) +// +// `contains-any` is the step-level spelling of the workflow-level `labels_any` +// [match] clause, and it is evaluated by the same intersection test (Matches). +// It exists so "kind X AND any of these labels" is ONE homogeneous-`and` +// predicate: without it that need can only be written by mixing `and` with +// `or`, which V22 refuses. A set-membership operator answers it without a +// connective at all, so the homogeneity rule never enters into it. +// +// The operator has two spellings and the list has two delimiters — `contains-any` +// or `contains_any`, `(…)` or `[…]` — and all four combinations mean the same +// thing (DKT-1000). Accepting both is not decoration: `contains-any (…)` is what +// already-registered definitions carry, and `contains_any [a, b]` is the +// spelling authors reach for because it is how a list is written everywhere else +// in a workflow TOML. Refusing either would make an operator's first correct +// guess a validation error. The delimiters must PAIR — `(a, b]` matches neither +// branch — because RE2 has no backreference to enforce it in one branch, so the +// two spellings are written out as two. +// +// The list branch is FIRST in the alternation deliberately. Go's regexp is +// leftmost-first, so ordering it after the one-value branch would let +// `labels contains-any(a,b)` match as `labels contains "-any(a,b)"` — a clause +// that registers and then quietly evaluates as something the author never +// wrote. +// +// A list element may not contain whitespace, which is what keeps the new form +// clear of whenConnective: that splitter runs over the whole predicate before +// any clause is shape-checked, and it only separates on a connective with +// whitespace on BOTH sides. A pathological `(a , and , b)` therefore splits +// into two clauses that both fail this regex — V22 refuses it, and WhenHolds +// reads it as false, because both go through the same splitter and the same +// shape. Fail-closed, never divergent. var whenShape = regexp.MustCompile( - `^\s*(kind|labels)\s*(==|!=|contains)\s*(\S+)\s*$`) + `^\s*(?:` + + `(labels)\s*(contains-any|contains_any)\s*(?:` + + `\(\s*(` + whenListElements + `)\s*\)` + + `|\[\s*(` + whenListElements + `)\s*\]` + + `)` + + `|(kind|labels)\s*(==|!=|contains)\s*(\S+)` + + `)\s*$`) + +// whenListElements is a non-empty comma-separated list of whitespace-free +// values, quoted or bare. It is spelled ONCE and shared by both delimiter +// branches of whenShape so `(…)` and `[…]` cannot drift into accepting +// different element vocabularies — a list that registers under one delimiter +// and is refused under the other would make the two spellings different +// grammars wearing the same name. +// +// Brackets are excluded from an element for the same reason parens are: an +// element that could contain the closing delimiter would let `[a, b` match by +// swallowing it. +const whenListElements = `[^\s,()\[\]]+(?:\s*,\s*[^\s,()\[\]]+)*` func validateWhen(step *Step) error { - for _, clause := range splitWhen(step.When) { + clauses, _, mixed := splitWhen(step.When) + // A mixed predicate is refused BEFORE its clauses are shape-checked: the + // operator wrote something whose meaning depends on a precedence rule the + // grammar does not have, and reporting a clause typo first would tell them + // to fix the wrong thing. + if mixed { + return &Error{ + Rule: "V22", Step: step.Name, Field: "when", + Message: fmt.Sprintf( + "step %q: `when` %q mixes `and` and `or`; a predicate must join its "+ + "clauses with one connective throughout, because the grammar has "+ + "no precedence rule and no parentheses to disambiguate the mix", + step.Name, strings.TrimSpace(step.When)), + } + } + for _, clause := range clauses { if !whenShape.MatchString(clause) { return &Error{ Rule: "V22", Step: step.Name, Field: "when", Message: fmt.Sprintf( "step %q: `when` clause %q must be a predicate over `kind` or `labels` only, "+ - "as ` <==|!=|contains> `", + "as ` <==|!=|contains> ` or "+ + "`labels contains-any (a, b, c)` (equivalently "+ + "`labels contains_any [a, b, c]`), with clauses joined by "+ + "`and` throughout or `or` throughout", step.Name, strings.TrimSpace(clause)), } } @@ -869,17 +1293,55 @@ func validateWhen(step *Step) error { return nil } -// splitWhen breaks a `when` expression into its conjuncts. `and` is the only -// connective: a disjunction over kind/labels is expressible as two steps, and -// admitting one operator keeps the predicate language small enough to stay -// obviously decidable. -func splitWhen(expr string) []string { - parts := strings.Split(expr, " and ") - out := make([]string, 0, len(parts)) - for _, p := range parts { - out = append(out, strings.TrimSpace(p)) +// The two connectives of the §11.1 `when` grammar. +const ( + WhenAnd = "and" + WhenOr = "or" +) + +// whenConnective matches a connective token BETWEEN two clauses — whitespace on +// both sides is part of the match, so a label or kind literal that merely +// contains the letters (`labels contains and-then`) is not a separator. It is +// the one place the connective vocabulary is spelled, for the same reason +// whenShape is the one place the clause shape is: validate and evaluate must +// split a predicate identically or a definition that registered could evaluate +// as something else (predicate.go, on §11.2's single-regex discipline). +var whenConnective = regexp.MustCompile(`\s+(and|or)\s+`) + +// splitWhen breaks a `when` expression into its clauses and reports which +// connective joins them. +// +// `and` and `or` may each join any number of clauses, but a single predicate +// must use ONE of them throughout — `mixed` is true otherwise, and V22 refuses +// it (DKT-548). That is the whole of the precedence question: `a and b or c` +// has two readings, the grammar has no parentheses to pick one, and picking a +// default silently would route work through a lane the author did not write. +// Requiring homogeneity costs an author one extra step in the rare mixed case +// and costs nothing in the common one. +// +// A single-clause predicate reports WhenAnd: with nothing to join, the two +// connectives agree, and `and` is the identity the old grammar already had. +func splitWhen(expr string) (clauses []string, connective string, mixed bool) { + seps := whenConnective.FindAllStringSubmatchIndex(expr, -1) + + clauses = make([]string, 0, len(seps)+1) + prev := 0 + for _, sep := range seps { + clauses = append(clauses, strings.TrimSpace(expr[prev:sep[0]])) + switch conn := expr[sep[2]:sep[3]]; { + case connective == "": + connective = conn + case connective != conn: + mixed = true + } + prev = sep[1] } - return out + clauses = append(clauses, strings.TrimSpace(expr[prev:])) + + if connective == "" { + connective = WhenAnd + } + return clauses, connective, mixed } // validateLimits is V24: `max` >= 1 and the durations parse. @@ -978,6 +1440,30 @@ func hasLoopStep(def *Definition) bool { return false } +// canRouteFixLoop reports whether a step has any routing that can resolve to +// `fix-loop`: an explicit `on_fail`, or a `threshold` key. The DECLARED values +// only — the `on_fail` default is `waiting-human`, so silence never routes +// there. +func canRouteFixLoop(step *Step) bool { + if step.OnFail == OnFailFixLoop { + return true + } + _, ok := step.Threshold[OnFailFixLoop] + return ok +} + +// anyBodyServes reports whether at least one `loop = true` step serves a +// trigger — V17c's question, asked with the same ServesTrigger reading the +// engine's loop entry uses, so "who answers this trigger" has one definition. +func anyBodyServes(def *Definition, trigger string) bool { + for _, step := range def.Steps { + if step.Loop && ServesTrigger(step, trigger) { + return true + } + } + return false +} + // humanOnFailValues is the closed vocabulary minus waiting-human — the legal // routings for a type="human" step's reject, per V13. func humanOnFailValues() []string { diff --git a/internal/workflow/validate_test.go b/internal/workflow/validate_test.go index 3990e38c..0afbdebc 100644 --- a/internal/workflow/validate_test.go +++ b/internal/workflow/validate_test.go @@ -241,6 +241,57 @@ inputs = ["a.findings"] `, wants: []string{`"b"`, "`inputs`", `"a.findings"`, "change-summary"}, }, + { + // V11's vote-record clause (DKT-545): `.vote-record` resolves + // only against `type="vote"` steps — any other step opens no proposal, + // so the input could never resolve to anything on any run. + rule: "V11", name: "vote-record input naming a non-vote step", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +[[step]] +name = "b" +after = ["a"] +executor = "y" +emits = "j" +inputs = ["a.vote-record"] +`, + wants: []string{`"b"`, "`inputs`", `"a.vote-record"`, "vote"}, + }, + { + // V11a (DKT-77): `gate-results` is a reserved kind — the engine-served + // input form would shadow an artifact of that kind. + rule: "V11a", name: "a step may not emit the reserved gate-results kind", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "gate-results" +`, + wants: []string{`"a"`, `"gate-results"`, "reserved"}, + }, + { + // V11b (DKT-545): `vote-record` is reserved for the same reason. + rule: "V11b", name: "a step may not emit the reserved vote-record kind", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "vote-record" +`, + wants: []string{`"a"`, `"vote-record"`, "reserved"}, + }, { rule: "V12", name: "on_fail outside the closed vocabulary", src: ` @@ -388,11 +439,62 @@ on_fail = "fix-loop" `, wants: []string{`"a"`, "fix-loop", "loop = true"}, }, + { + // V17c is V17b scoped per trigger (DKT-544): once every body declares + // `serves`, a fix-loop-capable step named by none of them has no body + // to instantiate — the same silent no-op, reintroduced per cluster. + rule: "V17c", name: "fix-loop trigger no scoped body serves", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +on_fail = "fix-loop" +[[step]] +name = "b" +after = ["a"] +executor = "x" +emits = "k" +threshold = { "fix-loop" = "any(status == unmet)" } +[[step]] +name = "fix" +executor = "y" +emits = "k" +loop = true +serves = ["b"] +after_loop = "b" +`, + wants: []string{`"a"`, "fix-loop", "serves"}, + }, { rule: "V18", name: "loop step with after", src: twoStepPipeline("after = [\"a\"]\nloop = true\nexecutor = \"y\"\nemits = \"k\"\nafter_loop = \"a\"\n"), wants: []string{`"b"`, "loop entry"}, }, + { + // V35 (DKT-544): `serves` is a loop-cluster declaration and only a + // `loop = true` body has fix-loop routings to serve. + rule: "V35", name: "serves on a non-loop step", + src: twoStepPipeline("after = [\"a\"]\nexecutor = \"y\"\nemits = \"k\"\nserves = [\"a\"]\n"), + wants: []string{`"b"`, "`serves`", "loop = true"}, + }, + { + rule: "V35", name: "serves naming an unknown step", + src: twoStepPipeline("loop = true\nexecutor = \"y\"\nemits = \"k\"\n" + + "serves = [\"ghost\"]\nafter_loop = \"a\"\n"), + wants: []string{`"b"`, "`serves`", `"ghost"`}, + }, + { + // A serves entry naming a step that never routes `fix-loop` declares a + // cluster that can never be entered — a misdeclaration, not a choice. + rule: "V35", name: "serves naming a step that never routes fix-loop", + src: twoStepPipeline("loop = true\nexecutor = \"y\"\nemits = \"k\"\n" + + "serves = [\"a\"]\nafter_loop = \"a\"\n"), + wants: []string{`"b"`, "`serves`", `"a"`, "fix-loop"}, + }, { rule: "V19", name: "max_attempts below one", src: ` @@ -454,6 +556,24 @@ when = "assignee == someone" `, wants: []string{`"a"`, "`when`", "`kind`", "`labels`"}, }, + { + rule: "V22", name: "when mixes and with or", + // The grammar has `or` (DKT-548) but no parentheses, so `a and b or c` + // has two readings and V22 refuses to pick one. The message must name + // both connectives — an author who reads only "invalid `when`" will + // re-check the clause spellings, which are fine. + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +when = "kind == bug and labels contains security or labels contains security-load-bearing" +`, + wants: []string{`"a"`, "`when`", "`and`", "`or`"}, + }, { rule: "V23", name: "class defaults to the executor value", // V23 is a DEFAULT, not a refusal: the case asserts the applied value @@ -628,6 +748,69 @@ params = { field = "risk", method = "medain", output = "k" } schemas: registeredRisk(), wants: []string{`"a"`, "params.method", "medain", `"median"`}, }, + { + rule: "V28", name: "aggregate with a route_at that is not a string", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "up" +executor = "x" +emits = "f" +[[step]] +name = "a" +after = ["up"] +action = "aggregate" +inputs = ["up.f"] +payload = "risk-report@1" +params = { field = "risk", method = "median", output = "k", route_at = 3 } +`, + schemas: registeredRisk(), + wants: []string{`"a"`, "params.route_at", "non-empty string"}, + }, + { + rule: "V28a", name: "route_at naming a value outside the declared order", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "up" +executor = "x" +emits = "f" +[[step]] +name = "a" +after = ["up"] +action = "aggregate" +inputs = ["up.f"] +payload = "risk-report@1" +params = { field = "risk", method = "median", output = "k", route_at = "urgent" } +`, + schemas: registeredRisk(), + wants: []string{`"a"`, "params.route_at", "urgent", "risk-report@1", + `"low"`, `"medium"`, `"high"`}, + }, + { + rule: "V28a", name: "route_at naming a declared value registers clean", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "up" +executor = "x" +emits = "f" +[[step]] +name = "a" +after = ["up"] +action = "aggregate" +inputs = ["up.f"] +payload = "risk-report@1" +params = { field = "risk", method = "median", output = "k", route_at = "high" } +`, + schemas: registeredRisk(), + }, { rule: "V29", name: "aggregate over a field with no declared order", src: ` @@ -788,6 +971,294 @@ emits = "k" `, wants: []string{`"issue.latest"`, "reserved"}, }, + { + // V11's `issue.linked..` shape half (DKT-547): the + // form takes a relation token and ONE kind — a wildcard answers no + // question a cross-issue consumer can ask. + rule: "V11", name: "issue.linked with a wildcard kind", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +inputs = ["issue.linked.depends_on.*"] +`, + wants: []string{`"a"`, "`inputs`", `"issue.linked.depends_on.*"`, + "issue.linked.."}, + }, + { + // And its vocabulary half: the relation token must be a relation type + // or an inverse form — "specified_by" is not in the model's closed + // vocabulary, so declaring it would bind nothing on every activation. + rule: "V11", name: "issue.linked naming an unknown relation", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +inputs = ["issue.linked.specified_by.ux-spec"] +`, + wants: []string{`"a"`, "`inputs`", `"specified_by"`, "not a relation type"}, + }, + { + // And its reserved-kind half: `gate-results` and `vote-record` are + // engine-reserved from every `emits` (V11a/V11b), so no linked issue + // could ever hold an artifact of them. + rule: "V11", name: "issue.linked naming an engine-reserved kind", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +inputs = ["issue.linked.depends_on.gate-results"] +`, + wants: []string{`"a"`, "`inputs`", `"gate-results"`, "engine-reserved"}, + }, + { + // V34 (DKT-547): `issue.linked` and everything under it are reserved + // step names — V34's issue.latest reasoning for the cross-issue form. + rule: "V34", name: "a step name may not claim the reserved issue.linked namespace", + src: ` +[pipeline] +name = "p" +version = 1 +[[step]] +name = "issue.linked.depends_on" +executor = "x" +emits = "k" +`, + wants: []string{`"issue.linked.depends_on"`, "reserved"}, + }, + { + // V36 (DKT-545): a vote step's threshold routes over the CAST SET, + // and step-name interposition is the saga's machinery — a step-name + // key would record a routing nothing downstream consumes. + rule: "V36", name: "vote threshold routing to a step name", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "seed" +executor = "x" +emits = "k" +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["a", "b"] +vote_rule = "majority" +on_fail = "waiting-human" +threshold = { "seed" = "any(vote == approve-with-concerns)" } +`, + wants: []string{`"gate"`, "`threshold`", `"seed"`, "fix-loop"}, + }, + { + // V36: casts have no registered schema, so an ordered comparison + // would be a guaranteed T3 park on every evaluation. + rule: "V36", name: "vote threshold with an ordered operator", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "seed" +executor = "x" +emits = "k" +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["a", "b"] +vote_rule = "majority" +on_fail = "waiting-human" +threshold = { "waiting-human" = "any(vote >= approve)" } +`, + wants: []string{`"gate"`, "`threshold`", ">=", "ordered"}, + }, + { + // V36: the field vocabulary is the cast's — anything else would + // silently never match, V22's kind/labels reasoning over casts. + rule: "V36", name: "vote threshold addressing a non-cast field", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "seed" +executor = "x" +emits = "k" +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["a", "b"] +vote_rule = "majority" +on_fail = "waiting-human" +threshold = { "pass" = "any(severity == high)" } +`, + wants: []string{`"gate"`, "`threshold`", `"severity"`, "cast"}, + }, + { + // V37 (DKT-870): a floor with half a declaration positions nothing. + rule: "V37", name: "pass_floor missing its at value", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +payload = "risk-report@1" +pass_floor = { field = "risk" } +`, + wants: []string{`"a"`, "pass_floor", "`field`", "`at`"}, + }, + { + // V37: without a pinned schema there is no order to position against, + // so the floor would be permanently inert — a misdeclaration. + rule: "V37", name: "pass_floor without a declared payload", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +pass_floor = { field = "risk", at = "high" } +`, + wants: []string{`"a"`, "pass_floor", "payload"}, + }, + { + // V37a (DKT-870): the exit bar is a position in the declared order, + // V28a's discipline applied to `pass_floor.at`. + rule: "V37a", name: "pass_floor at outside the declared order", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +payload = "risk-report@1" +pass_floor = { field = "risk", at = "urgent" } +`, + schemas: registeredRisk(), + wants: []string{`"a"`, "pass_floor", "urgent", "risk-report@1", + `"low"`, `"medium"`, `"high"`}, + }, + { + rule: "V37a", name: "pass_floor over a field with no declared order", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +payload = "risk-report@1" +pass_floor = { field = "stage", at = "final" } +`, + schemas: registeredRisk(), + wants: []string{`"a"`, "pass_floor", "stage", "ordered_enum"}, + }, + { + rule: "V37a", name: "pass_floor naming a declared value registers clean", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +payload = "risk-report@1" +pass_floor = { field = "risk", at = "high" } +`, + schemas: registeredRisk(), + }, + { + // V38 (DKT-870): the stall bound fires at THIS step's own loop entry, + // so a step that never routes `fix-loop` declares a bound that can + // never fire — V17/V35's inert-declaration discipline. + rule: "V38", name: "max_stalled_rounds on a step that never routes fix-loop", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +max_stalled_rounds = 3 +`, + wants: []string{`"a"`, "max_stalled_rounds", "fix-loop"}, + }, + { + // V38's other half: a `type` step records no artifact, so there is no + // payload whose per-round volume the bound could count. + rule: "V38", name: "max_stalled_rounds on a type step", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +[[step]] +name = "gate" +after = ["a"] +type = "human" +on_fail = "fix-loop" +max_stalled_rounds = 2 +[[step]] +name = "fix" +executor = "y" +emits = "k" +loop = true +after_loop = "a" +`, + wants: []string{`"gate"`, "max_stalled_rounds", "artifact"}, + }, + { + rule: "V38", name: "max_stalled_rounds on a routing step registers clean", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +[[step]] +name = "check" +after = ["a"] +executor = "y" +emits = "report" +threshold = { "fix-loop" = "any(status == unmet)" } +max_stalled_rounds = 3 +[[step]] +name = "fix" +executor = "z" +emits = "k" +loop = true +after_loop = "a" +`, + }, } // closedRiskSchema is riskSchema with `additionalProperties: false` — the exact @@ -1369,6 +1840,71 @@ func TestReservedActionsAreAllImplemented(t *testing.T) { } } +// TestLinkedInputRelaxesTheProducedKindTable is DKT-547's lint half, both +// directions at once. V11 holds every same-run form to the produced-kind table +// — `issue.latest.` requires SOME step of the workflow to produce the +// kind — and the cross-issue form is exactly the case where that requirement +// must NOT apply: the producer is another issue's workflow, so no step here +// produces `ux-spec` and the declaration is still registrable. L4's +// predecessor rule likewise does not apply — the artifact was recorded before +// this run existed — so the form is legal on a ROOT step, which is where a +// design-qa step consuming a spec actually sits. +func TestLinkedInputRelaxesTheProducedKindTable(t *testing.T) { + const src = ` +[pipeline] +name = "ui-change-mini" +version = 1 +[[step]] +name = "design-qa" +after = [] +executor = "qa" +emits = "qa-report" +inputs = ["issue.body", "issue.linked.depends_on.ux-spec"] +[[step]] +name = "judge-design" +after = ["design-qa"] +executor = "judge" +emits = "verdict" +inputs = ["design-qa.qa-report", "issue.linked.depends-on.ux-spec"] +` + def, err := Load([]byte(src)) + testsupport.Must(t, err, "loading: %v", err) + if err := Validate(def); err != nil { + t.Fatalf("Validate rejects the issue.linked form: %v", err) + } + if err := Lint(def); err != nil { + t.Fatalf("Lint rejects the issue.linked form: %v", err) + } +} + +// TestLinkedInputParser pins LinkedInput's grammar: one parser for the form, +// shared by V11, L4, and the engine's activation-time resolver. +func TestLinkedInputParser(t *testing.T) { + cases := []struct { + input string + relation, kind string + ok bool + }{ + {"issue.linked.depends_on.ux-spec", "depends_on", "ux-spec", true}, + {"issue.linked.depends-on.ux-spec", "depends-on", "ux-spec", true}, + {"issue.linked.blocked_by.doc", "blocked_by", "doc", true}, + {"issue.linked.depends_on.*", "", "", false}, + {"issue.linked.depends_on", "", "", false}, + {"issue.linked.depends_on.", "", "", false}, + {"issue.linked..doc", "", "", false}, + {"issue.linked.depends_on.a.b", "", "", false}, + {"issue.latest.doc", "", "", false}, + {"other.doc", "", "", false}, + } + for _, tc := range cases { + relation, kind, ok := LinkedInput(tc.input) + if relation != tc.relation || kind != tc.kind || ok != tc.ok { + t.Errorf("LinkedInput(%q) = (%q, %q, %v), want (%q, %q, %v)", + tc.input, relation, kind, ok, tc.relation, tc.kind, tc.ok) + } + } +} + // TestReRegisteringAnS3FileIsRefused is §4.9.3's honest consequence, stated // rather than hidden. // @@ -1485,3 +2021,254 @@ func TestV26KeepsTheOriginalRemedyForAGenuinelyNewRule(t *testing.T) { t.Errorf("the refusal does not name the key to set:\n%s", msg) } } + +// TestVoteThresholdAndVoteRecordRegister (DKT-545) is the positive half of +// V36/V11's vote clauses: a vote step routing its concerned approvals into the +// fix loop, with a downstream consumer reading the panel's record, is exactly +// the declaration this grammar exists to admit. +func TestVoteThresholdAndVoteRecordRegister(t *testing.T) { + src := ` +[pipeline] +name = "concern-loop" +version = 1 +[[step]] +name = "seed" +after = [] +executor = "x" +emits = "findings" +[[step]] +name = "gate" +after = ["seed"] +type = "vote" +voters = ["a", "b", "c"] +vote_rule = "majority" +on_fail = "waiting-human" +threshold = { "fix-loop" = "count>=2(vote == approve-with-concerns)" } +[[step]] +name = "fix" +executor = "x" +emits = "findings" +loop = true +after_loop = "gate" +inputs = ["gate.vote-record"] +[[step]] +name = "report" +after = ["gate"] +executor = "y" +emits = "record" +inputs = ["gate.vote-record"] +` + if _, err := Load([]byte(src)); err != nil { + t.Fatalf("a concern-routing vote workflow failed to register: %v", err) + } +} + +// TestV22ConnectiveGrammar pins the shape V22 admits after DKT-548: clauses +// joined by `and` throughout or by `or` throughout, and nothing else. +// +// It goes through Load — the real register path — rather than calling +// validateWhen, because the point of the rule is what an operator's TOML is +// allowed to say. The rejection rows assert the RULE ID too: a mixed predicate +// refused as V22 tells the author which grammar bit them; refused as a parse +// error it would not. +func TestV22ConnectiveGrammar(t *testing.T) { + cases := []struct { + name string + when string + wantErr bool + }{ + {name: "single clause", when: "kind == bug"}, + {name: "and throughout", when: "kind == bug and labels contains backend"}, + { + name: "three ands", + when: "kind == bug and labels contains backend and labels != docs-only", + }, + // The motivating case: DKT-548's two byte-identical author lanes existed + // only because this string could not be written. + { + name: "or throughout", + when: "labels contains security-load-bearing or labels contains security", + }, + { + name: "three ors over both fields", + when: "kind == bug or kind == chore or labels contains security", + }, + {name: "quoted values across a disjunction", when: `kind == "bug" or labels == "urgent"`}, + {name: "extra whitespace around the connective", when: "kind == bug or labels contains x"}, + + // `contains-any` (DKT-550): a set-membership CLAUSE, so it registers + // alone, beside an ordinary `and` clause, and beside an `or` one. + {name: "contains-any alone", when: "labels contains-any (security-change, security)"}, + {name: "contains-any single element", when: "labels contains-any (security)"}, + {name: "contains-any three elements", when: "labels contains-any (a, b, c)"}, + {name: "contains-any without inner spaces", when: "labels contains-any(a,b)"}, + {name: "contains-any with generous whitespace", when: "labels contains-any ( a , b )"}, + {name: "contains-any quoted elements", when: `labels contains-any ("a", "b")`}, + // DKT-550's motivating predicate, as ONE homogeneous-`and` clause list. + { + name: "contains-any beside an and clause", + when: "kind == doc:tdd and labels contains-any (security-change, security)", + }, + { + name: "contains-any beside an or clause", + when: "labels contains-any (a, b) or kind == bug", + }, + + // `contains_any [...]` (DKT-1000): the same clause under the underscore + // spelling and the bracketed list. Both are accepted alongside the + // registered `contains-any (…)` form, and the spelling and the + // delimiter are independent — all four combinations register. + {name: "contains_any bracketed", when: "labels contains_any [security-change, security]"}, + {name: "contains_any single element", when: "labels contains_any [security]"}, + {name: "contains_any three elements", when: "labels contains_any [a, b, c]"}, + {name: "contains_any without inner spaces", when: "labels contains_any[a,b]"}, + {name: "contains_any with generous whitespace", when: "labels contains_any [ a , b ]"}, + {name: "contains_any quoted elements", when: `labels contains_any ["a", "b"]`}, + {name: "contains_any with parens", when: "labels contains_any (a, b)"}, + {name: "contains-any with brackets", when: "labels contains-any [a, b]"}, + // DKT-1000's motivating predicate, verbatim. + { + name: "contains_any beside an and clause", + when: "labels contains doc:tdd and labels contains_any " + + "[security-change, security-load-bearing, security]", + }, + { + name: "contains_any beside an or clause", + when: "labels contains_any [a, b] or kind == bug", + }, + + // The delimiters must PAIR. RE2 has no backreference, so the two + // spellings are two branches — a mismatched pair that registered would + // mean the closing delimiter is decoration. + {name: "contains_any bracket opened, paren closed", when: "labels contains_any [a, b)", wantErr: true}, + {name: "contains_any paren opened, bracket closed", when: "labels contains_any (a, b]", wantErr: true}, + // The bracketed form inherits every list-shape refusal of the paren one. + {name: "contains_any empty list", when: "labels contains_any []", wantErr: true}, + {name: "contains_any unclosed list", when: "labels contains_any [a, b", wantErr: true}, + {name: "contains_any unopened list", when: "labels contains_any a, b]", wantErr: true}, + {name: "contains_any bare value, no brackets", when: "labels contains_any a", wantErr: true}, + {name: "contains_any trailing comma", when: "labels contains_any [a, b,]", wantErr: true}, + {name: "contains_any leading comma", when: "labels contains_any [, a]", wantErr: true}, + {name: "contains_any doubled comma", when: "labels contains_any [a,, b]", wantErr: true}, + {name: "contains_any nested brackets", when: "labels contains_any [a, [b]]", wantErr: true}, + {name: "contains_any element with whitespace", when: "labels contains_any [a b]", wantErr: true}, + // `contains_any` is labels-only, exactly as `contains-any` is: a scalar + // has no set form, and the field vocabulary is still kind/labels only. + {name: "contains_any over kind", when: "kind contains_any [bug, task]", wantErr: true}, + {name: "contains_any over a field outside the grammar", + when: "assignee contains_any [a, b]", wantErr: true}, + // A near-miss spelling is not the operator. It must be refused rather + // than falling through to `labels contains "_anyx"`-style readings. + {name: "contains_anyx is not the operator", when: "labels contains_anyx [a, b]", wantErr: true}, + {name: "containsany is not the operator", when: "labels containsany [a, b]", wantErr: true}, + + // Malformed list syntax. Each is refused as V22 rather than admitted + // and read as something else — an unclosed or empty list that + // registered would evaluate false forever and silently skip the lane. + {name: "contains-any empty list", when: "labels contains-any ()", wantErr: true}, + {name: "contains-any unclosed list", when: "labels contains-any (a, b", wantErr: true}, + {name: "contains-any unopened list", when: "labels contains-any a, b)", wantErr: true}, + {name: "contains-any bare value, no parens", when: "labels contains-any a", wantErr: true}, + {name: "contains-any trailing comma", when: "labels contains-any (a, b,)", wantErr: true}, + {name: "contains-any leading comma", when: "labels contains-any (, a)", wantErr: true}, + {name: "contains-any doubled comma", when: "labels contains-any (a,, b)", wantErr: true}, + {name: "contains-any nested parens", when: "labels contains-any (a, (b))", wantErr: true}, + // `kind` is one value, so it has no set form — membership over it is + // `==`, and admitting `kind contains-any` would invite a list read + // against a scalar. + {name: "contains-any over kind", when: "kind contains-any (bug, task)", wantErr: true}, + {name: "contains-any over a field outside the grammar", + when: "assignee contains-any (a, b)", wantErr: true}, + + {name: "and then or", when: "kind == bug and labels contains a or labels contains b", wantErr: true}, + {name: "or then and", when: "labels contains a or labels contains b and kind == bug", wantErr: true}, + {name: "or over a field outside the grammar", when: "kind == bug or assignee == someone", wantErr: true}, + {name: "dangling connective", when: "kind == bug or", wantErr: true}, + {name: "doubled connective", when: "kind == bug or or labels contains a", wantErr: true}, + // A literal that merely CONTAINS a connective is not a split point: the + // connective is a whitespace-delimited token, so this stays one clause + // and validates. + {name: "label whose name embeds a connective", when: "labels contains and-then"}, + {name: "value that is a bare connective is a clause, not a separator", when: "labels contains or"}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + src := fmt.Sprintf( + "\n[pipeline]\nname = \"w\"\nversion = 1\n"+ + "[[step]]\nname = \"a\"\nexecutor = \"x\"\nemits = \"k\"\nwhen = %q\n", + tc.when) + + _, err := Load([]byte(src)) + if !tc.wantErr { + testsupport.Must(t, err, "when = %q should register: %v", tc.when, err) + return + } + + if err == nil { + t.Fatalf("when = %q registered; V22 must refuse it", tc.when) + } + we, ok := err.(*Error) + if !ok { + t.Fatalf("when = %q: error is %T, want *workflow.Error: %v", tc.when, err, err) + } + if we.Rule != "V22" { + t.Errorf("when = %q refused as %s, want V22: %v", tc.when, we.Rule, err) + } + }) + } +} + +// TestV22AndWhenHoldsShareOneGrammar is the single-regex discipline as an +// assertion: every predicate V22 admits must be evaluable, and the evaluator +// must not accept a spelling the validator refuses. +// +// The direction that matters is the second one. A predicate that evaluates but +// cannot register is merely unreachable; one that registers but evaluates as +// something else routes work through a lane nobody wrote — which is exactly +// what a second spelling of the connective grammar would eventually produce. +func TestV22AndWhenHoldsShareOneGrammar(t *testing.T) { + subject := Subject{Kind: "bug", Labels: []string{"security"}} + + for _, when := range []string{ + "kind == bug", + "kind == bug and labels contains security", + "labels contains security-load-bearing or labels contains security", + "kind == task or labels contains security", + "labels contains and-then", + "labels contains-any (security-change, security)", + "kind == doc:tdd and labels contains-any (security-change, security)", + "labels contains-any (a, b) or kind == bug", + "labels contains_any [security-change, security-load-bearing, security]", + "labels contains doc:tdd and labels contains_any [security-change, security]", + "labels contains_any [a, b] or kind == bug", + `labels contains_any ["a", "b"]`, + } { + src := fmt.Sprintf( + "\n[pipeline]\nname = \"w\"\nversion = 1\n"+ + "[[step]]\nname = \"a\"\nexecutor = \"x\"\nemits = \"k\"\nwhen = %q\n", when) + if _, err := Load([]byte(src)); err != nil { + t.Errorf("V22 refused %q, which WhenHolds evaluates: %v", when, err) + continue + } + // Evaluability, not the value: every clause parses, so no clause falls + // through whenClauseHolds's unparseable branch. + clauses, _, mixed := splitWhen(when) + if mixed { + t.Errorf("%q registered but splitWhen calls it mixed", when) + } + for _, clause := range clauses { + if !whenShape.MatchString(clause) { + t.Errorf("%q registered but clause %q does not match whenShape", when, clause) + } + } + _ = WhenHolds(when, subject) + } + + // The other direction: a mix evaluates FALSE and never registers, so the two + // halves agree that it is not a predicate. + const mixed = "kind == bug and labels contains security or labels contains x" + if WhenHolds(mixed, subject) { + t.Errorf("WhenHolds(%q) = true; a mixed predicate V22 refuses must not hold", mixed) + } +} diff --git a/scripts/qa.sh b/scripts/qa.sh index 8a85580a..16b95cf9 100755 --- a/scripts/qa.sh +++ b/scripts/qa.sh @@ -178,9 +178,9 @@ fi # surface only, and is skipped when a single section was requested, since it is # a whole-repo check rather than a section's. # -# copy-verify, render-verify, and the CI-wiring gate-baseref-regression / -# gate-coverage-check run here for the same reason: each is a whole-repo, -# diff-independent check, not a section's. +# copy-verify, render-verify, and the CI-wiring gate-baseref-regression run +# here for the same reason: each is a whole-repo, diff-independent check, not +# a section's. # # ONE BLOCK SHAPE, defined once as `run_gate` in qa/helpers.sh. This # section used to hold five copies of the same twelve lines, which had already @@ -216,15 +216,11 @@ if [ -z "$SECTION" ]; then run_gate "RV" "RV_render_coverage" "render-verify gate" \ "render-verify.sh" 'MISSING|raw ANSI escape' - # gate-baseref-regression and gate-coverage-check pin the CI wiring - # itself (findings C1/C2/C5 and C3): the former runs the base-ref mode of - # secret-scan.sh/self-hygiene.sh through their real entry points, the - # latter checks ci.yaml's GATE COVERAGE block against the actual directory. + # gate-baseref-regression pins the CI wiring itself (findings C1/C2/C5): + # it runs the base-ref mode of secret-scan.sh/self-hygiene.sh through their + # real entry points. run_gate "BR" "BR_baseref_mode" "gate-baseref-regression" \ "gate-baseref-regression.sh" 'FAIL' - - run_gate "GC" "GC_ci_coverage_block" "gate-coverage-check" \ - "gate-coverage-check.sh" 'FAIL' fi # --- Report ------------------------------------------------------------------ diff --git a/scripts/qa/fixtures/context/step-10.golden b/scripts/qa/fixtures/context/step-10.golden index fbb3380d..6cdab97d 100644 --- a/scripts/qa/fixtures/context/step-10.golden +++ b/scripts/qa/fixtures/context/step-10.golden @@ -25,6 +25,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "commit-author", "executor": "commit-author", "expected_cost": 0.1, diff --git a/scripts/qa/fixtures/context/step-2.golden b/scripts/qa/fixtures/context/step-2.golden index c05b2392..5600e8d6 100644 --- a/scripts/qa/fixtures/context/step-2.golden +++ b/scripts/qa/fixtures/context/step-2.golden @@ -32,6 +32,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "judge-correctness", "executor": "judge-correctness", "expected_cost": 0.6, diff --git a/scripts/qa/fixtures/context/step-3.golden b/scripts/qa/fixtures/context/step-3.golden index 7019f3eb..1d3de0f4 100644 --- a/scripts/qa/fixtures/context/step-3.golden +++ b/scripts/qa/fixtures/context/step-3.golden @@ -32,6 +32,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "judge-architecture", "executor": "judge-architecture", "expected_cost": 0.6, diff --git a/scripts/qa/fixtures/context/step-4.golden b/scripts/qa/fixtures/context/step-4.golden index 9577b02b..bfe5e17e 100644 --- a/scripts/qa/fixtures/context/step-4.golden +++ b/scripts/qa/fixtures/context/step-4.golden @@ -32,6 +32,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "judge-simplicity", "executor": "judge-simplicity", "expected_cost": 0.6, diff --git a/scripts/qa/fixtures/context/step-5.golden b/scripts/qa/fixtures/context/step-5.golden index da0124cb..b52a131b 100644 --- a/scripts/qa/fixtures/context/step-5.golden +++ b/scripts/qa/fixtures/context/step-5.golden @@ -32,6 +32,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "judge-testing", "executor": "judge-testing", "expected_cost": 0.6, diff --git a/scripts/qa/fixtures/context/step-6.golden b/scripts/qa/fixtures/context/step-6.golden index 833bce93..4869e2b5 100644 --- a/scripts/qa/fixtures/context/step-6.golden +++ b/scripts/qa/fixtures/context/step-6.golden @@ -25,6 +25,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "synthesize-findings", "executor": "synthesize-findings", "expected_cost": 0.4, diff --git a/scripts/qa/fixtures/context/step-7.golden b/scripts/qa/fixtures/context/step-7.golden index 5a0039b2..28c1ec69 100644 --- a/scripts/qa/fixtures/context/step-7.golden +++ b/scripts/qa/fixtures/context/step-7.golden @@ -25,6 +25,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "expected_cost": 0, "instance": "reconcile@0", "issue": "DKT-1", diff --git a/scripts/qa/fixtures/context/step-8.golden b/scripts/qa/fixtures/context/step-8.golden index f4c41394..53061003 100644 --- a/scripts/qa/fixtures/context/step-8.golden +++ b/scripts/qa/fixtures/context/step-8.golden @@ -38,6 +38,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "class": "verify-ac", "executor": "verify-ac", "expected_cost": 0.5, diff --git a/scripts/qa/fixtures/context/step-9.golden b/scripts/qa/fixtures/context/step-9.golden index 596a303c..5171fc40 100644 --- a/scripts/qa/fixtures/context/step-9.golden +++ b/scripts/qa/fixtures/context/step-9.golden @@ -25,6 +25,7 @@ ], "step": { "attempt": 0, + "blocked_reason": "an `after` predecessor is not done", "expected_cost": 0, "instance": "commit-gate@0", "issue": "DKT-1", diff --git a/scripts/qa/gate-coverage-check.sh b/scripts/qa/gate-coverage-check.sh deleted file mode 100755 index 7c6fcb15..00000000 --- a/scripts/qa/gate-coverage-check.sh +++ /dev/null @@ -1,155 +0,0 @@ -#!/usr/bin/env bash -# -# gate-coverage-check — the GATE COVERAGE comment block at the top of -# .github/workflows/ci.yaml names, for every script under -# scripts/qa/, whether CI runs it and why not when it doesn't. That block is -# hand-maintained prose; nothing checked it against the directory it -# describes. This script does: every non-test script under -# scripts/qa/ must be named somewhere in the block, and every script named in -# the block must still exist. A script added to scripts/qa/ and forgotten -# here now fails CI instead of silently going unaccounted for. -# -# It checks TWO different claims, because the block makes two: -# -# ACCOUNTING (the original check) — the set of script NAMES in the block -# matches the set of files on disk. Answers "is every script spoken for?" -# -# INVOCATION — where the block says a script "RUNS HERE, by name, -# in ``", that job must exist in the `jobs:` mapping and must actually -# invoke that script. Answers "is the claim true?" -# -# Accounting alone was not enough, and the gap was not theoretical: with the -# entire `repo-gates` job deleted from ci.yaml, this script still printed -# `ok (18 scripts accounted for)` and exited 0, while the block went on -# asserting "RUNS HERE, by name, in `repo-gates`: secret-scan.sh, -# self-hygiene.sh". Only the names were pinned, never the invocations — so -# ci.yaml:12-13's promise that the listing "cannot drift silently" held for -# half of what the listing says. -# -# The invocation half derives BOTH of its inputs from the file: the job name -# and the script names come from parsing the block's own RUNS HERE clause, -# and the truth comes from the `jobs:` mapping. Neither is a hand-maintained -# list here, which is AC3 — a second list would just be a third thing -# to drift. -# -# Usage: -# ./scripts/qa/gate-coverage-check.sh - -set -euo pipefail - -REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" -CI_YAML="$REPO_ROOT/.github/workflows/ci.yaml" -FAIL=0 - -fail() { - echo "gate-coverage-check FAIL: $1" >&2 - FAIL=1 -} - -if [ ! -f "$CI_YAML" ]; then - fail "$CI_YAML does not exist" - exit 1 -fi - -# The block runs from the `name: ci` line to the `on:` trigger block — the -# comment lines in between are the block's own boundary, not a hand-picked -# line range that would silently stop tracking it if the file grew above it. -BLOCK=$(awk '/^on:/{exit} {print}' "$CI_YAML") - -# Every non-test script actually on disk. -DISK_SCRIPTS=$(cd "$REPO_ROOT/scripts/qa" && ls *.sh | grep -v '^test_' | sort) - -# Every `.sh` filename token the block itself names. `qa.sh` (singular, -# `scripts/qa.sh`) is the runner the block mentions as the thing INVOKING -# several gates — it lives one directory above scripts/qa/ and is not itself -# a member of the directory this check accounts for. -# -# THE `.sh` MUST END THE TOKEN. Without a trailing boundary, `[A-Za-z0-9_-]+\.sh` -# matches the PREFIX of any longer word: the prose `base.sha` in this block -# scraped as a phantom `base.sh`, and the check then reported "the block names -# a script that no longer exists" for a file nobody had ever written. Any -# `.sh?????` word in the surrounding explanation would do the same. `[^A-Za-z0-9_.-]` -# after the extension (or end of line) confines the match to a real filename -# token. -# -# Comment lines only, additionally: the extent above runs to the `on:` key, so -# any YAML between `name: ci` and `on:` is inside the block too, and the -# listing this check verifies lives entirely in the prose. -BLOCK_SCRIPTS=$(printf '%s\n' "$BLOCK" | grep '^[[:space:]]*#' | - grep -oE '[A-Za-z0-9_-]+\.sh([^A-Za-z0-9_.-]|$)' | - sed 's/[^A-Za-z0-9_.-]$//' | grep -v '^qa\.sh$' | sort -u) - -MISSING_FROM_BLOCK=$(comm -23 <(printf '%s\n' "$DISK_SCRIPTS") <(printf '%s\n' "$BLOCK_SCRIPTS")) -STALE_IN_BLOCK=$(comm -13 <(printf '%s\n' "$DISK_SCRIPTS") <(printf '%s\n' "$BLOCK_SCRIPTS")) - -if [ -n "$MISSING_FROM_BLOCK" ]; then - fail "scripts/qa/ has a script the GATE COVERAGE block never names: $(printf '%s ' $MISSING_FROM_BLOCK)" -fi - -if [ -n "$STALE_IN_BLOCK" ]; then - fail "the GATE COVERAGE block names a script that no longer exists: $(printf '%s ' $STALE_IN_BLOCK)" -fi - -# --- INVOCATION ------------------------------------------------------------- -# -# The block's RUNS HERE clause names a job and the scripts it claims that job -# invokes. Both halves are read out of the file rather than restated here. -# -# The clause spans several lines: a `RUNS HERE, by name, in \`\`` opener, -# then parenthetical prose, then an indented list of script names. So the job -# name is taken from the opener and the script names from every line between -# that opener and the next `RUNS`/`EXCLUDED` heading — the block's own -# structure, not a line count. - -RUNS_HERE_JOB=$(printf '%s\n' "$BLOCK" | - sed -n 's/.*RUNS HERE, by name, in `\([A-Za-z0-9_-]*\)`.*/\1/p' | head -1) - -if [ -z "$RUNS_HERE_JOB" ]; then - # The clause is the thing this half verifies. If it is gone or reworded, - # this check has silently stopped checking anything — which is the exact - # failure mode this check is about, so it fails rather than passing vacuously. - fail "the GATE COVERAGE block has no \`RUNS HERE, by name, in \`\`\` clause; the invocation check cannot run" -else - # The scripts the clause claims that job runs: the lines from the opener up - # to the next section heading. - RUNS_HERE_SCRIPTS=$(printf '%s\n' "$BLOCK" | - awk '/RUNS HERE, by name, in/{inblock=1} - inblock && /RUNS ELSEWHERE|EXCLUDED/{inblock=0} - inblock{print}' | - grep -oE '[A-Za-z0-9_-]+\.sh' | sort -u) - - # The job's own text, from the `jobs:` mapping. A job body runs from its - # ` :` key to the next key at the same indent — the YAML's own - # structure. This is a targeted extract, not a YAML parse: the question is - # only "does this job exist, and does its text invoke this script", and - # both are answerable from the raw lines without adding a yq dependency to - # a gate that has none. - JOB_BODY=$(awk -v job=" $RUNS_HERE_JOB:" ' - $0 == job {injob=1; next} - injob && /^ [A-Za-z0-9_-]+:/ {exit} - injob {print} - ' "$CI_YAML") - - if [ -z "$JOB_BODY" ]; then - fail "the GATE COVERAGE block says scripts run in the \`$RUNS_HERE_JOB\` job, but ci.yaml's jobs: mapping has no such job" - else - # A job body that is entirely comments invokes nothing. Steps are what - # run, so the check is against the step text, with comment lines dropped - # — otherwise the `repo-gates` job comment naming the scripts would - # satisfy the check on its own, and a job stripped to its comments would - # still pass. - JOB_STEPS=$(printf '%s\n' "$JOB_BODY" | sed 's/[[:space:]]*#.*//') - - for script in $RUNS_HERE_SCRIPTS; do - if ! printf '%s\n' "$JOB_STEPS" | grep -qF "$script"; then - fail "the GATE COVERAGE block says \`$RUNS_HERE_JOB\` runs $script, but that job never invokes it" - fi - done - fi -fi - -if [ "$FAIL" -eq 0 ]; then - echo "gate-coverage-check: ok ($(printf '%s\n' "$DISK_SCRIPTS" | wc -l | tr -d ' ') scripts accounted for)" -fi - -exit "$FAIL" diff --git a/scripts/qa/test_zg_workflow.sh b/scripts/qa/test_zg_workflow.sh index 4a1d180f..829dc83e 100644 --- a/scripts/qa/test_zg_workflow.sh +++ b/scripts/qa/test_zg_workflow.sh @@ -716,6 +716,12 @@ SQL fi # `run status` renders the rollup and WRITES NOTHING. + # + # Nine `pending`, not ten: `implement@0` has no `after` predecessor, ample + # budget headroom (cap 25 against its 1.5 declared cost), and no other R1-R7 + # clause holds it back, so it reads `ready` the instant the run activates — + # the correct effective status, per EffectiveStatusCounts. The rollup counts + # by EFFECTIVE status, so a leaf step's readiness moves it out of `pending`. local RUN_STATE_BEFORE RUN_STATE_AFTER RUN_STATE_BEFORE=$(sqlite3 "$ZG_DB" \ "SELECT group_concat(id||status||row_version||updated_at_ms) FROM runs;") @@ -723,7 +729,7 @@ SQL assert_exit "ZG" "ZG9_status_exit" 0 assert_json "ZG" "ZG9_status_issues" ".data.issues" "2" assert_json "ZG" "ZG9_status_step_rollup" \ - '.data.steps | map(select(.status == "pending")) | .[0].count' "10" + '.data.steps | map(select(.status == "pending")) | .[0].count' "9" RUN_STATE_AFTER=$(sqlite3 "$ZG_DB" \ "SELECT group_concat(id||status||row_version||updated_at_ms) FROM runs;") check_cond "ZG" "ZG9_status_writes_nothing" "`run status` mutated the run row" [ "$RUN_STATE_BEFORE" = "$RUN_STATE_AFTER" ] @@ -3385,7 +3391,13 @@ TOML # coverage of their own. This is what a future retire verb must break # LOUDLY, matching the Go side's literal `wantCandidates` in # TestRenamePlusBumpRefusesActivation. - assert_stdout_contains "ZG" "ZG32_names_joined" "zg-gone@1, zg-gone-renamed@2" + # + # zg-gone@1 carries the orphan annotation (orphanAnnotation, + # internal/engine/orphan_registration.go): its source file is the one ZG31 + # deleted above, so the candidate list names it as an orphaned registration, + # same as the Go side's `wedgeCandidatesOrphaned`. + assert_stdout_contains "ZG" "ZG32_names_joined" \ + "zg-gone@1 (no source on disk — orphaned registration, deprecation candidate), zg-gone-renamed@2" # The BRANCH DISCRIMINATOR. The joined candidate list above is # rendered by refList in BOTH refusal branches and in the same order, so it # does not distinguish them. This literal is the only text that does; a diff --git a/skills/docket/SKILL.md b/skills/docket/SKILL.md deleted file mode 100644 index 1f0cc1ba..00000000 --- a/skills/docket/SKILL.md +++ /dev/null @@ -1,4293 +0,0 @@ ---- -name: docket -description: > - Comprehensive reference for using the Docket CLI (`docket`), a local-first, - SQLite-backed issue tracker. Use this skill whenever the user asks to - create, edit, list, move, close, or reopen issues; attach files, add - comments, apply labels, or link relations between issues; generate an - execution plan or find work-ready issues; create or cast a consensus vote - ("run a vote", "start a proposal"); author, edit, or link a document; define - or register a workflow ("set up a workflow", "register a pipeline"); watch - live-updating output; export or import a Docket database; or any request - to "use docket", "track this in docket", "create a docket issue", "check - docket status", "run docket plan/next", "show the docket board", etc. ---- - -# Docket CLI Skill - -Docket (`docket`) is a local-first, SQLite-backed issue tracker driven -entirely through a single CLI binary. There is no server and no network -call — all state lives in one SQLite `issues.db` file inside a **store**, -resolved via `internal/config` in this order: - -1. **`DOCKET_PATH`** — taken as the store directory, normalized to absolute. -2. **A repo-local `.docket/` store** containing `issues.db`, discovered by - walking from the cwd up to the git worktree toplevel (just the cwd - outside a repository). The legacy per-repo layout keeps working wherever - it already exists. -3. **The shared per-user store, `~/.docket`** — the default. Every - repository resolving here is a **project** row in one database (see - `docket project` below); issue ids are store-wide numbers. - -This skill teaches an agent how to drive `docket` end to end: issue CRUD -and lifecycle, file attachments, comments, labels, relations, dependency -graphs, execution planning, consensus voting, docs, watch mode, and -export/import. - -Every command supports **two output modes**: human-readable (default, -colorized via lipgloss when the terminal supports it) and machine-readable -JSON (`--json`). **Agents should always pass `--json`** for reliable -parsing — the examples below show both. - -## Quick Start - -```bash -docket init # initialize the resolved store (~/.docket by default) -docket init --local # opt out: create a repo-local .docket store in the cwd -docket issue create -t "Fix login bug" --json # create an issue, get its ID back -docket issue list --json # list open issues -docket issue show DKT-1 --json # show full detail incl. comments/activity -docket next --json # what's ready to work on right now? -``` - -Issue IDs are formatted `DKT-` (e.g. `DKT-42`), document IDs `DOC-`, -and proposal (vote) IDs `DKT-V` — all three accept either the bare number -or the formatted string as CLI arguments (`model.ParseID`, `ParseDocID`, -`ParseProposalID` all strip the prefix case-insensitively). The issue prefix -is a **per-project display setting** (`docket project set-prefix`): another -project's issues may render `VOR-42`, but the number is the store-wide -identity and `DKT-` always parses whatever the prefix. - -## Global Flags & Output Contract - -Defined once on `rootCmd` in `internal/cli/root.go` and inherited by every -subcommand: - -| Flag | Shorthand | Type | Default | Behavior | -|---|---|---|---|---| -| `--json` | — | string | `""` | Switch to machine-readable JSON envelope on stdout. Bare `--json` selects v1; `--json=v2` selects the uniform envelope. See below. | -| `--quiet` | `-q` | bool | `false` | Suppress non-essential human-mode info/warning lines on stderr. No effect in `--json` mode (already silent). | -| `--watch` | `-w` | bool | `false` | Re-run the command on an interval and refresh output. Accepted only on the exact allowlist below; every other command — read-only or not — rejects it with a `VALIDATION_ERROR`. | -| `--interval` | — | duration | `2s` | Poll interval for `--watch` **and** `events list --follow`. Minimum `500ms`; anything lower is a `VALIDATION_ERROR`. | - -### `--watch` eligibility - -`--watch`/`-w` and `--interval` are hidden from the `--help` of every command -NOT in this allowlist (hidden at help-render time — the flags are shared -persistent globals, so they exist on every command even when help omits -them), defined in `internal/cli/watch_commands.go`: - -``` -docket board -docket issue list -docket issue show -docket issue log -docket issue graph -docket issue comment list -docket doc list -docket doc show -docket doc comment list -docket next -docket plan -docket stats -docket config -docket vote list -docket vote show -docket vote result -docket events list -docket step list -``` - -`docket events list` is on the list because its `--follow` polls on the same -`--interval`; the rest is the tracker surface. - -Attempting `--watch` on any command off the list — read-only (`docket -project list`) or write alike — fails with a `VALIDATION_ERROR` whose -message enumerates the live allowlist: `--watch is limited to: docket board, -docket config, ...` (generated from the allowlist by -`watchRejectionMessage()`). The message's own enumeration is authoritative -if this copy ever drifts. - -### JSON envelope shape - -All JSON output (`internal/output/json.go`) is a single-line JSON object -written to stdout via `json.Encoder` (HTML-escaping disabled). - -Success: -```json -{"ok": true, "data": { ... }, "message": "Created DKT-1: Fix login bug"} -``` -`message` is `omitempty` — it is present on success responses but callers -should not depend on it being non-empty for every command. - -Error: -```json -{"ok": false, "error": "issue DKT-99 not found", "code": "NOT_FOUND"} -``` - -### `--json` values (v1 vs v2) - -`--json` is a string flag with `NoOptDefVal = "v1"`, so a bare `--json` behaves -exactly as it did when the flag was a boolean. - -| Value | Mode | Notes | -|---|---|---| -| (flag absent) | human | | -| `--json` (bare) | JSON v1 | byte-identical to the pre-v2 output | -| `--json=v1` | JSON v1 | explicit form | -| `--json=true`, `--json=1` | JSON v1 | retained from the boolean-flag era | -| `--json=v2` | JSON v2 | uniform envelope, see below | -| `--json=false`, `--json=0` | human | retained from the boolean-flag era | -| anything else | `VALIDATION_ERROR` (exit 3) | e.g. `--json=v3` | - -**v1 is frozen, with the recorded amendments below.** New response data appears -under v2 only, which is what makes `--json` safe for existing scripts. Every -amendment is additive — no key has ever changed meaning or disappeared — so a -script selecting a key it already read keeps reading exactly that. - -| Amendment | Payload | Shape | -|---|---|---| -| DKT-55 | `issue show`, `issue list` | `scope` appears **when the issue declares one, and only then** — no declared scope emits no key and stays byte-identical to the pre-scope shape; a declared-but-empty scope emits `[]` | -| DKT-245 | `issue show`, `issue list` | `resolution` appears **when a routing has set one**, so an abandoned issue stops being indistinguishable from a finished one | -| DKT-404 | `issue show` | `run_disposition` appears **when a run abandoned its work on the issue** | -| DKT-452 | every issue payload | `issue` mirrors `id`, **unconditionally** — see below | - -### Primary-key naming (`id` vs the noun) - -Every verb keys its primary entity by the entity's **noun**: `run status` keys -`run`, `step show` keys `step`, `dispatch open` keys `dispatch`, `issue claim` / -`issue release` / `issue heartbeat` key `issue`. The issue READ verbs were the -exception — they keyed `id` and nothing else — so a caller that had just parsed -a run or a step reached for `.data.issue` on `issue show`, got `null`, and had -to dump the key set to recover (DKT-452). - -Both keys now carry the id, on **v1 and v2 alike**, wherever an issue is -serialized — `issue show` (including its nested `sub_issues`), `issue list` -rows, and the issue returned by `issue create` / `edit` / `close` / `reopen` / -`move`: - -```console -$ docket issue show DKT-1 --json | jq -r '.data.id, .data.issue' -DKT-1 -DKT-1 -$ docket issue list --json | jq -c '.data.issues[] | {id, issue}' -{"id":"DKT-1","issue":"DKT-1"} -``` - -`id` is the older spelling and is not deprecated; `issue` is the one that -matches every other verb. Prefer whichever your surrounding code already uses — -they cannot disagree, since both are written from the same value. - -This is the one amendment that is NOT conditional, so v1 issue payloads are no -longer byte-identical to their pre-DKT-452 bytes. That is deliberate: a -conditional alias would be absent in precisely the case the alias exists to -serve, and an *added* key cannot break a reader that selects `.data.id`. - -Under **`--json=v2`**, list commands return a uniform envelope instead of their -per-command key (`issues`, `docs`, `proposals`, `entries`): - -```json -{"ok": true, "data": {"items": [...], "total": 42, "truncated": true}} -``` - -- `total` is the number of matching records **before** `--limit` is applied. -- `truncated` is `true` when `--limit` dropped records. - -This closes a silent-drop bug: under v1, `docket next --limit 10` and -`docket issue log --limit 10` report `"total": 10` whether 10 or 10 000 records -matched. Under v2 both report the true total and set `truncated`. - -Commands returning a single entity (`issue show`, `issue create`, …) are not -wrapped — v2 returns the same object as v1, plus a `version` field (see -Optimistic concurrency below). - -Under v2, a **negative** `--limit` is a `VALIDATION_ERROR` on every list verb. -Under v1 the legacy behaviors are preserved unchanged (`issue list` and `next` -treat it as unlimited; `issue log` clamps it to 1). - -### Error codes & exit codes - -Defined in `internal/output/json.go`. The process exit code always matches -the table below, in both JSON and human mode (`ExitCodeForError`): - -| `code` | Exit code | Meaning | -|---|---|---| -| `GENERAL_ERROR` | 1 | Unclassified failure (DB error, I/O error, etc.) | -| `NOT_FOUND` | 2 | Referenced issue/doc/proposal/label/relation does not exist | -| `VALIDATION_ERROR` | 3 | Bad input: invalid enum value, missing required flag, mutually exclusive flags, non-interactive environment without required flags, invalid `--json` value, negative `--limit` under v2, `--if-version < 1` | -| `CONFLICT` | 4 | State conflict: duplicate relation, cycle detected, already-voted, non-empty DB on import without `--merge`/`--replace`, `--if-version` mismatch, a dispatch already open for the run, `next --run` while a dispatch is open or discrepancies exist, `dispatch verify` byte mismatch, `dispatch close` over an unreconciled discrepancy, `dispatch backfill-usage` repeating a `(step, attempt, unit)` already recorded, any dispatch verb finding no manifest open, `step annotate` on a step that has not finished, or `issue move --project` on an issue a run holds | -| `AUTH_ERROR` | 5 | The supplied capability token does not hold this lease (or the entity is unclaimed) | -| `STALE_LEASE` | 6 | The token is correct but the lease has expired — claim again | -| `TIMEOUT` | 7 | Reserved — no verb emits this yet | -| `UNTRUSTED` | 8 | Reserved — no verb emits this yet | -| `GONE` | 9 | `events list --since` names a cursor below the retained minimum: those events no longer exist | - -Codes 1–4 and their exit numbers are a frozen contract and are never -renumbered. Codes 5–8 are declared ahead of the verbs that will raise them, so -the taxonomy is fixed once rather than extended per feature. `GONE` **appends** -at 9 rather than taking exit 6 — that number is `STALE_LEASE`, which emits today, -and two codes sharing one exit would be indistinguishable to a script testing -`$?`. New codes only ever append. - -**`GONE` is reached by `docket events prune`, and by nothing else.** The engine -never deletes an event on its own — there is no retention sweep, no compaction at -`run done`, and no prune inside `next` — so a repo whose operator has never run -that verb retains the first event ever written and can never answer `GONE` to -anybody. Once something *has* been pruned, a cursor naming events below what -survives gets this code and a message naming the `seq` to resume from. - -Exit code `0` is success. Note `PersistentPreRunE` also returns `NOT_FOUND` -(exit 2) if the resolved store has no database yet — run `docket init` first. - -### Optimistic concurrency (`--if-version`) - -Every mutable entity carries a `version` counter that increments on each -mutation. Pass `--if-version N` to a mutating verb to apply the change only if -the entity is still at version `N`: - -```bash -V=$(docket issue show DKT-1 --json=v2 | jq -r '.data.version') -docket issue edit DKT-1 --json=v2 --if-version "$V" -s in-progress -``` - -- Version matches → the write applies and the version increments. -- Version differs → `CONFLICT` (exit 4) and **nothing is written**. -- Entity is missing → `NOT_FOUND` (exit 2), not `CONFLICT`. -- `--if-version` below 1 → `VALIDATION_ERROR` (exit 3); versions start at 1. - -Omitting `--if-version` preserves the previous last-writer-wins behavior. The -version still increments, so a concurrent CAS writer detects the change. - -Read the current version from `.data.version` under `--json=v2`; v1 payloads do -not carry the field. - -Verbs accepting `--if-version`: `issue edit`, `issue move`, `issue close`, -`issue reopen`. - -### Idempotency keys (`--idempotency-key`) - -Create verbs accept `--idempotency-key KEY`. Repeating a create with the same -key returns the **original** entity with exit 0 and creates nothing new — so a -retry after a dropped connection cannot duplicate work: - -```bash -docket issue create --json -t "Deploy checklist" --idempotency-key deploy-2026-08-02 -docket issue create --json -t "Deploy checklist" --idempotency-key deploy-2026-08-02 -# → same DKT-N both times, one issue created -``` - -Keys are scoped per verb, so the same key on `issue create` and `doc create` is -two independent records. An empty `--idempotency-key ""` is a -`VALIDATION_ERROR`. Without the flag, creates are never deduplicated. - -Verbs accepting `--idempotency-key`: `issue create`, `doc create`, -`vote create`, `issue comment add`, `doc comment add`. - -### Claims, leases, and capability tokens - -`docket issue claim` takes a lease on an issue and mints a **capability -token**. The claim is atomic: exactly one of any number of concurrent claimants -wins, and the losers get `CONFLICT` (exit 4). - -```bash -TOKEN=$(docket issue claim DKT-1 --owner ci-runner-7 --json | jq -r '.data.token') -DOCKET_TOKEN="$TOKEN" docket issue heartbeat DKT-1 # extend while working -DOCKET_TOKEN="$TOKEN" docket issue release DKT-1 # or close, when done -``` - -**The token is returned exactly once.** Only its hash is stored, so it cannot -be read back from the database — capture it from the claim response or claim -again. It never appears in `issue show`, `issue list`, or `issue log`. - -**Tokens pass via `DOCKET_TOKEN` or stdin, never argv.** There is no `--token` -flag on any verb: `ps` exposes argv to every user on a shared host. Pipe it -(`echo "$TOKEN" | docket issue heartbeat DKT-1`) or export it. - -Refusals: - -| Situation | Code | Exit | -|---|---|---| -| No token supplied to a token-requiring verb | `VALIDATION_ERROR` | 3 | -| Token does not hold the lease, or issue unclaimed | `AUTH_ERROR` | 5 | -| Token is correct but the lease expired | `STALE_LEASE` | 6 | -| Claiming an issue whose lease is live | `CONFLICT` | 4 | - -`AUTH_ERROR` covers both "wrong token" and "unclaimed" deliberately — a caller -holding no capability learns nothing about whether a lease exists. -`STALE_LEASE` is distinct because it means *re-claim*: the token was right, -time ran out. - -**Expiry is the liveness mechanism.** A lease that lapses without release -returns the issue to the unclaimed pool: the next claim simply wins, and -`attempt` records that a claim was made. No reaper runs — expiry is resolved by -the next claim — and **no read verb ever writes**. Reads report *effective* -status, so an expired lease shows `"live": false` the instant it lapses: - -```bash -docket issue show DKT-1 --json=v2 | jq '.data.lease' -# {"owner":"ci-runner-7","expires_ms":1754161200000,"attempt":1,"live":false} -``` - -`attempt` counts claims for all time — never decremented, never reset — so a -killed worker's claim and its successor's both appear. - -The `lease` object is **`--json=v2` only**, and absent entirely when the issue -is unclaimed. An issue that is never claimed behaves exactly as it did before -leases existed on every verb. - -`docket issue close` ends a live lease as a side effect. The holder may always -close; a non-holder closing a live-leased issue is refused `AUTH_ERROR`, so a -bystander cannot silently evict a working holder. Closing an **unclaimed** -issue needs no token. - -See `docs/spec/security.md` for the full token model, including what a lease -does and does not defend against. - -### Engine configuration (`docket config set|get`) - -Engine defaults live in the database and are read by the claim machinery: - -| Key | Type | Default | Meaning | -|---|---|---|---| -| `lease.ttl.default` | duration | `15m` | fallback lease TTL | -| `lease.ttl.` | duration | (falls back to default) | per-class lease TTL. **The class is a step's `class` field, which defaults to its `executor` name** — see below | -| `attempt.max` | int ≥ 1 | `3` | maximum claims per entity | -| `budget.default` | number ≥ 0 | `0` | default per-run budget cap; 0 is unlimited. Resolved at `run start` and stored on the run, so setting it later does not re-cap a run already started | -| `budget.unit` | unit name or `""` | `""` | which recorded usage unit the run cap counts. Empty (the default) means the cap rests on the declared-cost floor alone | -| `dispatch.ttl` | duration | `30m` | how long a dispatch manifest stays open before `next` auto-abandons it | -| `dispatch.grace` | duration | `15m` | how long a claimed step may go unrecorded before it counts as a dispatch discrepancy | -| `events.retain` | duration or `0` | `0` | how long events are protected from `events prune`. `0` (the default) retains **everything**, so prune deletes nothing until a policy is set | -| `context.warn_bytes` | int ≥ 0 | `65536` | context size that triggers a warning | -| `context.error_bytes` | int ≥ 0 | `131072` | context size that triggers an error | -| `vote.rule..threshold` | float in (0,1] | (unset) | approval threshold a `vote_rule` tallies at | -| `vote.rule..criticality` | low\|medium\|high\|critical | `medium` | the proposal's criticality | -| `vote.hold.rule` | rule name or `""` | `""` | vote rule a **materialized held step** is tallied under. Empty (the default) mints held steps as `human` for one operator to decide | -| `vote.hold.voters` | comma-separated names or `""` | `""` | who casts on a materialized held step. Empty (the default) mints held steps as `human` | - -**A class is whatever string your steps carry, and a step that declares no -`class` carries its `executor` name** (DKT-260). So `lease.ttl.read` binds -nothing unless your definitions actually say `class = "read"` — and in a corpus -where steps declare no class, the only strings that ever appear in that column -are executor names, which is not what "per-class lease TTL" suggests. The cost -of the mismatch is not cosmetic: every unforced reap of one epoch killed a -*healthy* step 15–30 minutes into its claim against a 15m default, and reap is -a database fence — it marks the claim dead and cannot stop a process that is -still running, so a wrongly-reaped write step means two processes editing one -tree. - -**Docket cannot supply `read` and `write` classes of its own.** The class name -is the workflow author's; core is deliberately blind to it (the headroom -mechanism keys on the declared `[limits] max`, never on a name, and a source -guard asserts core contains no `"write"` literal). Declare the classes you want -to configure. - -`docket config set lease.ttl.` **warns** when no registered workflow -declares that class, naming the ones that do. A warning and not a refusal — -configuring ahead of a workflow is ordinary — but a TTL that binds nothing is -otherwise discovered when a healthy step is reaped mid-run. - -```bash -docket config set lease.ttl.write 45m -docket config get lease.ttl.write --json # {"key":...,"value":"45m","source":"set"} -docket config get # every key, with its source -``` - -`source` is `set` or `default`, so "nobody configured this" is distinguishable -from "configured to the same value". An unknown key or a value of the wrong -type is a `VALIDATION_ERROR` (exit 3) at `set` time — a typo never silently -stores a key nothing reads. The class in `lease.ttl.` is an opaque -string; docket never interprets it. - -Under the shared store, config is **layered per project**: a read resolves the -current project's override, then the store-wide value, then the built-in -default (the first two both report `source: set`). `config set --global` -writes the store-wide default instead of this project's override, and -`config get --global` reads the store-wide values ignoring this project's -overrides. - -**Vote rules.** A workflow's `type="vote"` step names a rule rather than passing -flags, because a step cannot pass flags. A rule *exists* iff its `.threshold` is -set — criticality has a default, so it cannot be the existence test: - -```bash -docket config set vote.rule.majority.threshold 0.6 -docket config set vote.rule.majority.criticality high -``` - -`` is opaque, exactly as the lease class is. A step whose `vote_rule` -names an unregistered rule is refused at `workflow register` (V26), naming the -rule and listing the registered ones — a workflow that cannot possibly tally -should not register. Note that `required_voters` comes from the step's own -`voters` list, not from the rule: a rule is about *how strictly to tally*, the -step about *who casts*. - -**Held steps by tally.** A materialized `-held` step is the one row in a -run no author wrote, so it has no `[[step]]` table to carry `voters` and -`vote_rule`. `vote.hold.rule` and `vote.hold.voters` are where an instance says -them instead: - -```bash -docket config set vote.rule.panel.threshold 0.6 -docket config set vote.hold.rule panel -docket config set vote.hold.voters alice,bob,carol -``` - -**BOTH keys are required, and unset is a strict no-op:** with either missing, -holds are minted `human` exactly as they always were, and one operator approves -or rejects them. With both set they are minted `vote` and flow through the -ordinary vote lifecycle. The escalation is one-directional: - -- **pass** — the computed value stands, identical to `step approve` on the - cluster. The payload records `operator_resolved` and the aggregate resumes. -- **anything else** — the step **parks** at `waiting-human` and the question - passes to an operator, who answers it with `step approve` (including - `--value`) or `step reject`. A tally may confirm the engine's own computation - and may never overrule it, so a panel that could not agree cannot produce the - effect of an operator who declined. - -The MINTED KIND is what persists: config supplies the roster, the step row -supplies the question's type. Editing or clearing these keys mid-run changes who -casts on holds minted *after* the edit, never what an already-open question is. - -### Interactive forms - -Several write commands (`issue create`, `issue delete` with sub-issues, -`vote create`, `vote cast`, `doc create`, `doc delete`, `label delete`) -fall back to an interactive `huh` form when required -flags are omitted and stdin is a TTY. `import --replace` is the one -exception: it requires `--yes` unconditionally, in every output mode and -regardless of terminal attachment — never a prompt, and never `--json` as -consent (DKT-15). (`issue comment` and `doc comment` -use a different fallback — they open `$EDITOR` when no message is piped -and stdin is a TTY; see the Comments section below.) **In non-interactive/agent contexts -(no TTY) these commands return a `VALIDATION_ERROR` listing the missing -flags instead of hanging** — always pass all required flags explicitly -when scripting or running as an agent. `--json` mode never launches an -interactive form; missing required fields are always a hard -`VALIDATION_ERROR` in JSON mode. - ---- - -## Workflow: Issue Creation & Editing - -Create an issue (only `--title` is required in JSON mode): - -```bash -docket issue create --json \ - -t "Add rate limiting to API" \ - -d "Prevent abuse on public endpoints" \ - -s todo -p high -T feature \ - -l backend -l must-have \ - -f internal/api/router.go \ - -a "@alice" -``` - -Description can be piped from stdin with `-d -`: - -```bash -echo "Long description..." | docket issue create --json -t "Title" -d - -``` - -Edit only the fields you pass — `issue edit` uses `cmd.Flags().Changed(...)` -so omitted flags are left untouched, not reset to zero values: - -```bash -docket issue edit DKT-1 --json -s in-progress -a "@bob" -docket issue edit DKT-1 --json --parent DKT-5 # reparent -docket issue edit DKT-1 --json --parent none # make it a root issue again -docket issue edit DKT-1 --json -f a.go -f b.go # REPLACES the file list (not additive) -``` - -Reparenting validates against cycles (`db.IsDescendant`) and rejects -self-parenting with `VALIDATION_ERROR`/`CONFLICT`. - -Status transitions and other lifecycle commands: - -```bash -docket issue move DKT-1 review --json # arbitrary status transition -docket issue move DKT-1 --project vorpal --json # migrate issue + subtree to another project -docket issue close DKT-1 --json # shorthand for: move done -docket issue reopen DKT-1 --json # shorthand for: move backlog (only if currently done) -docket issue delete DKT-1 --json --force # cascade-delete issue + all sub-issues -docket issue delete DKT-1 --json --orphan # delete issue, promote sub-issues to root -``` - -Valid `--status` values: `backlog`, `todo`, `in-progress`, `review`, `done`. -Valid `--priority` values: `none`, `low`, `medium`, `high`, `critical`. -Valid `--type`/`-T` values: `task`, `bug`, `feature`, `epic`, `chore`. - -List and inspect: - -```bash -docket issue list --json -s todo -s in-progress -p high --tree -docket issue show DKT-1 --json # full detail: sub-issues, relations, comments, activity, docs -docket issue log DKT-1 --json --limit 50 -``` - ---- - -## Workflow: File Attachment (`docket issue file`) - -```bash -docket issue file add DKT-1 --json internal/api/router.go internal/api/middleware.go -docket issue file list DKT-1 --json -docket issue file remove DKT-1 --json internal/api/router.go -``` - -`add`/`remove` take 2+ positional args (`id` then one or more file paths) — -there is no `-f` flag on `issue file add`; that's only on `issue create -f` -and `issue edit -f`. Files are additive on `file add` (unlike `issue edit --f`, which replaces the whole list). - ---- - -## Workflow: Comments - -```bash -docket issue comment add DKT-1 --json -m "Investigated — root cause is a stale cache key" -docket issue comment list DKT-1 --json -``` - -`-m`/`--message` is optional: if omitted and stdin is a pipe, the body is -read from stdin; if omitted and stdin is a TTY (human mode only), `$EDITOR` -(default `vi`) is opened. In `--json` mode, `-m` (or piped stdin) is -required — there is no editor fallback. - ---- - -## Workflow: Labels & Relations - -```bash -docket issue label add DKT-1 --json backend must-have --color "#ff0000" -docket issue label rm DKT-1 --json must-have -docket issue label list --json -docket issue label delete backend --json --force # --force skips the attached-issue-count confirmation - -docket issue link add DKT-1 --json blocks DKT-2 # DKT-1 blocks DKT-2 -docket issue link add DKT-1 --json depends_on DKT-3 # DKT-1 depends_on DKT-3 -docket issue link remove DKT-1 --json blocks DKT-2 -docket issue link list DKT-1 --json -``` - -Valid `` values (`model.RelationType`): `blocks`, `depends_on`, -`relates_to`, `duplicates`. - ---- - -## Workflow: Dependency Graph (`docket issue graph`) - -```bash -docket issue graph DKT-1 --json --direction both --depth 2 -docket issue graph DKT-1 --mermaid --direction down # Mermaid flowchart, human-readable only -``` - -`--direction` must be one of `up` (what blocks this), `down` (what this -blocks), or `both` (default). `--depth 0` (default) means unlimited BFS -traversal. Use this before touching a shared interface to assess blast -radius. - ---- - -## Workflow: Planning (`docket plan` and `docket next`) - -`docket plan` groups all non-done issues into dependency-ordered execution -phases (topological sort; a cycle returns `CONFLICT`): - -```bash -docket plan --json -docket plan --json --root DKT-1 # scope to a parent issue's subtree -docket plan --json -s backlog -s todo -l must-have # filter by status/label -docket plan --json -p high -p critical -T bug -a alice # filter by priority/type/assignee -``` - -`docket next` finds work-ready issues — no incomplete blockers, in one of -the ready statuses (default `backlog`,`todo`): - -```bash -docket next --json -docket next --json -s todo -p high -p critical -l must-have --limit 5 -``` - -On docket.git itself (the engine repo), backlog defect sweeps run directly -in-session — plain edits, tests, commits — unless the operator asks for a -run: the plan/conduct pipeline is for corpus-governed work, and routing an -engine-repo sweep into `run start` has cost an operator interrupt and an -abandoned run (2026-08-17). - ---- - -## Workflow: Workflow definitions (`docket workflow`) - -A **workflow** is a declarative description of the steps a piece of work goes -through: what runs, in what order, what each step needs from the ones before -it, and what happens when something fails. It is a TOML file you register into -the database; nothing about it is specific to any kind of work or any kind of -worker. - -Start from a shipped template and register what it writes: - -```bash -docket workflow init --template standard-dev -# → wrote .docket/config/workflows/standard-dev.toml -docket workflow lint .docket/config/workflows/standard-dev.toml --json -docket workflow register .docket/config/workflows/standard-dev.toml --json -docket workflow list --json -docket workflow show standard-dev --json -``` - -At this stage registration is all that happens: a registered workflow is -stored, inspectable, and validated, but nothing runs it yet. A repo that never -registers one behaves exactly as it did before workflows existed. - -### Shipped templates - -| Template | Shape | -|---|---| -| `standard-dev` | two steps — run the checks, then have a person approve. One fenced-command gate, no fanout. | -| `parallel-check` | prepare → several checks in parallel → summarize → verify. Fanout with a join quorum. | - -Both are ordinary definitions, byte-identical to files you could have typed, -and both are asserted to register cleanly by a test. - -### Registration is content-addressed and immutable - -A registered `name@version` is **frozen**: - -| Second registration of… | Result | -|---|---| -| the same bytes | success, returns the existing row, changes nothing | -| **different** bytes at the same `name@version` | `CONFLICT` (exit 4), naming both hashes | -| any bytes at a new `version` | an ordinary registration | - -To change a workflow, bump `[pipeline].version`. This exists so that pinning -means something: a run that pinned `name@version` cannot have the definition -swapped underneath it. - -`register` accepts `-` to read stdin, so configuration generated in a pipeline -needs no temp file: - -```bash -generate-workflow | docket workflow register - --json -``` - -### Checking a draft without registering it (`docket workflow lint`) - -Registration is a persistent write: it inserts a row and freezes a -`name@version`. Checking a draft that way either accumulates versions nobody -wanted or does not happen at all — so `lint` runs the identical validation and -**writes nothing**: - -```bash -docket workflow lint .docket/config/workflows/standard-dev.toml --json -# → {"name":"standard-dev","version":1,"sha256":"…","registration":"new"} -generate-workflow | docket workflow lint - --json # `-` reads stdin -``` - -It is the **same pipeline `register` runs**, call for call: grammar and step -rules, vote rules against the config registry (V26), and threshold fields and -literals against the registered schemas. The two cannot report different -verdicts on the same bytes. - -The registry is **consulted, never written**, and the verdict says what a real -register would do: - -| `registration` | Meaning | -|---|---| -| `new` | nothing holds this `name@version`; a register would insert | -| `unchanged` | the same bytes are already registered; a register would be an idempotent success | -| *(refusal)* | **different** bytes hold this `name@version` — `CONFLICT` (exit 4), naming both hashes and the version to bump `[pipeline].version` to | - -**The conflict case fails the lint rather than reporting a third outcome.** A -definition that cannot register as it stands has not passed a check — and that -case is the trap this verb exists for: an edited file at a frozen `name@version` -validates cleanly and then refuses the *whole activation* the next time a run -starts. - -### Retiring a version from binding (`docket workflow deprecate`) - -A registered **name** binds forever at its highest version, and deleting its -TOML does not unregister it — nothing removes a `workflows` row. `deprecate` -retires one version from binding without deleting it: - -```bash -docket workflow deprecate standard-dev@1 --json -docket workflow deprecate standard-dev@1 --json --restore # back into binding -``` - -The row survives and stays fully readable: `workflow show` renders it, -`--source` emits the exact registered bytes, and a run that already **pinned** -it still resolves it and still completes. Only its candidacy for NEW bindings -stops — matching picks the highest **non-retired** version of each name, so -retiring the top version falls back to the one beneath it, and retiring every -version of a name takes that name out of routing entirely. - -A retired version reports `deprecated_at_ms` under `--json=v2` (see -`workflow list` below) and prints `[deprecated]` in human mode. - -### `docket workflow` refusals - -| Situation | Code | Exit | -|---|---|---| -| Grammar, validation, or lint failure | `VALIDATION_ERROR` | 3 | -| `payload` names a schema that is not registered | `VALIDATION_ERROR` | 3 | -| A threshold names a field its declared schema does not declare | `VALIDATION_ERROR` | 3 | -| A threshold literal is not a value its declared schema allows | `VALIDATION_ERROR` | 3 | -| An ordered comparison (`>=`, `>`, `<=`, `<`) on a field with no `ordered_enum` | `VALIDATION_ERROR` | 3 | -| A step emitting the reserved kind `gate-results` | `VALIDATION_ERROR` | 3 | -| Definition file not found | `NOT_FOUND` | 2 | -| Re-registering different bytes at an existing `name@version` | `CONFLICT` | 4 | -| `workflow lint` on a draft whose `name@version` is registered with different bytes | `CONFLICT` | 4 | -| `workflow show` on an unregistered name or version | `NOT_FOUND` | 2 | -| `workflow init` target exists without `--force` | `CONFLICT` | 4 | -| `workflow deprecate` without an explicit `@version` | `VALIDATION_ERROR` | 3 | -| `workflow deprecate` on an already-retired version | `CONFLICT` | 4 | - -### The definition grammar - -A definition has `[pipeline]`, an optional `[match]`, an optional `[limits]`, -and one or more `[[step]]` tables. **Unknown keys are an error**, naming the -key and its step — a typo'd `max_attempt` silently taking a default is exactly -the bug that makes a workflow behave differently from what its author read. - -```toml -[pipeline] -name = "standard-dev" # required; runs pin name@version -version = 1 # required, integer >= 1 -description = "…" # optional - -[match] # which issues bind to this workflow - # evaluated over the HIGHEST version of each name -kind = ["task", "bug"] -labels_any = ["…"] -labels_all = ["…"] -unless_labels = ["…"] # evaluated last and wins - -[limits] # per executor CLASS, not per step -write = { max = 1, lease_ttl = "45m", max_step_duration = "2h" } -read = 4 # bare int is shorthand for { max = 4 } - -[[step]] -name = "check" -after = [] # [] means root; see below -executor = "author" -emits = "check-report" -``` - -**`[limits]` keys, and what each bounds.** A class is an opaque string; these -are the only three things a bound on one does. - -| Key | Bounds | -|---|---| -| `max` | how many steps of the class may be `claimed`/`running` at once. A **finite** `max` is also what makes the class write-class for the reap acknowledgment below — a class with no `max` gets neither ack rows nor a headroom hold | -| `lease_ttl` | the lease a claim of this class takes, overriding `docket config lease.ttl.` | -| `max_step_duration` | a **schedule-to-close** bound measured from the claim, **independent of heartbeats** | - -`max_step_duration` is the one worth stating plainly: a step past it is reaped -**even with a live lease and a fresh heartbeat**, which is the whole difference -between it and a lease TTL. A runaway holder cannot renew forever. It is a -`[limits]` key on the workflow class — there is no `docket config` key for it. - -**Both reaps are scoped to an ACTIVE run.** A lease that lapses while a run sits -in `waiting-human` is not reaped, and neither is a step past its -`max_step_duration` there. A TTL is a bet that a silent worker is dead and its -step is better re-offered; on a run that is not active nothing would be offered -anyway, so the reap is pure loss — it clears a live worker's lease and takes a -write-reap hold on its class. This is a **suspension, not an exemption**: no -expiry is rewritten, so the first `next` after the run returns to `active` reaps -what came due meanwhile. It matters because the state is ordinary rather than -exotic — a run parks when *any* step is parked, leaving that step's siblings -legitimately `claimed` at `waiting-human`. - -`[[step]]` fields, in full: - -| Field | Type / default | Meaning | -|---|---|---| -| `name` | string, required, unique in workflow | step identity | -| `executor` | string (opaque hint) | a worker step; docket never interprets the value | -| `action` | string | a deterministic computation step | -| `type` | `"human"` \| `"vote"` | an operator gate | -| `fanout` | [hints] | expands to one parallel sibling per entry | -| — | | **exactly one** of `executor` / `action` / `type` / `fanout` per step | -| `class` | string, default = the `executor` value | the key `[limits]` accounts against | -| `emits` | artifact-kind string | **required on executor steps**; what the step records | -| `payload` | `name@version` | a registered payload schema; the step's `--payload-file` is validated against it at `complete`, from the bytes the run pinned | -| `voters`, `vote_rule` | [hints], name | **required on `type="vote"`**, forbidden elsewhere | -| `after` | [step names], **required** except on the first step and `loop = true` steps | predecessors; `[]` means root | -| `inputs` | [`"."` \| `".*"` \| `".gate-results"` \| `"issue.body"` \| `"issue.diff"`] | artifacts delivered to the step | -| `holds_tree` | bool, **default true** | whether this step OCCUPIES its issue's scope while it runs. It is what scope exclusion consults, and what decides whether the step's completion records an `issue.diff` | -| `gates` | [name \| `{name, source="fence:", pre=bool}`] | checks; `pre = true` runs at claim | -| `params` | opaque table | arguments to an `action` step | -| `min_siblings` | int, default = all | how many fanout siblings the join needs | -| `threshold` | table: routing → predicate | routing computed from the step's results | -| `on_fail` | `"fix-loop"` \| `"waiting-human"` \| `"skip"` \| `"abandon-issue"`; default `"waiting-human"` | where a failure routes. **Required explicitly on `type="human"` and `type="vote"` steps** — the default is a routing nobody chose | -| `loop` | bool, default false | marks a loop-body step | -| `after_loop` | step name | where execution re-enters after a loop body | -| `max_attempts` | int ≥ 1 | per-instance retry budget | -| `max_fix_loops` | int ≥ 0 | loop-entry budget per issue | -| `expected_cost` | number ≥ 0, default 0 | the step's contribution to the run's budget floor, accrued **per claim**. Per expanded sibling on a fanout — four siblings accrue four times, no proration | -| `when` | predicate over `kind` / `labels` | step is skipped when false | -| `metadata` | opaque table | recorded and delivered verbatim | -| `packet` | list of paths, relative to `.docket/config/` | files inlined into the step's rendered work packet, **in declared order**. Each must be pinned by the run (they are, automatically, if they live under `.docket/config/`); an entry the run did not pin is refused **at activation**. An entry may carry the `{executor}` token, substituted with that sibling's executor hint — which is how one `fanout` step gives each sibling a different file. Docket reads their bytes and never interprets them | - -**`packet` inlines files; it never points at them.** The rendered packet carries -each file's **body**, delimited and labeled with its path and hash, so a worker -receives one document rather than a list of things to go read. Bytes are -admitted only when they hash to what the run pinned: a file edited after -activation is `CONFLICT` (exit 4) naming **both** hashes, and one deleted is -`NOT_FOUND` (exit 2). That is what makes a packet reproducible — same step, same -packet, byte-identical, even mid-run. - -A packet file may declare more files in a `packet_includes:` frontmatter list, -and those are inlined immediately after it. **That is the only frontmatter key -Docket reads** — every other key is ignored entirely, not validated and not -surfaced — and includes are followed **exactly one level deep**. A malformed -`packet_includes` is `VALIDATION_ERROR` at render, naming the file; a declared -include that is missing or unpinned is refused rather than silently omitted. - -#### Engine-produced inputs - -Three `inputs` forms resolve to something the engine recorded rather than to a -step's declared artifact. - -`issue.body` is the activation snapshot. `issue.diff` is the run's computed VCS -diff, and **it is recorded only at the completion of a step that holds the -tree** — `holds_tree`, default true, the same declaration scope exclusion reads. -A non-holding step records nothing, and its consumers resolve to the artifact -the last **holding** step recorded: the reviewed object, pinned at the moment -the change existed, byte-identical for every sibling that reads it. Recomputing -at every executor completion was the defect this closes — read-shaped fanout -siblings each re-diffed the *live* tree at their own record time, so a diff -taken after the change landed came out empty and one taken beside a sibling's -in-flight probe carried the probe. Action, human, and vote steps never record a -diff either; with no diff artifact at all, the input resolves to an **empty -diff**, never an error and never a live `git diff`. - -**The bundle carries a machine-readable target ref (DKT-24).** Context -assembly lifts the resolved `issue.diff` artifact's round record onto the -bundle as `target_sha` — the commit the diff's tree stood at — and -`target_worktree` — the producing record's declared worktree path, good while -that checkout is still on disk (it is swept at integration). Both are omitted -entirely when the resolved diff carries no round record. The default packet -template states them in its header, so a reviewing consumer no longer -re-derives the tree from a prose convention in the change-summary's first -line. They are exactly as reproducible as the input they describe. - -`.gate-results` is the named step's **recorded** gate results, served from -the ledger rather than re-run — one input per `done` producer instance, carrying -a JSON array in §11.4's gate-result shape (`gate`, `ordinal`, `argv`, `exit`, -`duration_ms`, `output`, `truncated`, `verdict`, `pre`, `reason`), the same -shape a claim response's `pre_gates` carries. Instance selection mirrors -ordinary artifact resolution — same issue, `done` only, ordinal-scoped with -the per-input fallback, siblings in index order — with one departure (DKT-12): -the **requesting step admits itself regardless of status**, so a self-declared -`.gate-results` reads the step's own claim-time `pre = true` rows (which -commit before context assembly, while the step is still `claimed`). -Completion-side rows, not yet recorded at claim, show up as the empty array. - -```toml -inputs = ["implement.change-summary", "implement.gate-results"] -``` - -The producer **must be a step of this workflow**, but need not declare gates: -a producer that recorded none resolves to an **empty array**, not an absent -input, because "this step ran no checks" is an answer a consumer can act on -while a missing input reads as a resolution failure. (Gates can also arrive -from a `fence:` source the definition does not enumerate, so requiring a -declaration would refuse correct workflows.) `gate-results` is a **reserved -kind**: a step emitting it is refused at `workflow register` (V11a), since an -artifact of that kind could never be addressed — the engine-served form would -shadow it. - -**`after` is required, and `after = []` is how you declare a root.** Implicit -topology was a footgun: a step that forgets `after` would silently become a -root and run first. Only the first step and `loop = true` steps may omit it. - -**Every gate step must declare `on_fail` explicitly — `type="human"` and -`type="vote"` alike (V13a).** The default is `waiting-human`, so a gate that -declares nothing has a routing its author never chose. - -**A `type="human"` step additionally may not route rejects to `waiting-human`** -(V13): it would park the issue on the resolution of the very thing that just -rejected it, a deadlock. Legal values there are `fix-loop`, `skip`, and -`abandon-issue`. - -**A `type="vote"` step MAY route to `waiting-human`**, and often should: on a -vote gate that routing is the ESCALATION — a tally that did not reach its -threshold decided nothing, so the question passes to an operator who has not -been asked yet, which is not a wait on the decider that just declined. All four -values are legal there. Because both readings are defensible, the grammar makes -you say which you mean rather than inheriting one silently. - -`threshold` predicates have the shape `agg(field op literal)` with -`agg ∈ {any, all, count>=n}` and `op ∈ {==, !=, >=, >, <=, <}`; routings are -`fix-loop`, `waiting-human`, `pass`, or a step name (which interposes that step -as a gate). Routings are evaluated **top to bottom, first match routes**, and -no match routes `pass`. - -**An interposed gate runs only when routed to.** A step named as a step-name -routing target — author it with `after = [routing-step]` — is latched by -readiness until a routing predecessor's **recorded** routing names it, and when -the routing resolves anywhere else it is terminalized `skipped` in the same -routing transaction, so joins and issue completion resolve without it. A -`next --run` offer may still carry such a gate in its staged closure, marked -`conditional`: confirm the predecessor actually routed to it before spawning -anything for it. - -**Fields and literals are still opaque tokens to docket — but they are checked -against your schema.** When a step declares a `payload`, `workflow register` -verifies that every predicate's field is one the schema declares, that every -literal is a value that field accepts, and that any ordered operator -(`>=`, `>`, `<=`, `<`) names a field the schema marks `ordered_enum`. Docket -learns that `high` comes after `medium` because your document said so; it holds -no opinion about what either word means. - -A step with a `threshold` and **no** `payload` is legal and unchanged: -equality has never needed an order. An ordered comparison over such a field -**parks the step** `waiting-human` with a reason naming the predicate — docket -declines rather than guessing an order. - -**Executor hints are opaque.** `executor`, `fanout` entries, `voters`, and -`class` are strings docket stores, echoes back, and uses as map keys. There is -no registry of known executors and no behavior keyed on the value: put role -names, team names, or people's names there and they mean what you intend. -`params` and `metadata` are likewise opaque — docket never reads a key inside -them. - -### Gates — what actually runs - -A gate is a check a step must pass. It comes in two spellings: - -```toml -gates = ["tests"] # a named gate -gates = [{ name = "checks", source = "fence:checks" }] # commands from the issue body -gates = [{ name = "measure", pre = true }] # runs at claim, not at complete -``` - -**A gate name is an opaque string.** Docket looks it up in your trust store and -never interprets it — there is no registry of known gates, no gate whose name -has behavior, and no default gate. - -**Every gate needs a matching trust entry or it does not run** (see -`docket trust` below and security.md §7). An unmatched gate is recorded -`verdict: "unmatched"` with null `argv` and null `exit`, nothing spawns, and -**the step fails** and routes per `on_fail`. A workflow whose check cannot run -has not passed its check. - -| Spelling | Where the command comes from | -|---|---| -| `"name"` or `{name}` | the trust entry's own argv — the entry *is* the command | -| `{name, source="fence:"}` | fenced blocks in the issue body whose info string is ``, harvested and hashed at activation, one command per line | -| `{name, pre=true}` | runs at **claim**, with its result in the context bundle rather than judging the step | - -A fence tag is opaque too: `source = "fence:checks"` harvests ```` ```checks ```` -blocks and docket never knows what the word means. Fenced commands are matched -**per line**, each its own decision with its own recorded result. - -Gate results are recorded as `{gate, ordinal, argv, exit, duration_ms, output, -truncated, verdict, pre, reason}` with -`verdict ∈ {pass, fail, unmatched, skipped}`. A `pre = true` gate's results ride -in the claim response under `context.pre_gates` — present only when the step -declares them. A failing pre-gate does **not** refuse the claim: it is a -measurement the step consumes, and the judging is the step's job. A later step -reads the same rows by declaring `inputs = [".gate-results"]` (see -*Engine-produced inputs* above). - -**`skipped` means nothing was measured**, and it is a different fact from -`fail`. A gate measures the tree its step is about to judge; when that tree -cannot be bound, docket **records `skipped` rather than measuring a different -tree**. A pass collected in the shared checkout, while the change under review -lives somewhere else, is a verdict with no evidence value — and one that reads -as green. - -Docket tries to avoid the skip first. A worktree that has been swept (integration -removes them between waves) is **reconstructed from the object database**: the -commit is still there, so the tree is rebuilt in a throwaway detached checkout, -measured, and removed. Those rows say so in their `reason`. Only when the commit -itself is unreachable does the gate skip, and the reason names the sha so you -know what to fetch. - -A step whose gates recorded `skipped` **parks at `waiting-human`** — not its -`on_fail`. Nothing is known about the change, so a fix loop would ask a worker to -fix a tree nobody read, and a judge panel would deliberate over an infrastructure -condition. What *is* known is something an operator can act on. `skipped` is -counted in its own column in `run report`, beside `pass` and `fail`. - -#### What a gate's child process sees - -The child environment is **constructed, not inherited**: a variable is present -only because the allowlist names it (`PATH`, `HOME`, `USER`, `LOGNAME`, `SHELL`, -`LANG`, `LC_ALL`, `LC_CTYPE`, `TZ`, `TMPDIR`, `SSL_CERT_FILE`, `SSL_CERT_DIR`, -`XDG_CACHE_HOME`). An unset parent variable is omitted rather than set empty. -`DOCKET_TOKEN` and `DOCKET_PATH` are denied outright — a capability token in a -child would convert code execution into engine authority — and a spawn aborts -if either is ever found in the constructed set. - -Docket then **sets** these itself: - -| Variable | Value | -|---|---| -| `TERM` | always `dumb` — a gate's output is captured, not displayed, and inherited ANSI escapes would pollute the run report | -| `CI` | always `1`, the near-universal "non-interactive" convention | -| `DOCKET_GATE` | the gate name (opaque to core) | -| `DOCKET_REPO` | the repository root | -| `DOCKET_ISSUE` | `DKT-N`, the issue the gated step belongs to | -| `DOCKET_SCOPE` | the issue's declared scope globs, **newline-joined**; absent entirely when there are no globs to carry | -| `DOCKET_GATE_NETWORK` | the trust entry's declared hosts, comma-joined — set only when it declared any, alongside the proxy variables | - -`DOCKET_ISSUE` and `DOCKET_SCOPE` are what let a **diff-shaped** gate evaluate -the change it is actually gating instead of the whole dirty tree. The globs are -newline-joined rather than JSON because the consumer is a shell check reading -its own environment, where `while IFS= read -r glob` needs no parser. **Absent -is not empty**: an issue that declared no scope gives the check no narrower -answer than the tree, and inventing one would be docket deciding what the issue -touches. - -The variable carries globs or nothing, so it is the one surface where declaring -no scope and declaring an empty one look alike — a declared-but-empty scope -leaves `DOCKET_SCOPE` unset too, rather than setting it to the empty string. -Everywhere the two are distinguishable they stay distinguished: v1's `scope` -key, and the activation lint that warns about the first and not the second. A -gate that must tell them apart reads `docket issue show`, not its environment. - -There is no way to extend the allowlist — no flag, no config key, no trust-entry -field. - -### Action steps — computations, not workers - -An `action` step has no worker. It declares a computation and `params.output`, -which is the artifact kind it produces: - -```toml -[[step]] -name = "reconcile" -after = ["synthesize"] -action = "aggregate" -inputs = ["synthesize.findings"] -payload = "findings@1" -params = { field = "severity", method = "median", hold_spread = 2, output = "findings" } -``` - -**Nothing claims an action step.** `docket step claim` refuses one with -`CONFLICT` — "resolved by the engine, not by a worker" — the same way it refuses -a `human` or `vote` gate. The engine runs it, records its artifact, and routes. -It still appears in `docket next --run` so a dispatcher can see what a run is -doing; the row simply carries no `executor` to spawn. - -**Resolution is builtin-first.** `aggregate` is the one computation docket -performs itself; every other action name is looked up in your trust store and -run as a **user-trusted command**, through the same matching, argv resolution, -env allowlist, timeout, capture, and repo containment a gate goes through. There -are no exceptions and no second execution path. The name `aggregate` is -reserved, so a trust entry cannot shadow it — `workflow register` says so rather -than leaving you to wonder why your command never ran. - -An unmatched action name records `verdict: "unmatched"` with null `argv` and -null `exit`, spawns nothing, and **fails the step**, which routes per `on_fail`. -A computation that could not run has not succeeded. - -#### The trusted-command contract - -| Direction | Shape | -|---|---| -| **stdin** | the step context object, exactly as `docket step context --json` emits it — one JSON document, then EOF | -| **stdout** | one JSON object `{"body": "", "payload": [ … ]}`. `body` defaults to `""`, `payload` to `[]`. No other keys, one document | -| **exit 0** | success; the artifact records with `kind = params.output` | -| **non-zero exit** | failure; the step routes per `on_fail`, the captured output is recorded, and **no artifact is written** | -| **unparseable stdout on exit 0** | failure, with the first 200 bytes quoted back with control characters escaped | - -An object rather than "stdout is the payload", because every artifact has a -human-readable body and a command needs a channel for it. `stderr` is the -diagnostic stream and is what `action_results.output` records; it cannot corrupt -the document docket parses. - -If the step declares a `payload`, the produced payload is validated against that -schema exactly as a worker's is. A failure there is a step failure routed per -`on_fail`, not a refusal to a caller — there is no caller. - -Every attempt is recorded as an **action result**: -`{action, ordinal, argv, exit, duration_ms, output, truncated, verdict, builtin, -reason}` with `verdict ∈ {pass, fail, unmatched}`. `builtin` marks a computation -docket performed itself, and `argv`/`exit` are null there because nothing -spawned. A `flaky` trust entry re-runs and each attempt gets its own row, with -the **last** one deciding the routing. - -#### `aggregate` — the one builtin - -`aggregate` reduces clustered values to one value per cluster, over an order -**your schema declares**. It works for severities, priorities, tiers, T-shirt -sizes, or ripeness grades alike: docket knows position, never significance. - -| Param | Type | Required | Meaning | -|---|---|---|---| -| `field` | string | yes | the payload property to reduce | -| `method` | `median` \| `max` \| `min` | yes | the reduction | -| `hold_spread` | integer ≥ 0, default 0 | no | hold when the spread reaches this; `0` never holds | -| `output` | string | yes | the artifact kind this step produces | - -No other keys are accepted — a typo'd `method = "medain"` is refused at -`workflow register`, not discovered hours into a run. - -An `aggregate` step **must** declare `payload = "name@version"`, and that -schema must mark `params.field` as `ordered_enum`. Median, max, and min are -defined only over an order, so an aggregate without one could never compute. - -**The input.** The builtin reduces the **concatenated payloads of the step's -declared `inputs` artifacts**, resolved by the ordinary input rules — `done` -producers only, in declared order, and scoped to the step's own loop ordinal. So -`inputs = ["synthesize.findings"]` means "reduce what `synthesize` recorded". -`inputs` must be non-empty on an `aggregate` step, refused at `workflow -register`: a step with nothing to read can never compute. - -Each element of that payload is one cluster. The element's `field` is either an -**array** of values — the cluster's members — or a **scalar**, which is a -one-member cluster. Every other key of the element is carried through verbatim. - -Over a flat payload of scalars, `aggregate` is the **identity**: every value -passes through, nothing is held, nothing is demoted. You can introduce -clustering later without a behavior change anywhere else. - -**The even-count rule.** With members sorted by their position in your declared -order, the reduction is `m[0]` for `min`, `m[len-1]` for `max`, and -`m[(len-1)/2]` for `median` — **the LOWER of the two central values when the -count is even**. So a cluster of `{low, blocker}` medians to `low`. - -That is not caution. Docket does not know which end of your order is worse: a -rule that took "the more severe of the two" would be docket holding an opinion -about severities, which is simply wrong for a `confidence` or a `ripeness` -enum and invisible when it is. One expression, no special case, and the -standard lower median for ordinal data where no average exists. - -**If that is the wrong end for your order, say so in the schema.** Docket cannot -know which end is worse, but *your order can*. Add `"conservative_end": "upper"` -beside the `ordered_enum` annotation and that field's even-count median ties -resolve toward the top of the declared order instead — `{low, blocker}` medians -to `blocker`. Declare nothing and the lower median is unchanged, which is what -keeps a `confidence` or `ripeness` order behaving exactly as it always has. See -[The `conservative_end` annotation](#the-conservative_end-annotation). - -The direction moves the **median tie and nothing else**: `min` and `max` already -name an end explicitly, and an odd-count median has no tie to break. If you want -the top of the order in *every* case and not only on ties, that is `method = -"max"`, not a direction. - -**Spread and holds.** `spread` is the distance between the extreme members' -**positions** — so with `["info","low","medium","high","blocker"]`, both -`{low, high}` and `{low, medium, high}` have spread 2. A cluster holds when -`hold_spread > 0 && spread >= hold_spread`. - -**The demotion trail.** When the computed value's position is strictly below its -highest member's, the output records `demoted_from` with the value that was not -taken. When nothing was demoted the key is **absent**, not empty. `max` never -demotes. - -**The output**, one element per input element, validated against both the -shipped `aggregate@1` schema and your own: - -```json -{ "severity": "medium", "members": ["low","medium","high"], "held": true, - "demoted_from": "high", "operator_resolved": false, "…your other keys…": "…" } -``` - -#### Held clusters, and what you do about one - -When `hold_spread` trips, docket materializes a `type="human"` step **per held -cluster**, named `-held` at the same ordinal with the cluster's payload -index as its sibling suffix — `reconcile-held@0#0`, `reconcile-held@0#2`, … — -and the routing step **stops**. Concretely: - -- The routing step's status stays `gated`, which is non-terminal, so every - downstream step waits. Its threshold is **not** evaluated yet. -- Each held step is offered by `next --run` immediately, takes no claim and no - token, and shows up as an ordinary human gate — or as a vote gate, when - `vote.hold.*` is configured (see *Engine configuration*). Everything below is - the same either way; a tally simply answers first, and escalates to the - operator's verbs below when it does not pass. -- `guard stop` **denies** while any is open, exactly as it does for a declared - human gate awaiting approval. Stopping means resolving or abandoning first. -- The step-name suffix `-held` is **reserved**: you cannot declare a step whose - name ends in it. -- **`#N` is the cluster's POSITION in the payload, not a cluster id.** A held - step names its own provenance so the two cannot be confused: `step show` - carries `held_cluster` — `cluster_index`, `cluster_count`, the `artifact` - the payload lives on, and the `producer_step` that recorded it — and `step - artifacts` on the row, which is legitimately empty because a hold produces - nothing, names that artifact instead of stopping at "produced no artifacts" - (DKT-239). Two clusters of one payload point at the SAME artifact, which is - exactly what the index disambiguates. - -**One step per cluster, so you can answer them differently.** A hold carrying -four clusters gives you four approve/reject decisions, not one. The suffix is -the cluster's index in the payload, which is stable across re-reads of the same -immutable artifact — so a resumed saga re-derives the same step for the same -cluster, and a cluster that was never held has no step. (A hold where only the -second cluster trips materializes `#1` and no `#0`.) - -| Verb | Effect | -|---|---| -| `docket step approve [--note N] [--value V]` | records a **new** artifact on the routing step with `operator_resolved: true` on **that cluster**, marks the held step `done` | -| `docket step reject [--note N]` | records **no** artifact for that cluster, marks the held step `done` | - -**`--value V` is the corrected value for the cluster's aggregated field.** It -lands on the **field itself**, so every threshold and every downstream input -routes on the number the operator actually endorsed, and the computed value it -replaced is recorded beside it as `operator_set_from` — the two records stay -distinguishable rather than one overwriting the other. `--note`, when given, -travels with the decision as `operator_note` on the same element, so a fixer -reading the resolved payload learns *what was decided* and not merely *that a -decision happened*. - -| Rule about `--value` | | -|---|---| -| It is validated against the **pinned schema's declared enum** before anything is written | a correction must be a member of the membership set the run agreed to; a value outside it is a `VALIDATION_ERROR` | -| It is **never parsed from `--note`** | docket does not read a disposition out of prose. Refusing to infer one was always right; what it argued for was a structured field, not no field | -| It accompanies **approve** only | reject records no artifact for the cluster, so there is no value to set — `--value` with `reject` is a `VALIDATION_ERROR` | -| It applies to **materialized** `-held` steps only | a declared human gate has no payload of its own to correct, so the flag would reach nothing there; that too is a `VALIDATION_ERROR` naming the step | -| The routing step must declare an aggregated field and a `payload` schema | otherwise there is no field to set and no enum to check against | - -The value rides in the `step-approved` event beside the note, so the feed's -account of the decision carries the decision's content. - -The routing step waits until **every** cluster has an answer, then routes once: -per its effective `on_fail` if **any** cluster was rejected, otherwise through -the threshold over the resolved payload. Reject is the escalating answer, so a -mixed set does not silently pass — but each cluster keeps its own status, -routing, and note, so the record stays per-cluster even though what routes is -one decision. - -Approval means *accept the cluster* — at the computed value, or at the one -`--value` names. The originally-held artifact stays addressable forever: what -docket computed and what you accepted are two records, not one overwritten one. - -A step parked because its clusters were **rejected** cannot be retried: -`docket step resolve --as retry` refuses there rather than silently re-parking -it. The rejection is sticky — re-running the aggregate re-reads the same -rejected decision and routes to the same place, so the attempt counter was never -what blocked it. What moves such a step is `override-pass`, `skip`, or -`abandon-issue`, and the rejected verdict stays addressable through all three. - -A loop entry supersedes an unresolved held step along with everything else at -that ordinal — the question was about that ordinal's computation, and the loop -has moved past it. - -### Fanout and joins - -A `fanout` step expands to one sibling per hint, in declared order: -`review@0#0 … review@0#3`. A step declaring `after = ["review"]` waits for the -**join**, and the rules are worth knowing exactly: - -| Rule | Behavior | -|---|---| -| The join releases when **every** sibling is terminal | terminal means `done`, `skipped`, `superseded`, or `failed-routed`. A sibling that ended any of those ways has ended; waiting for all of them to be `done` would deadlock on the first one that failed or was skipped. | -| A sibling in `waiting-human` **parks the issue** | `waiting-human` is not terminal, so the join stays open until an operator resolves it with `docket step resolve`. | -| Downstream `inputs` resolve over **`done` siblings only** | a sibling that failed produced no result, so its artifact is not an input. | -| `on_fail` applies **per sibling** | one sibling failing routes that sibling. The other three still finish on their own terms. | -| `min_siblings` is a **quorum**, compared after the join | if fewer than `min_siblings` siblings are `done` once every sibling is terminal, the fanned step routes per its `on_fail`. | - -**`min_siblings` does not cancel early.** Reaching the quorum does not release -the join: docket waits for every sibling to finish and *then* compares. A -4-way fanout with `min_siblings = 2` and two siblings already `done` still waits -for the other two. This is deliberate for v1 — cancelling work that is already -running, to save time on a quorum that is already met, is a decision docket -declines to make on your behalf. - -### Loops - -A `threshold` (or an `on_fail`) that routes `fix-loop` enters a loop. There is -**no other loop construct** — a threshold routing to a *step name* interposes -that step as a one-off gate and is not a loop. - -What happens on loop entry, in one transaction: - -1. **The issue's loop counter increments.** The counter is per-issue, not - per-step. If the new count would exceed `max_fix_loops`, the routing becomes - `waiting-human` instead and no loop is entered — loops are bounded by - construction, and the parked step's routing records why. -2. **Unclaimed work downstream of `after_loop` is superseded.** Instances at a - lower ordinal that are still `pending` become `superseded` — a terminal - status, not a deletion. Already-claimed and running instances are **left - alone to finish**; their eventual routing is recorded for the ledger but - applies no downstream effect, so a slow step from the previous ordinal cannot - re-route an issue that has already moved on. -3. **`loop = true` steps instantiate at the new ordinal**, along with the - `after_loop` step and everything transitively after it. Gates re-run and - thresholds re-apply on the new instances — they are fresh, with no gate trail - and no routing carried over. - -Steps **upstream** of `after_loop` do not re-run. That is why `inputs` bind -**per input**: a step at ordinal 1 resolves each declared input at ordinal 1 if -something produced it there, and otherwise falls back to the highest earlier -ordinal that did. The fixture's `fix` step binds `reconcile.findings` fresh at -ordinal 1 and `implement.change-summary` from ordinal 0, in the same step. - -**Issue completion is evaluated over highest-ordinal instances only.** A `done` -step at ordinal 0 whose ordinal-1 instance is still pending does not count as -finished, and superseded ordinal-0 instances do not block completion. Prior -instances and their artifacts stay immutable and addressable for the ledger. - ---- - -## Workflow: Payload schemas (`docket schema`) - -A step can declare `payload = "name@version"`. That names a **payload schema**: -a JSON Schema document you register, against which the step's `--payload-file` -is checked, and — this is the part that matters — the place where **order comes -from**. - -Docket's threshold predicates include ordered comparisons: `any(risk >= medium)`. -Docket does not know that `medium` outranks `low`. It knows it because your -schema said so. - -```bash -docket schema register risk-report@1 .docket/config/schemas/risk-report.json -docket schema list -docket schema show risk-report@1 --body -``` - -### The `ordered_enum` annotation - -An ordinary JSON Schema, plus one key: - -```json -{ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "array", - "items": { - "type": "object", - "properties": { - "risk": { - "type": "string", - "enum": ["info", "low", "medium", "high", "blocker"], - "ordered_enum": true - } - }, - "required": ["risk"] - } -} -``` - -| Rule | | -|---|---| -| The order **is** the `enum` array, as written, ascending | there is no second list to disagree with it, and adding a value to `enum` cannot leave the order stale | -| `ordered_enum` must sit beside an `enum` of **at least two unique strings** | otherwise it is a `VALIDATION_ERROR` naming the property path | -| It constrains **nothing** | a document that validated before the annotation validates after it; the annotation declares position, not a new rule | -| Only **top-level properties of the array's item schema** are indexed | that is exactly what a threshold predicate's bare field token can name | - -Docket learns *position*, never *significance*. It does not know which end of -your order is worse, more urgent, or better — and it never has to. A median over -a declared order is the same computation for risk levels, priorities, tiers, -T-shirt sizes, or ripeness grades. - -A payload is an array of objects. A schema over it is therefore usually -`{"type": "array", "items": {"type": "object", …}}` — and a document that is not -that shape still registers, it simply declares nothing a threshold can name. - -### The `conservative_end` annotation - -Docket learns position, never significance — but *an order can know its own bad -end even when docket cannot*. One optional key says which one it is: - -```json -"risk": { - "type": "string", - "enum": ["info", "low", "medium", "high", "blocker"], - "ordered_enum": true, - "conservative_end": "upper" -} -``` - -| Rule | | -|---|---| -| The value is `"upper"` or `"lower"` — **positions in the ascending `enum`**, never the values it holds | `"high"` is a `VALIDATION_ERROR`: it is a member of *this* enum and could not name an end of any other | -| It must sit beside `"ordered_enum": true` | a direction names an end of an order; declared over an unordered field it is a `VALIDATION_ERROR` naming the property path, not a silently-ignored key | -| It is **optional**, and absent means `"lower"` | every schema written before this key existed computes exactly what it did before | -| It changes exactly one decision: the **even-count median tie** | it does not reach `min`, `max`, an odd-count median, a threshold predicate, `spread`, or a hold | - -Declare it on a severity, a priority, or any order where a tie should fall on -the cautious side; leave it off for a confidence, a ripeness, or a tier, where -the two central values are simply two values and neither end is "bad". A -direction is not part of the order, so adding one to a schema does not change -what any threshold predicate compares — only which of two tied medians is taken. - -### Registration is content-addressed and immutable - -Exactly as workflows are: - -| Second registration of… | Result | -|---|---| -| the same bytes | success, returns the existing row, changes nothing | -| **different** bytes at the same `name@version` | `CONFLICT` (exit 4), naming both hashes | -| any bytes at a new `version` | an ordinary registration | - -To change a schema, register a new version. This is not ceremony: a schema -decides whether a worker's payload is *accepted*, so a mutable `risk-report@1` -would change a running job's acceptance criteria mid-flight. A workflow that -wants the new schema declares it and bumps its own `[pipeline].version`. - -Registering `risk-report@2` does nothing to a workflow that declares -`payload = "risk-report@1"`. Version references are exact. - -### Register schemas before the workflows that name them - -A workflow declaring `payload = "risk-report@1"` is **checked against the -registered schema when you register the workflow**, not later: - -| Check | What it refuses | -|---|---| -| the schema is registered | `payload` naming something absent — with the `docket schema register` line to run | -| the field exists | `any(rsk >= medium)` when the schema declares `risk`, listing what it does declare | -| the literal is valid | `any(risk == severe)` when the enum is `low, medium, high` | -| ordered means ordered | `any(stage >= final)` when `stage` declares an `enum` but no `ordered_enum` | - -So register schemas first. A workflow that could never route correctly should -not register at all — the alternative is discovering a typo hours into a run, on -a step whose work is already done. - -**Auto-registration already does this for you.** Activation registers everything -under `.docket/config/schemas/` before anything under `.docket/config/workflows/`, -so a workflow and the schema it names can live side by side in the same tree and -the ordering is never yours to arrange. The rule above matters when you register -by hand, and as the reason the auto-registration order is what it is. - -**A threshold on a step with no `payload` is untouched by all of this.** It is -grammar-checked and nothing more, and at runtime an ordered comparison over a -field with no declared order parks the step for a human rather than guessing. -`any(status == unmet)` never needed a schema to be correct, and still does not: -equality has never needed an order. - -**Changing a schema does not change an already-registered workflow.** -`payload = "risk-report@1"` names a version. Registering `risk-report@2` creates -a new row and touches nothing; a workflow that wants it declares it and bumps its -own `[pipeline].version`. That is worth knowing because "the schema was updated" -is the first thing anyone assumes when a threshold behaves as it did yesterday. - -### The one shipped schema - -`docket schema list` reports one row in a fresh repo: - -``` -aggregate@1 1e0a0be39394 builtin -``` - -`aggregate@1` ships with docket and describes the output of the builtin -`aggregate` action step. It is inert unless such a step runs, and nothing else -in the registry arrives without someone registering it. - -### Validation at `complete` - -A step that declares `payload = "name@version"` has its `--payload-file` -validated when it completes: - -``` -Error: step assess@0: payload does not satisfy risk-report@1: - payload[3].risk: value "urgent" is not one of ["info","low","medium","high","blocker"] - (+2 more) -``` - -Three things about that refusal are deliberate: - -- **It is path-precise.** `payload[3].risk` is the element and the property, in - the notation the file itself is written in. "The payload is invalid" would be - something a worker can only re-submit against blindly. -- **It is capped at five lines.** A worker's log is not improved by a hundred, - and the count of what was dropped is reported so you know the list is partial. -- **It validates against the bytes the RUN PINNED**, not against whatever the - registry holds now. Two runs of the same work at the same pins reach the same - verdict on the same payload. - -Authorization is checked first, always: a caller that does not hold the step's -token gets `AUTH_ERROR` and learns nothing about the schema. - -### `docket schema` refusals - -| Situation | Code | Exit | -|---|---|---| -| Schema file not found or unreadable | `NOT_FOUND` | 2 | -| Reference is not `name@version` with a version ≥ 1 | `VALIDATION_ERROR` | 3 | -| Malformed JSON, or a document that does not compile as JSON Schema | `VALIDATION_ERROR` | 3 | -| `ordered_enum` without a usable sibling `enum` | `VALIDATION_ERROR` | 3, naming the property path | -| Re-registering different bytes at an existing `name@version` | `CONFLICT` | 4 | -| `schema show` on an unregistered name or version | `NOT_FOUND` | 2 | -| `--payload-file` fails the step's declared schema at `complete` | `VALIDATION_ERROR` | 3 | -| `--payload-file` omitted on a step that declares `payload` | `VALIDATION_ERROR` | 3 | - ---- - -## Workflow: Voting (`docket vote`, consensus proposals) - -**A workflow step can open one of these.** A `type="vote"` step creates a -proposal when it becomes ready, fans out to the voters it names, and routes on -the outcome — `approved` passes the step, `rejected` routes per its `on_fail`, -which such a step must declare explicitly (V13a). `waiting-human` is legal there -and means *escalate to an operator*, unlike on a human gate where it is refused. -The proposal's id rides on the step row as `proposal`, and the roster as -`voters`, so a caller holding a `next` row can cast without reading the pinned -definition. Nothing about the machinery below changes: voters cast with the same -`docket vote cast` shown here, the tally is the same weighted score and quorum, -and there is **no new verb**. The step names a `vote_rule`, which is a pair of -`vote.rule..*` config keys, and `required_voters` is the length of its own -`voters` list. Nothing casts a vote automatically — a voter is a person or a -process running the CLI. - -Create a proposal: - -```bash -docket vote create --json \ - -d "Adopt Result for all internal/db error returns" \ - -r "Panics currently propagate uncaught in 3 call sites" \ - -c high -n 3 --threshold 0.67 \ - --domain-tags "database,error-handling" \ - --files-changed "internal/db/issue.go,internal/db/doc.go" -``` - -Cast a vote: - -```bash -docket vote cast DKT-V1 --json \ - -v approve --confidence 0.9 --domain-relevance 0.8 \ - --findings "Reviewed all call sites, no blockers" \ - --summary "LGTM" \ - --metadata '{"model_resolved":"claude-sonnet-5","effort_resolved":"high"}' -``` - -`--metadata` is optional and opaque: a seat records what produced its vote -there, and the value is stored whole and read by nobody. Pass it whenever the -vote was cast by something whose spend an operator may later want to attribute. -Treat the value as public — it is visible to anyone who can list processes, it -is stored verbatim in the store, and `docket export` re-emits it verbatim with -no redaction. It reads back through `vote show --json`, `vote result --json`, -and the export document; the human-readable tables do not render it. - -Valid `--verdict`/`-v` values: `approve`, `approve-with-concerns`, `reject`. -Valid `--criticality`/`-c` values: `low`, `medium`, `high`, `critical`. -`--confidence` and `--domain-relevance` are floats in `[0.0, 1.0]`. - -Inspect and finalize: - -```bash -docket vote show DKT-V1 --json -docket vote result DKT-V1 --json -docket vote list --json --all # default: open proposals only -docket vote commit DKT-V1 --json --outcome "Approved: adopting Result" -docket vote close DKT-V1 --json --reason "decided out-of-band; operator ran the verb directly" -docket vote backfill-usage DKT-V1 --json --voter tribunal-security --unit output_tokens --quantity 48211 -docket vote link DKT-V1 --json --issue DKT-1 -docket vote unlink DKT-V1 --json --issue DKT-1 -``` - ---- - -## Workflow: Docs (`docket doc`) - -```bash -docket doc create --json -t "ADR-0003: SQLite over Postgres" -T adr -s accepted \ - -d "@docs/adr/0003-sqlite.md" # "@path" loads body from a file -docket doc create --json -t "Quick note" -d "-" # "-" reads body from stdin -docket doc show DOC-1 --json -docket doc show DOC-1 --json --rev 2 # a specific revision -docket doc list --json -T adr -s accepted -docket doc edit DOC-1 --json -s superseded -docket doc delete DOC-1 --json --force -docket doc link add DOC-1 --json --issue DKT-1 -docket doc link remove DOC-1 --json --issue DKT-1 -docket doc comment add DOC-1 --json -m "Needs a follow-up on migration path" -docket doc comment list DOC-1 --json -``` - -`--description`/`-d` on `doc create`/`doc edit` supports the same three -input modes as `issue create -d`: literal string, `@path/to/file` (loads -file contents, 1 MiB cap), or `-` (stdin, 1 MiB cap). - ---- - -## Workflow: Watch Mode - -Any watch-eligible command (see table above) can be re-run on an interval -instead of polling manually: - -```bash -docket issue list --json --watch --interval 5s -docket board --watch # human-mode live board, default 2s interval -docket vote result DKT-V1 --watch --interval 1s -``` - -`--watch` is rejected with `VALIDATION_ERROR` on any write command (e.g. -`docket issue create --watch` fails immediately). Watch mode runs until -`Ctrl-C` (SIGINT) or SIGTERM. - ---- - -## Workflow: Export / Import - -```bash -docket export --json -o json -f backup.json -docket export -o csv -f issues.csv -s todo -s in-progress -docket export -o markdown > issues.md - -docket import backup.json --json --merge # skip duplicates by ID -docket import backup.json --json --replace --yes # destructive: wipes DB first; --yes is required - # in every output mode, --json never substitutes (DKT-15) -docket import backup.json --json # default: requires an EMPTY database, else CONFLICT -``` - -`export` streams to stdout when `-f`/`--file` is omitted. `import` requires -`--merge` XOR `--replace`, or an empty database — passing both is a -`VALIDATION_ERROR`, and importing into a non-empty DB without either flag -is a `CONFLICT`. - ---- - -## Complete Command & Flag Reference - -Every flag below is transcribed directly from the `cmd.Flags().*` calls in -the corresponding `internal/cli/*.go` file's `init()`. "Req." marks flags -enforced via `cmd.Flags().MarkFlagRequired` (Cobra rejects the command -before `RunE` runs if absent) — distinct from flags that are merely -required *in JSON mode* by manual checks inside `RunE` (noted in Notes). - -### `docket issue` (alias `i`) — `internal/cli/issue.go` - -#### `docket issue create` — `issue_create.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--title` | `-t` | string | `""` | Required in `--json` mode | -| `--description` | `-d` | string | `""` | `"-"` reads from stdin | -| `--status` | `-s` | string | `"backlog"` | | -| `--priority` | `-p` | string | `"none"` | | -| `--type` | `-T` | string | `"task"` | | -| `--label` | `-l` | stringSlice | `nil` | repeatable | -| `--file` | `-f` | stringSlice | `nil` | repeatable | -| `--assignee` | `-a` | string | `""` | | -| `--parent` | — | string | `""` | parent issue ID | -| `--scope` | — | stringSlice | `nil` | repeatable; path glob this issue is expected to touch | -| `--idempotency-key` | — | string | `""` | replay protection; repeat returns the original issue | - -#### `docket issue edit [id]` — `issue_edit.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--title` | `-t` | string | `""` | only applied when explicitly set | -| `--description` | `-d` | string | `""` | `"-"` reads from stdin | -| `--status` | `-s` | string | `""` | | -| `--priority` | `-p` | string | `""` | | -| `--type` | `-T` | string | `""` | | -| `--assignee` | `-a` | string | `""` | | -| `--file` | `-f` | stringSlice | `nil` | repeatable; **replaces** existing file list | -| `--parent` | — | string | `""` | `"0"` or `"none"` clears parent | -| `--scope` | — | stringSlice | `nil` | repeatable; **replaces** the declaration, `--scope=` clears it | -| `--if-version` | — | int | `0` | apply only at this version; `CONFLICT` otherwise | - -**`--scope` is not `--file`.** `--file` records the concrete paths an issue -concerns, and `docket plan` uses them to split colliding work. `--scope` is a -list of path **globs** declaring what an issue is *expected* to touch — a -judgment made ahead of the work, which activation snapshots and which the -scheduler will use for mutual exclusion between steps. They differ in -cardinality, in semantics (actual vs. intended), and in matching rule (equality -vs. glob intersection). - -An issue created without `--scope` stores SQL `NULL`, not `[]`: "no scope -declared" and "declared to touch nothing" are different facts. An `issue edit` -that never mentions `--scope` leaves an earlier declaration alone. - -**Reading it back:** `issue show` and `issue list` carry `scope` **when the -issue declares one**, under plain `--json` as well as `--json=v2` — the one -amendment to the v1 freeze (DKT-55, see above). The three states are -distinguishable on the wire: no key at all is undeclared, `[]` is -declared-to-touch-nothing, and a populated array is the declaration. A -declared scope also survives `export`/`import` intact, `NULL` included. - -#### `docket issue show [id]` — `issue_show.go` - -No local flags. Watch-eligible. - -**It says when a run gave up on the issue** (DKT-404). Abandoning an issue -deliberately leaves its tracker status alone, so a `todo` or `review` an -abandon froze is byte-identical to work nobody has started — RUN-14's four -abandoned issues sat at `todo` for days, and a later session read two -operator-abandoned issues as "still parked on a gate that was never resolved" -and re-asked both decisions. `issue show` now prints a **`Run disposition`** -section — `work abandoned in RUN-32 by implement@0 (2026-08-20)` plus the -recorded reason **verbatim** — and `--json` carries `run_disposition` -`{run, disposition, by, reason, at}`, emitted **only when a run abandoned its -work**, so an ordinary issue's payload is unchanged. `by` is absent when an -operator abandoned the issue from outside the graph with `run abandon ---issue`, where no step decided it. - -It is the **latest** such ruling, keyed by ISSUE and not bound to any run: an -issue abandoned two runs ago and never resurfaced still reports the run that -stopped. It also survives `issue reopen` — the line is a dated fact about a -run, not a claim about the issue's current status, and it is usually the -context a reopened issue most needs. Earlier rulings stay in `events list`, -which is where a history belongs. - -#### `docket issue list` (alias `ls`) — `issue_list.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--status` | `-s` | stringSlice | `nil` | repeatable | -| `--priority` | `-p` | stringSlice | `nil` | repeatable | -| `--label` | `-l` | stringSlice | `nil` | repeatable | -| `--type` | `-T` | stringSlice | `nil` | repeatable | -| `--assignee` | `-a` | string | `""` | | -| `--parent` | — | string | `""` | | -| `--roots` | — | bool | `false` | root issues only | -| `--tree` | — | bool | `false` | indented hierarchy | -| `--sort` | — | string | `""` | `field:direction`, e.g. `priority:asc` | -| `--limit` | — | int | `50` | | -| `--all` | — | bool | `false` | include `done` issues | -| `--project` | — | string | `""` | list ANOTHER project's issues, by prefix, name, identity path, or row id (DKT-72, DKT-453) | -| `--run` | — | string | `""` | list one RUN's roster — the issues bound to `RUN-N` (DKT-405) | - -Watch-eligible. - -`--run RUN-N` is the run roster as a LISTING (DKT-405). Before it, the roster -existed only inside `run status --json` — a document to parse rather than a -listing to read — and `issue list --run` answered `unknown flag`. It shows the -whole roster **including `done` issues**: a roster is a closed set, and the -post-mortem case the flag exists for is mostly finished issues, so hiding them -would answer "which issues did RUN-14 carry" with "the ones it did not -finish". Ids render under the RUN's project prefix, for the `--project` reason -below. An unknown run is `NOT_FOUND`, not an unfiltered listing; `--run` and -`--project` together are refused, since a run belongs to one project and -silently picking one of the two scopes would be a lie. - -Listing is otherwise cwd-scoped: the project the working directory resolves to. -`--project` is the escape hatch a machine-global store needs — without it, -reading another project's issues means changing directory into it, and is -impossible for a project whose checkout is not on this machine. Ids render -under the NAMED project's prefix, not the caller's, for the same reason -`events list --all-projects` does (DKT-67): the prefix is the only thing on the -row that says whose issue it is. - -**The target resolves four ways** — exact `identity` path, numeric row `id`, -`name`, or display `prefix` (name and prefix case-insensitively) — every column -`docket project list` prints, through the same resolver `issue move --project` -and `project delete` use. The PREFIX is the key that matters most: it is the -only project identifier an issue id carries (`FLX-141`), so it is the one a -reader has actually seen, and until DKT-453 it was the one key this flag -refused — a conductor guessed it, got `NOT_FOUND`, and fell back to listing -every project's issues. An ambiguous name or prefix is a `VALIDATION_ERROR` -naming the candidates (id, name, identity) rather than a guess. - -#### `docket issue close [id]` — `issue_close.go` - -Shorthand for `move done`. - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--if-version` | — | int | `0` | apply only at this version; `CONFLICT` otherwise | - -Ends a live lease. The holder must supply its token (`DOCKET_TOKEN` or stdin); -a non-holder gets `AUTH_ERROR` (exit 5). An unclaimed issue needs no token. - -#### `docket issue claim ` — `issue_claim.go` - -Takes a lease and mints a capability token, returned exactly once. - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--owner` | — | string | — | **required**; identifies the lease holder | -| `--ttl` | — | duration | configured | lease duration; overrides the configured TTL | -| `--class` | — | string | `""` | executor class whose configured TTL applies (opaque) | - -Response (`--json`): `{"issue":"DKT-N","token":"<64 hex>","lease_expires_ms":N}`. -Under `--json=v2` it also carries `attempt` and `version`. Exit 4 if a live -lease is held; exit 2 if the issue does not exist. - -#### `docket issue heartbeat ` — `issue_heartbeat.go` - -Extends a lease you hold. Token via `DOCKET_TOKEN` or stdin. Does not change -`attempt` — a heartbeat is not a new claim. - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--ttl` | — | duration | configured | extension length | -| `--class` | — | string | `""` | executor class whose configured TTL applies | - -#### `docket issue release ` — `issue_release.go` - -Releases a lease you hold, returning the issue to the unclaimed pool -immediately. No local flags. Token via `DOCKET_TOKEN` or stdin. `attempt` -survives; the released token never works again. - -#### `docket issue move ` — `issue_move.go` - -Two modes: a **status move** (two positional args, `id` and target status) or -a **project migration** (`` plus `--project`, no status arg). - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--if-version` | — | int | `0` | apply only at this version; `CONFLICT` otherwise (enforced even on a no-op move). Status moves only — with `--project` it is a `VALIDATION_ERROR` | -| `--note` | — | string | `""` | why the issue moved; recorded as an issue **comment** in the **same transaction** as the move, so a refused move records no comment (DKT-480). Recorded even on a no-op move. Status moves only — with `--project` it is a `VALIDATION_ERROR` | -| `--project` | — | string | `""` | migrate the issue **and its whole sub-issue tree** to another project in the shared store | - -**Migration (`--project`)** re-homes work that landed in the wrong project — -most commonly a gap recorded by `step complete --gap-file`, which lands in the -run's own project unconditionally. The target resolves in order: exact -`identity`, then numeric `id`, then unique `name`, then unique `prefix` (the -name/prefix matches are case-insensitive) — the same resolver `issue list ---project` and `project delete` use (DKT-453); an ambiguous name or prefix is a -`VALIDATION_ERROR` naming the candidates. Labels re-map **by -name** into the target project (created there when missing, color preserved); -comments, relations, and activity ride along untouched — ids are store-wide, -so nothing referencing the issue goes stale. The response carries the target -project and the full list of migrated ids. - -| Refusal | Code | Exit | -|---|---|---| -| a status arg **and** `--project` together | `VALIDATION_ERROR` | 3 | -| the issue has a parent — a sub-issue migrates with its root, never alone (migrate the root, or `issue edit --parent none` first) | `VALIDATION_ERROR` | 3 | -| the issue (or any sub-issue) belongs to a run — a run's snapshots and steps are project-scoped bookkeeping | `CONFLICT` | 4 | -| issue not found | `NOT_FOUND` | 2 | - -#### `docket issue reopen [id]` — `issue_reopen.go` - -Only transitions if currently `done`, sets status to `backlog`. - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--if-version` | — | int | `0` | apply only at this version; `CONFLICT` otherwise | - -#### `docket issue delete ` — `issue_delete.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--force` | `-f` | bool | `false` | cascade-delete sub-issues; mutually exclusive with `--orphan` | -| `--yes` | `-y` | bool | `false` | alias for `--force` (DKT-72) | -| `--orphan` | — | bool | `false` | promote sub-issues to root; mutually exclusive with `--force` | - -`--yes` is an ALIAS, not a third behavior: the confirmation it answers is a -three-way choice (cascade, orphan, cancel), and a flag meaning "yes" without -saying to what would have to pick one silently. `--force` names the choice; -`--yes` is the spelling scripted cleanup reaches for. An issue with no -sub-issues never asks anything and needs neither flag. - -#### `docket issue log [id]` — `issue_log.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--limit` | — | int | `20` | clamped to min 1 | - -Watch-eligible. - -#### `docket issue comment add [id]` / `docket issue comment list [id]` — `issue_comment.go`, `issue_comment_list.go` - -| Flag | Short | Type | Default | Notes | -|---|---|---|---|---| -| `--message` | `-m` | string | `""` | (`add` only) required in `--json` mode if stdin isn't piped | -| `--idempotency-key` | — | string | `""` | (`add` only) replay protection | - -`comment list` has no local flags; watch-eligible. - -#### `docket issue file add/remove/list` — `issue_file.go` - -`add ...` and `remove ...` take -`cobra.MinimumNArgs(2)` — no flags. `list ` takes `cobra.ExactArgs(1)` — -no flags. - -#### `docket issue link add/remove/list` — `issue_link.go` - -`add ` and `remove ` -take `cobra.ExactArgs(3)` — no flags. `list ` — no flags. - -#### `docket issue label add/rm/list/delete` — `issue_label.go` - -| Command | Flag | Short | Type | Default | -|---|---|---|---|---| -| `add
diff --git a/internal/cli/board.go b/internal/cli/board.go index f36ef52b..d8bd1118 100644 --- a/internal/cli/board.go +++ b/internal/cli/board.go @@ -11,10 +11,14 @@ import ( ) // boardColumn represents a single status column in the board JSON output. +// +// Issues is typed `any` for the same reason listResult's is: summary rows +// (issueRowsPayload) by default, the full issue shape (issueListPayload) under +// `--with-body` (DKT-1053; see issue_row.go). type boardColumn struct { - Status string `json:"status"` - Count int `json:"count"` - Issues []*model.Issue `json:"issues"` + Status string `json:"status"` + Count int `json:"count"` + Issues any `json:"issues"` } // boardResult is the JSON output structure for the board command. @@ -37,6 +41,7 @@ func runBoard(cmd *cobra.Command, args []string, w *output.Writer) error { priorities, _ := cmd.Flags().GetStringSlice("priority") assignee, _ := cmd.Flags().GetString("assignee") expand, _ := cmd.Flags().GetBool("expand") + withBody, _ := cmd.Flags().GetBool("with-body") // Validate filter enum values. for _, p := range priorities { @@ -86,7 +91,7 @@ func runBoard(cmd *cobra.Command, args []string, w *output.Writer) error { columns = append(columns, boardColumn{ Status: string(status), Count: len(col), - Issues: col, + Issues: issuesPayload(col, withBody), }) } @@ -125,5 +130,6 @@ func init() { boardCmd.Flags().StringSliceP("priority", "p", nil, "Filter by priority (repeatable)") boardCmd.Flags().StringP("assignee", "a", "", "Filter by assignee") boardCmd.Flags().Bool("expand", false, "Show sub-issues individually instead of rolling up") + boardCmd.Flags().Bool("with-body", false, withBodyHelp) rootCmd.AddCommand(boardCmd) } diff --git a/internal/cli/issue_list.go b/internal/cli/issue_list.go index bd4af536..4aa1f05f 100644 --- a/internal/cli/issue_list.go +++ b/internal/cli/issue_list.go @@ -13,27 +13,47 @@ import ( "github.com/spf13/cobra" ) +// listResult is `issue list`'s payload. `Issues` holds SUMMARY rows +// (issueRowsPayload) by default and the full issue shape (issueListPayload) +// under `--with-body`, which is why it is typed `any`: the two shapes differ, +// and the default one is what keeps a filtered listing small enough to read +// (DKT-1053; see issue_row.go). Both payloads marshal the v1 array and +// implement output.Versioned, so the v2 envelope reaches each row's version +// either way. type listResult struct { - Issues []*model.Issue `json:"issues"` - Total int `json:"total"` + Issues any `json:"issues"` + Total int `json:"total"` // limit is the effective --limit, retained (unexported, so v1 output is // untouched) to compute truncation for the v2 envelope. limit int + // count is the number of rows in Issues, kept alongside because Issues is + // `any`. + count int } // listResult implements output.Collection for the v2 envelope. Total comes // from a COUNT(*) that ignores LIMIT, so it is already the true pre-limit // count and truncation is directly computable. -func (r listResult) CollectionItems() any { return issueListPayload{issues: r.Issues} } +func (r listResult) CollectionItems() any { return r.Issues } func (r listResult) CollectionTotal() int { return r.Total } func (r listResult) CollectionTruncated() bool { - return output.IsTruncated(r.limit, r.Total, len(r.Issues)) + return output.IsTruncated(r.limit, r.Total, r.count) } var listCmd = &cobra.Command{ Use: "list", Short: "List issues", Aliases: []string{"ls"}, + Long: `List issues as summary rows. + +Under --json each row carries every issue field except description — id +(and its alias issue), parent_id, title, status, priority, kind, assignee, +labels, files, docs, created_at, updated_at, and scope and resolution when +set — plus description_bytes, the length of the description it does not +carry. A listing is for picking; read the issue you picked with +docket issue show. Pass --with-body to have every row carry its full +description instead. next, plan and board emit the same rows and take the +same flag.`, RunE: func(cmd *cobra.Command, args []string) error { return watchable(cmd, args, runIssueList) }, @@ -55,6 +75,7 @@ func runIssueList(cmd *cobra.Command, args []string, w *output.Writer) error { all, _ := cmd.Flags().GetBool("all") runRef, _ := cmd.Flags().GetString("run") projectRef, _ := cmd.Flags().GetString("project") + withBody, _ := cmd.Flags().GetBool("with-body") if err := validateLimit(cmd, limit); err != nil { return err @@ -189,7 +210,12 @@ func runIssueList(cmd *cobra.Command, args []string, w *output.Writer) error { return cmdErr(fmt.Errorf("fetching linked docs: %w", err), output.ErrGeneral) } - result := listResult{Issues: issues, Total: total, limit: limit} + result := listResult{ + Issues: issuesPayload(issues, withBody), + Total: total, + limit: limit, + count: len(issues), + } // Fetch parent issues and sub-issue progress for the grouped display. // Only needed for human-readable output (JSON stays flat). @@ -338,5 +364,6 @@ func init() { "roster, done ones included") listCmd.Flags().Int("limit", 50, "Maximum number of results") listCmd.Flags().Bool("all", false, "Include done issues") + listCmd.Flags().Bool("with-body", false, withBodyHelp) issueCmd.AddCommand(listCmd) } diff --git a/internal/cli/issue_row.go b/internal/cli/issue_row.go new file mode 100644 index 00000000..7e43821b --- /dev/null +++ b/internal/cli/issue_row.go @@ -0,0 +1,161 @@ +package cli + +import ( + "encoding/json" + "time" + + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// issueRow is one SUMMARY row of a list-shaped issue verb — `issue list`, +// `next`, `plan`, `board` — and the reason those verbs are usable as pickers +// at all (DKT-1053, the issue-side twin of DKT-1045). +// +// It is the frozen v1 issue shape with exactly one substitution: `description` +// is replaced by `description_bytes`. Every other key keeps its name, its +// position and its conditionality — `issue` mirrors `id` (DKT-452), `scope` +// appears when declared (DKT-55), `resolution` when set (DKT-245) — so a +// consumer that selects any key other than `description` off a list row reads +// exactly what it read before. +// +// Before this every row carried its whole description, so a filtered listing +// of a few dozen issues with real descriptions ran to tens of kilobytes — +// larger than a harness tool result may be — to answer a question ("which of +// these do I want?") that needs ids, titles and status. The description now +// lives where ONE issue is asked for: `issue show`. The byte count stands in +// for it so a caller can still tell a stub from a substantial issue, and an +// empty description from an absent one. +// +// It is a CLI-side projection rather than a change to model.Issue's marshaler +// because that marshaler is the frozen v1 wire format every single-issue verb +// (`issue show`, `create`, `edit`, `claim`, ...) emits, and those verbs are +// asked for one issue's whole body. +type issueRow struct { + ID string `json:"id"` + Issue string `json:"issue"` + ParentID *string `json:"parent_id,omitempty"` + Title string `json:"title"` + DescriptionBytes int `json:"description_bytes"` + Status string `json:"status"` + Priority string `json:"priority"` + Kind string `json:"kind"` + Assignee string `json:"assignee"` + Labels []string `json:"labels"` + Files []string `json:"files"` + Docs []model.DocRef `json:"docs"` + Scope *[]string `json:"scope,omitempty"` + Resolution string `json:"resolution,omitempty"` + CreatedAt string `json:"created_at"` + UpdatedAt string `json:"updated_at"` +} + +// issueRowV2 is the summary row under --json=v2: the v1 row plus the CAS +// version, plus the lease when one is held — the same two additions +// model.VersionedIssue makes to the full shape, so a v2 consumer that read +// `version` or `lease` off a full list row reads them off a summary row too. +type issueRowV2 struct { + issueRow + Version int `json:"version"` + Lease any `json:"lease,omitempty"` +} + +// summarizeIssue projects an issue onto its list row. Ids, timestamps and the +// nil-slice normalization match model.Issue's marshaler exactly, so the summary +// row and the `--with-body` row for one issue differ only in the field this +// type exists to drop. +func summarizeIssue(i *model.Issue) issueRow { + labels := i.Labels + if labels == nil { + labels = []string{} + } + files := i.Files + if files == nil { + files = []string{} + } + docs := i.Docs + if docs == nil { + docs = []model.DocRef{} + } + + id := model.FormatID(i.ID) + row := issueRow{ + ID: id, + Issue: id, + Title: i.Title, + DescriptionBytes: len(i.Description), + Status: string(i.Status), + Priority: string(i.Priority), + Kind: string(i.Kind), + Assignee: i.Assignee, + Labels: labels, + Files: files, + Docs: docs, + Scope: i.Scope, + Resolution: i.Resolution, + CreatedAt: i.CreatedAt.UTC().Format(time.RFC3339), + UpdatedAt: i.UpdatedAt.UTC().Format(time.RFC3339), + } + if i.ParentID != nil { + pid := model.FormatID(*i.ParentID) + row.ParentID = &pid + } + return row +} + +// summarizeIssues projects a slice. The result is never nil: a listing that +// matches nothing must emit `[]`, not `null`. +func summarizeIssues(issues []*model.Issue) []issueRow { + rows := make([]issueRow, 0, len(issues)) + for _, i := range issues { + rows = append(rows, summarizeIssue(i)) + } + return rows +} + +// summarizeIssuesV2 is summarizeIssues with each row's version and lease. The +// lease's `live` flag is computed at nowMS, per read, the way +// model.VersionedIssue computes it (engine-spec.md §2: effective status at +// read time, no write). +func summarizeIssuesV2(issues []*model.Issue, nowMS int64) []issueRowV2 { + rows := make([]issueRowV2, 0, len(issues)) + for _, i := range issues { + rows = append(rows, issueRowV2{ + issueRow: summarizeIssue(i), + Version: i.Version, + Lease: i.Lease.LeaseWire(nowMS), + }) + } + return rows +} + +// issueRowsPayload wraps issues for emission as summary rows: v1 marshals +// []issueRow, and the v2 collection envelope — which consults output.Versioned +// on the items CONTAINER, as issueListPayload's comment explains — gets +// []issueRowV2. It is issueListPayload's summary-shaped twin. +type issueRowsPayload struct{ issues []*model.Issue } + +// MarshalJSON emits the v1 summary-row array. Never null. +func (p issueRowsPayload) MarshalJSON() ([]byte, error) { + return json.Marshal(summarizeIssues(p.issues)) +} + +// VersionedPayload implements output.Versioned for a list of summary rows. +func (p issueRowsPayload) VersionedPayload() any { + return summarizeIssuesV2(p.issues, model.NowMS()) +} + +// issuesPayload picks a list-shaped verb's row shape: summary rows by default, +// and under `--with-body` the pre-DKT-1053 full issue shape, byte-identical to +// what the verb used to print — for the one caller that wants every +// description in a single call (an export, a grep across a backlog). +func issuesPayload(issues []*model.Issue, withBody bool) any { + if withBody { + return issueListPayload{issues: issues} + } + return issueRowsPayload{issues: issues} +} + +// withBodyHelp is the `--with-body` flag's help text, shared by the four +// list-shaped verbs so `--help` describes the same escape hatch everywhere. +const withBodyHelp = "Include each issue's full description in --json output " + + "(rows carry description_bytes instead by default; use issue show for one issue's body)" diff --git a/internal/cli/issue_rows_test.go b/internal/cli/issue_rows_test.go new file mode 100644 index 00000000..6c7b12b0 --- /dev/null +++ b/internal/cli/issue_rows_test.go @@ -0,0 +1,551 @@ +package cli + +import ( + "bytes" + "database/sql" + "encoding/json" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// DKT-1053: list-shaped issue verbs emit summary rows. +// +// These tests are the issue-side twin of doc_list_test.go (DKT-1045): the +// same corpus shape — a dozen rows with real multi-kilobyte bodies — measured +// on every verb that lists issues, plus the promise that makes the trim safe: +// `issue show` still carries the whole description. + +// listCmdWithBody is listCmdWithDB plus the --with-body flag, set as asked. +func listCmdWithBody(conn *sql.DB, withBody bool) *cobra.Command { + cmd := listCmdWithDB(conn) + cmd.Flags().Bool("with-body", withBody, "") + return cmd +} + +// boardCmdWithDB registers the flags runBoard reads. +func boardCmdWithDB(conn *sql.DB, withBody bool) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().StringSlice("label", nil, "") + cmd.Flags().StringSlice("priority", nil, "") + cmd.Flags().String("assignee", "", "") + cmd.Flags().Bool("expand", false, "") + cmd.Flags().Bool("with-body", withBody, "") + return cmd +} + +// issueDescription is a stand-in for a real issue body: the claim, remedy and +// acceptance criteria a planned issue actually carries. Size is what makes the +// corpus a reproduction of DKT-1053 rather than a shape test, so it is long. +func issueDescription(n int) string { + var b strings.Builder + fmt.Fprintf(&b, "## Claim %d\n\n", n) + for i := 0; i < 60; i++ { + fmt.Fprintf(&b, "Criterion %d: the conductor dispatched the wave, the executor "+ + "reported, and the gate seated a panel of three judges.\n", i) + } + return b.String() +} + +// seedDescribedIssues creates 11 todo feature issues with real descriptions, +// plus one bug so the type filter has something to exclude and one done issue +// so the default listing has something to hide. It returns the 11 +// descriptions in id order. +func seedDescribedIssues(t *testing.T, conn *sql.DB) []string { + t.Helper() + + descriptions := make([]string, 0, 11) + for i := 1; i <= 11; i++ { + desc := issueDescription(i) + descriptions = append(descriptions, desc) + _, err := db.CreateIssue(conn, &model.Issue{ + Title: fmt.Sprintf("Planned issue %d", i), + Description: desc, + Status: model.StatusTodo, + Priority: model.PriorityHigh, + Kind: model.IssueKindFeature, + }, nil, nil) + testsupport.Must(t, err, "CreateIssue(%d): %v", i, err) + } + + _, err := db.CreateIssue(conn, &model.Issue{ + Title: "Not a feature", + Description: issueDescription(98), + Status: model.StatusTodo, + Priority: model.PriorityLow, + Kind: model.IssueKindBug, + }, nil, nil) + testsupport.Must(t, err, "CreateIssue(bug): %v", err) + + _, err = db.CreateIssue(conn, &model.Issue{ + Title: "Already finished", + Description: issueDescription(99), + Status: model.StatusDone, + Priority: model.PriorityLow, + Kind: model.IssueKindFeature, + }, nil, nil) + testsupport.Must(t, err, "CreateIssue(done): %v", err) + + return descriptions +} + +// summaryRowKeys is every key a summary row must carry: the frozen v1 keys +// minus `description`, plus `description_bytes`. +var summaryRowKeys = []string{ + "id", "issue", "title", "description_bytes", "status", "priority", "kind", + "assignee", "labels", "files", "docs", "created_at", "updated_at", +} + +// assertSummaryRows checks that every row is a summary row whose +// description_bytes matches the description it does not carry. +func assertSummaryRows(t *testing.T, rows []map[string]json.RawMessage, descriptions []string) { + t.Helper() + if len(rows) != len(descriptions) { + t.Fatalf("rows = %d, want %d", len(rows), len(descriptions)) + } + for i, row := range rows { + if _, ok := row["description"]; ok { + t.Errorf("row %d carries a description key; listings must be description-free", i) + } + for _, key := range summaryRowKeys { + if _, ok := row[key]; !ok { + t.Errorf("row %d is missing %q; got keys %v", i, key, keysOf(row)) + } + } + var descBytes int + if err := json.Unmarshal(row["description_bytes"], &descBytes); err != nil { + t.Fatalf("row %d description_bytes: %v", i, err) + } + if descBytes != len(descriptions[i]) { + t.Errorf("row %d description_bytes = %d, want %d", i, descBytes, len(descriptions[i])) + } + } +} + +// assertFullRows checks that every row is the pre-DKT-1053 shape: the whole +// description present, and no description_bytes standing in for it. +func assertFullRows(t *testing.T, rows []map[string]json.RawMessage, descriptions []string) { + t.Helper() + if len(rows) != len(descriptions) { + t.Fatalf("rows = %d, want %d", len(rows), len(descriptions)) + } + for i, row := range rows { + if _, ok := row["description_bytes"]; ok { + t.Errorf("row %d carries description_bytes under --with-body; the full shape has no such key", i) + } + var desc string + if err := json.Unmarshal(row["description"], &desc); err != nil { + t.Fatalf("row %d description: %v", i, err) + } + if desc != descriptions[i] { + t.Errorf("row %d description = %d bytes, want the full %d-byte description", + i, len(desc), len(descriptions[i])) + } + } +} + +// TestIssueListJSON_IsSummaryRows is DKT-1053's acceptance criterion, +// measured: a type-filtered `issue list --json` over eleven issues with real +// descriptions stays under DKT-1045's 4KB bar, and no row carries a +// description. Before the fix the same corpus produced tens of kilobytes. +func TestIssueListJSON_IsSummaryRows(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + cmd := listCmdWithBody(conn, false) + setFlags(t, cmd, map[string]string{"type": "feature", "status": "todo", "sort": "id:asc"}) + w, buf := bufWriter(true) + if err := runIssueList(cmd, nil, w); err != nil { + t.Fatalf("runIssueList: %v", err) + } + + if buf.Len() >= 4096 { + t.Errorf("issue list -T feature -s todo --json = %d bytes, want < 4096", buf.Len()) + } + + var env struct { + Data struct { + Issues []map[string]json.RawMessage `json:"issues"` + Total int `json:"total"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Total != 11 { + t.Errorf("total = %d, want 11", env.Data.Total) + } + assertSummaryRows(t, env.Data.Issues, descriptions) + + // The description is absent, not emptied: no row may smuggle it back. + if strings.Contains(buf.String(), "the conductor dispatched the wave") { + t.Error("listing output contains description prose; the description must live in issue show") + } +} + +// TestIssueListJSON_WithBodyRestoresDescriptions pins the escape hatch: +// --with-body re-emits the pre-DKT-1053 row, description and all, for the +// caller that genuinely wants every description in one call. +func TestIssueListJSON_WithBodyRestoresDescriptions(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + cmd := listCmdWithBody(conn, true) + setFlags(t, cmd, map[string]string{"type": "feature", "status": "todo", "sort": "id:asc"}) + w, buf := bufWriter(true) + if err := runIssueList(cmd, nil, w); err != nil { + t.Fatalf("runIssueList --with-body: %v", err) + } + + var env struct { + Data struct { + Issues []map[string]json.RawMessage `json:"issues"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + assertFullRows(t, env.Data.Issues, descriptions) + if buf.Len() < 4096 { + t.Errorf("--with-body output = %d bytes; the descriptions did not come back", buf.Len()) + } + + // The full row is model.Issue's own marshaler, byte for byte: what this + // verb printed by default before DKT-1053. + issues, _, err := db.ListIssues(conn, db.ListOptions{ + ProjectID: db.DefaultProjectID, Types: []string{"feature"}, + Statuses: []string{"todo"}, Sort: "id", SortDir: "asc", Limit: 50, + }) + testsupport.Must(t, err, "ListIssues: %v", err) + testsupport.Must(t, db.HydrateDocs(conn, issues), "HydrateDocs: %v", err) + want, err := json.Marshal(issues) + testsupport.Must(t, err, "Marshal: %v", err) + var got struct { + Data struct { + Issues json.RawMessage `json:"issues"` + } `json:"data"` + } + testsupport.Must(t, json.Unmarshal(buf.Bytes(), &got), "unmarshal: %v", nil) + if !bytes.Equal(bytes.TrimSpace(got.Data.Issues), want) { + t.Errorf("--with-body rows are not model.Issue's v1 marshaling:\n got %s\nwant %s", + got.Data.Issues, want) + } +} + +// TestIssueListJSON_SummaryRowKeepsConditionalKeys: the row drops exactly one +// key. `parent_id`, and `scope` when declared, survive the projection — a +// consumer selecting either off a list row (test_zd_jsonv2.sh does, for +// scope) reads what it read before. +func TestIssueListJSON_SummaryRowKeepsConditionalKeys(t *testing.T) { + conn := newTestDB(t) + parent := createIssue(t, conn, "parent", model.StatusTodo, model.PriorityHigh) + child, err := db.CreateIssue(conn, &model.Issue{ + Title: "scoped child", + ParentID: &parent, + Status: model.StatusTodo, + Priority: model.PriorityHigh, + Kind: model.IssueKindTask, + }, []string{"engine"}, nil) + testsupport.Must(t, err, "CreateIssue(child): %v", err) + testsupport.Must(t, db.SetIssueScopeGlobs(conn, child, `["internal/db/**"]`), + "SetIssueScopeGlobs: %v", nil) + + cmd := listCmdWithBody(conn, false) + w, buf := bufWriter(true) + if err := runIssueList(cmd, nil, w); err != nil { + t.Fatalf("runIssueList: %v", err) + } + + var env struct { + Data struct { + Issues []struct { + ID string `json:"id"` + ParentID *string `json:"parent_id"` + Labels []string `json:"labels"` + Scope json.RawMessage `json:"scope"` + } `json:"issues"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + var found bool + for _, row := range env.Data.Issues { + switch row.ID { + case model.FormatID(child): + found = true + if row.ParentID == nil || *row.ParentID != model.FormatID(parent) { + t.Errorf("child parent_id = %v, want %s", row.ParentID, model.FormatID(parent)) + } + if string(row.Scope) != `["internal/db/**"]` { + t.Errorf("child scope = %s, want [\"internal/db/**\"]", row.Scope) + } + if len(row.Labels) != 1 || row.Labels[0] != "engine" { + t.Errorf("child labels = %v, want [engine]", row.Labels) + } + case model.FormatID(parent): + if row.Scope != nil { + t.Errorf("parent declares no scope but its row carries scope = %s", row.Scope) + } + if row.ParentID != nil { + t.Errorf("root row carries parent_id = %q", *row.ParentID) + } + } + } + if !found { + t.Fatalf("child %s not listed:\n%s", model.FormatID(child), buf.String()) + } +} + +// TestIssueListJSON_EmptyIsAnArray guards the nil-slice trap on the summary +// path: a listing that matches nothing must emit `[]`, not `null`. +func TestIssueListJSON_EmptyIsAnArray(t *testing.T) { + conn := newTestDB(t) + + for _, withBody := range []bool{false, true} { + cmd := listCmdWithBody(conn, withBody) + w, buf := bufWriter(true) + if err := runIssueList(cmd, nil, w); err != nil { + t.Fatalf("runIssueList (with-body=%v): %v", withBody, err) + } + var env struct { + Data struct { + Issues json.RawMessage `json:"issues"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if got := string(env.Data.Issues); got != "[]" { + t.Errorf("issues (with-body=%v) = %s, want []", withBody, got) + } + } +} + +// TestIssueListJSONV2_SummaryRowsCarryVersion is the v2 half: the collection +// envelope still counts and truncates across the payload change, and each +// summary item carries the CAS `version` a VersionedIssue carried — the one +// thing a v2 list consumer reads that a v1 row does not have. +func TestIssueListJSONV2_SummaryRowsCarryVersion(t *testing.T) { + for _, withBody := range []bool{false, true} { + t.Run(fmt.Sprintf("with-body=%v", withBody), func(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + cmd := listCmdWithBody(conn, withBody) + setFlags(t, cmd, map[string]string{ + "type": "feature", "status": "todo", "sort": "id:asc", "limit": "5", + }) + w, buf := bufWriter(true) + w.JSONVersion = output.JSONV2 + if err := runIssueList(cmd, nil, w); err != nil { + t.Fatalf("runIssueList --json=v2 --limit 5: %v", err) + } + + var env struct { + Data struct { + Items []map[string]json.RawMessage `json:"items"` + Total int `json:"total"` + Truncated bool `json:"truncated"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if len(env.Data.Items) != 5 || env.Data.Total != 11 || !env.Data.Truncated { + t.Fatalf("v2 envelope = %d items, total %d, truncated %v; want 5, 11, true", + len(env.Data.Items), env.Data.Total, env.Data.Truncated) + } + if withBody { + assertFullRows(t, env.Data.Items, descriptions[:5]) + } else { + assertSummaryRows(t, env.Data.Items, descriptions[:5]) + } + for i, item := range env.Data.Items { + var version int + raw, ok := item["version"] + if !ok { + t.Fatalf("v2 item %d has no version key; keys = %v", i, keysOf(item)) + } + if err := json.Unmarshal(raw, &version); err != nil || version < 1 { + t.Errorf("v2 item %d version = %s, want a positive CAS version", i, raw) + } + if _, ok := item["lease"]; ok { + t.Errorf("v2 item %d carries a lease key while unclaimed", i) + } + } + }) + } +} + +// TestNextJSON_IsSummaryRows: `next` lists the same rows `issue list` does, +// with the same escape hatch. +func TestNextJSON_IsSummaryRows(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + for _, withBody := range []bool{false, true} { + cmd := nextCmdWithDB(conn, 50) + cmd.Flags().Bool("with-body", withBody, "") + setFlags(t, cmd, map[string]string{"type": "feature"}) + w, buf := bufWriter(true) + if err := runNext(cmd, nil, w); err != nil { + t.Fatalf("runNext (with-body=%v): %v", withBody, err) + } + var env struct { + Data struct { + Issues []map[string]json.RawMessage `json:"issues"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if withBody { + assertFullRows(t, env.Data.Issues, descriptions) + continue + } + assertSummaryRows(t, env.Data.Issues, descriptions) + if buf.Len() >= 4096 { + t.Errorf("next -T feature --json = %d bytes, want < 4096", buf.Len()) + } + } +} + +// TestPlanJSON_IsSummaryRows: a plan row is the shared summary row plus +// `blocked_by`; --with-body re-emits the pre-DKT-1053 plan row. +func TestPlanJSON_IsSummaryRows(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + for _, withBody := range []bool{false, true} { + cmd := planCmdWithDB(conn) + cmd.Flags().Bool("with-body", withBody, "") + setFlags(t, cmd, map[string]string{"type": "feature"}) + w, buf := bufWriter(true) + if err := runPlan(cmd, nil, w); err != nil { + t.Fatalf("runPlan (with-body=%v): %v", withBody, err) + } + var env struct { + Data struct { + Phases []struct { + Issues []map[string]json.RawMessage `json:"issues"` + } `json:"phases"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + var rows []map[string]json.RawMessage + for _, phase := range env.Data.Phases { + rows = append(rows, phase.Issues...) + } + for i, row := range rows { + if got := string(row["blocked_by"]); got != "[]" { + t.Errorf("row %d blocked_by = %s, want [] (with-body=%v)", i, got, withBody) + } + } + if withBody { + assertFullRows(t, rows, descriptions) + continue + } + assertSummaryRows(t, rows, descriptions) + if buf.Len() >= 4096 { + t.Errorf("plan -T feature --json = %d bytes, want < 4096", buf.Len()) + } + } +} + +// TestBoardJSON_IsSummaryRows: every column's rows are summary rows; empty +// columns stay `[]`; --with-body restores the full shape. +func TestBoardJSON_IsSummaryRows(t *testing.T) { + conn := newTestDB(t) + descriptions := seedDescribedIssues(t, conn) + + for _, withBody := range []bool{false, true} { + cmd := boardCmdWithDB(conn, withBody) + setFlags(t, cmd, map[string]string{"priority": "high"}) + w, buf := bufWriter(true) + if err := runBoard(cmd, nil, w); err != nil { + t.Fatalf("runBoard (with-body=%v): %v", withBody, err) + } + var env struct { + Data struct { + Columns []struct { + Status string `json:"status"` + Count int `json:"count"` + Issues json.RawMessage `json:"issues"` + } `json:"columns"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + var todo []map[string]json.RawMessage + for _, col := range env.Data.Columns { + if col.Status == string(model.StatusTodo) { + if err := json.Unmarshal(col.Issues, &todo); err != nil { + t.Fatalf("todo column: %v", err) + } + continue + } + if got := string(col.Issues); got != "[]" { + t.Errorf("%s column issues = %s, want [] (with-body=%v)", col.Status, got, withBody) + } + } + if withBody { + assertFullRows(t, todo, descriptions) + continue + } + assertSummaryRows(t, todo, descriptions) + if buf.Len() >= 4096 { + t.Errorf("board -p high --json = %d bytes, want < 4096", buf.Len()) + } + } +} + +// TestIssueShowJSON_StillCarriesTheDescription is the other half of DKT-1053: +// the description moved OUT of the listings on the promise that `issue show` +// still has it. This test is what makes that promise checkable. +func TestIssueShowJSON_StillCarriesTheDescription(t *testing.T) { + conn := newTestDB(t) + desc := issueDescription(1) + id, err := db.CreateIssue(conn, &model.Issue{ + Title: "Planned issue 1", + Description: desc, + Status: model.StatusTodo, + Priority: model.PriorityHigh, + Kind: model.IssueKindFeature, + }, nil, nil) + testsupport.Must(t, err, "CreateIssue: %v", err) + + for _, version := range []output.JSONVersion{output.JSONV1, output.JSONV2} { + w, buf := bufWriter(true) + w.JSONVersion = version + if err := runIssueShow(cmdWithDB(conn), []string{model.FormatID(id)}, w); err != nil { + t.Fatalf("runIssueShow: %v", err) + } + var env struct { + Data map[string]json.RawMessage `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + var got string + if err := json.Unmarshal(env.Data["description"], &got); err != nil { + t.Fatalf("issue show description: %v", err) + } + if got != desc { + t.Errorf("issue show description = %d bytes, want the full %d-byte description", + len(got), len(desc)) + } + if _, ok := env.Data["description_bytes"]; ok { + t.Error("issue show grew a description_bytes key; the detail view is unchanged") + } + } +} diff --git a/internal/cli/next.go b/internal/cli/next.go index a0387ef8..bb5b3a8e 100644 --- a/internal/cli/next.go +++ b/internal/cli/next.go @@ -12,9 +12,13 @@ import ( "github.com/spf13/cobra" ) +// nextResult is `next`'s issue-mode payload. `Issues` is typed `any` for the +// same reason listResult's is: summary rows (issueRowsPayload) by default, +// the full issue shape (issueListPayload) under `--with-body` (DKT-1053; see +// issue_row.go). type nextResult struct { - Issues []*model.Issue `json:"issues"` - Total int `json:"total"` + Issues any `json:"issues"` + Total int `json:"total"` // readyTotal is the size of the ready set BEFORE --limit truncated it, and // limit the effective limit. Both are unexported so the v1 payload is // untouched: v1's Total is len(Issues) — a post-limit count that cannot @@ -22,14 +26,17 @@ type nextResult struct { // That is the silent drop the v2 envelope exists to close. readyTotal int limit int + // count is the number of rows in Issues, kept alongside because Issues is + // `any`. + count int } // nextResult implements output.Collection for the v2 envelope, reporting the // honest pre-limit total rather than v1's len(Issues). -func (r nextResult) CollectionItems() any { return issueListPayload{issues: r.Issues} } +func (r nextResult) CollectionItems() any { return r.Issues } func (r nextResult) CollectionTotal() int { return r.readyTotal } func (r nextResult) CollectionTruncated() bool { - return output.IsTruncated(r.limit, r.readyTotal, len(r.Issues)) + return output.IsTruncated(r.limit, r.readyTotal, r.count) } var nextCmd = &cobra.Command{ @@ -63,6 +70,7 @@ func runNextIssues(cmd *cobra.Command, args []string, w *output.Writer) error { labels, _ := cmd.Flags().GetStringSlice("label") types, _ := cmd.Flags().GetStringSlice("type") limit, _ := cmd.Flags().GetInt("limit") + withBody, _ := cmd.Flags().GetBool("with-body") if err := validateLimit(cmd, limit); err != nil { return err @@ -127,16 +135,22 @@ func runNextIssues(cmd *cobra.Command, args []string, w *output.Writer) error { return cmdErr(fmt.Errorf("fetching linked docs: %w", err), output.ErrGeneral) } - // FindReady/filterReady return a nil slice when nothing is ready, and - // nextResult has no custom MarshalJSON (unlike issueListPayload) — a nil - // `ready` would serialize `.data.issues` as JSON null instead of `[]`. + // FindReady/filterReady return a nil slice when nothing is ready. Both row + // payloads marshal a nil slice as `[]`, but the human table is handed + // `ready` directly, so it is normalized here once for every consumer. if ready == nil { ready = []*model.Issue{} } // Total stays len(ready) — the v1 field is frozen. The honest pre-limit // count rides in readyTotal and surfaces only under --json=v2. - result := nextResult{Issues: ready, Total: len(ready), readyTotal: readyTotal, limit: limit} + result := nextResult{ + Issues: issuesPayload(ready, withBody), + Total: len(ready), + readyTotal: readyTotal, + limit: limit, + count: len(ready), + } var message string if !w.JSONMode { @@ -200,5 +214,6 @@ func init() { // is exactly what it was: a workflow-free repo never passes this flag and // never leaves the issue-mode path. nextCmd.Flags().String("run", "", "Show ready STEPS of this run instead of ready issues") + nextCmd.Flags().Bool("with-body", false, withBodyHelp) rootCmd.AddCommand(nextCmd) } diff --git a/internal/cli/plan.go b/internal/cli/plan.go index 247e0232..4be86f5d 100644 --- a/internal/cli/plan.go +++ b/internal/cli/plan.go @@ -29,10 +29,28 @@ type planPhaseJSON struct { type planIssue struct { Issue *model.Issue BlockedBy []string + // withBody selects the full pre-DKT-1053 row (planIssueJSON) over the + // summary row (planIssueRowJSON). Unexported: it is a rendering choice, + // never a wire field. + withBody bool } -// planIssueJSON is the explicit JSON wire format for planIssue, avoiding the -// fragile marshal-unmarshal-remarshal pattern (mirrors showResultJSON). +// planIssueRowJSON is `plan`'s default row: the summary row every +// list-shaped verb emits (issue_row.go), plus `blocked_by` (DKT-1053). +// +// Sharing issueRow means a plan row now also carries the keys the shared row +// has and planIssueJSON never gained — `issue` (DKT-452), and `scope` and +// `resolution` when set (DKT-55, DKT-245). Those are additions a consumer of +// the existing keys cannot notice; `description` is the only key that moved. +type planIssueRowJSON struct { + issueRow + BlockedBy []string `json:"blocked_by"` +} + +// planIssueJSON is the explicit JSON wire format for a `--with-body` plan +// row, avoiding the fragile marshal-unmarshal-remarshal pattern (mirrors +// showResultJSON). It is the shape `plan` emitted before DKT-1053, kept +// byte-identical for the caller that wants every description in one call. type planIssueJSON struct { ID string `json:"id"` ParentID *string `json:"parent_id,omitempty"` @@ -50,10 +68,23 @@ type planIssueJSON struct { BlockedBy []string `json:"blocked_by"` } -// MarshalJSON implements custom JSON serialization for planIssue. +// MarshalJSON implements custom JSON serialization for planIssue: the summary +// row by default, the full row under --with-body. func (p planIssue) MarshalJSON() ([]byte, error) { i := p.Issue + blockedBy := p.BlockedBy + if blockedBy == nil { + blockedBy = []string{} + } + + if !p.withBody { + return json.Marshal(planIssueRowJSON{ + issueRow: summarizeIssue(i), + BlockedBy: blockedBy, + }) + } + labels := i.Labels if labels == nil { labels = []string{} @@ -66,10 +97,6 @@ func (p planIssue) MarshalJSON() ([]byte, error) { if docs == nil { docs = []model.DocRef{} } - blockedBy := p.BlockedBy - if blockedBy == nil { - blockedBy = []string{} - } j := planIssueJSON{ ID: model.FormatID(i.ID), @@ -121,6 +148,7 @@ func runPlan(cmd *cobra.Command, args []string, w *output.Writer) error { types, _ := cmd.Flags().GetStringSlice("type") assignee, _ := cmd.Flags().GetString("assignee") rootFlag, _ := cmd.Flags().GetString("root") + withBody, _ := cmd.Flags().GetBool("with-body") // Validate status filter values. for _, s := range statuses { @@ -198,6 +226,7 @@ func runPlan(cmd *cobra.Command, args []string, w *output.Writer) error { planIssues[j] = planIssue{ Issue: issue, BlockedBy: collectDeps(issue.ID, dag), + withBody: withBody, } } phases[i] = planPhaseJSON{ @@ -379,5 +408,6 @@ func init() { planCmd.Flags().StringSliceP("priority", "p", nil, "Filter by priority (repeatable)") planCmd.Flags().StringSliceP("type", "T", nil, "Filter by type (repeatable)") planCmd.Flags().StringP("assignee", "a", "", "Filter by assignee") + planCmd.Flags().Bool("with-body", false, withBodyHelp) rootCmd.AddCommand(planCmd) } From d54c1f7a921733cab7f4917fb8c0b65da0834dd8 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:05:29 -0700 Subject: [PATCH 008/397] fix(engine): read a re-entry round's prior inputs from its own lineage - the previous-round pass matched on emitted kind alone, so a re-synthesis received the prior round's whole raw review fanout beside the digest of it - a re-entry step now reads only its own name at the prior ordinal plus the standing set the loop body declares it acted on - one security-change round-2 synthesis carried a 263KB packet of four raw judge payloads it was also carrying as the reconciled aggregate --- internal/engine/context.go | 98 ++++++++++++++++++++----- internal/engine/dkt491_test.go | 13 ++-- internal/engine/loop_test.go | 126 +++++++++++++++++++++++++++++++++ 3 files changed, 214 insertions(+), 23 deletions(-) diff --git a/internal/engine/context.go b/internal/engine/context.go index 6d235f56..3e617032 100644 --- a/internal/engine/context.go +++ b/internal/engine/context.go @@ -499,26 +499,41 @@ func contextPins(tx *sql.Tx, runID int) ([]ContextPin, error) { } // previousRoundInputs binds the PREVIOUS round's artifacts of a loop -// re-entry step's own emitted kind (DKT-63): review@k reads review@(k-1)'s -// fanout findings and the reconciled set beside the fix's change-summary, -// so a re-review judge reads the critique it must answer rather than the -// loop body's summary of it. RUN-5's re-review judges had only the fix -// step's own account — the author's summary of the critique of the author's -// work — and could not tell an answered ask from an ask answered as the fix -// characterised it, nor compute finding volume across rounds from one side -// of the exchange. +// re-entry step's own emitted kind, from the step's OWN LINEAGE (DKT-63, +// narrowed by DKT-1055): review@k reads review@(k-1)'s fanout findings and +// the reconciled set beside the fix's change-summary, so a re-review judge +// reads the critique it must answer rather than the loop body's summary of +// it. RUN-5's re-review judges had only the fix step's own account — the +// author's summary of the critique of the author's work — and could not tell +// an answered ask from an ask answered as the fix characterised it, nor +// compute finding volume across rounds from one side of the exchange. // // The rule is structural, read from the pinned definition // (loopReentryEmitter): the step is at ordinal > 0, inside the loop closure // (the same merged set blockingLoopBodyAbsent consults), and emits a kind — -// then every DONE same-issue producer at ordinal k-1 contributes its newest -// artifact of that kind (latestPerProducer, f6b28cc's binding rule), in -// §6.7's within-input order (sortArtifacts) — the same collapse and the same -// order every declared input resolves under. "Its own emitted kind" is what -// keeps this generic: the critique a step answers is definitionally the kind -// it produces, and core never reads what the kind means. nil when the step -// is not a re-entry — ordinal 0, an unlooped definition, or a kindless step -// — so every other packet is byte-identical. +// then every DONE same-issue producer at ordinal k-1 IN THE STEP'S LINEAGE +// (previousRoundLineage) contributes its newest artifact of that kind +// (latestPerProducer, f6b28cc's binding rule), in §6.7's within-input order +// (sortArtifacts) — the same collapse and the same order every declared +// input resolves under. "Its own emitted kind" is what keeps this generic: +// the critique a step answers is definitionally the kind it produces, and +// core never reads what the kind means. nil when the step is not a re-entry +// — ordinal 0, an unlooped definition, or a kindless step — so every other +// packet is byte-identical. +// +// THE LINEAGE IS NOT EVERY PRODUCER OF THE KIND. DKT-63's first cut matched +// on kind alone, and in every change workflow `review`, `synthesize`, and +// `reconcile` all emit `findings` — so synthesize@k received review@(k-1)'s +// whole raw fanout beside synthesize@(k-1) and reconcile@(k-1), on top of +// its declared `review.*` (already ordinal-scoped to round k). A +// security-change round-2 synthesis carried a 263KB packet — four raw judge +// payloads from the round before, whose digest it was ALSO carrying as +// reconcile@(k-1)'s aggregate — and spent 2.13M cache-read tokens on it. The +// previous round a step answers is the prior round of ITS OWN lineage: the +// same step name (review's own lineage IS the fanout, which is how DKT-63's +// re-review behaviour survives without a special case) plus the standing set +// the round's fix acted on. Nothing about this is scopeable from the +// definition's `inputs`, because the pass is engine-side. // // `artifacts` is the caller's already-loaded snapshot (AssembleContext loads // it once for this and the declared-input pass together). @@ -528,6 +543,7 @@ func previousRoundInputs( if !loopReentryEmitter(sched, step, spec) { return nil } + lineage := previousRoundLineage(sched.defs[step.WorkflowID], step.StepName, spec.Emits) var matched []*db.Artifact for _, a := range artifacts { @@ -536,7 +552,7 @@ func previousRoundInputs( p.Ordinal != step.Ordinal-1 || !recordedProducer(p.Status) { continue } - if a.Kind != spec.Emits { + if a.Kind != spec.Emits || !lineage[p.StepName] { continue } matched = append(matched, a) @@ -546,6 +562,54 @@ func previousRoundInputs( return artifactInputs(matched, sched.stepByID) } +// previousRoundLineage is the set of step NAMES whose previous-round +// artifacts of `kind` a re-entry step `name` reads (DKT-1055): the step's own +// name, plus the STANDING-SET producers — the steps each `loop = true` body +// whose `after_loop` chain contains `name` declares as `.` inputs +// of this kind, when the named step is itself inside that body's chain. +// +// The body's declared inputs are the definition's own statement of what the +// round was entered to act on: `fix` reads `reconcile.findings`, so the +// reconciled set is, by construction, the standing set that round k's +// re-review and re-synthesis annotate against — and the definition says so +// without core learning what a finding or a reconciliation is. The chain +// restriction keeps "previous round" honest: only a step that re-runs each +// round has a round to be previous, and a one-shot upstream producer the body +// also reads (`implement.change-summary`) is bound by §7.4's per-input +// fallback, not by this pass. Kind is compared per declaration, so a body's +// `.gate-results` or `.vote-record` never admits that step's +// artifacts of some other kind. +// +// Read per body rather than over the merged closure so a serves-scoped +// cluster (DKT-544) contributes only its own standing set: a body whose chain +// does not contain the consumer has nothing to say about what it answers. +func previousRoundLineage(def *workflow.Definition, name, kind string) map[string]bool { + lineage := map[string]bool{name: true} + if def == nil { + return lineage + } + for _, body := range def.Steps { + if !body.Loop || body.AfterLoop == "" { + continue + } + chain := downstreamClosure(def, []string{body.AfterLoop}) + if !chain[name] { + continue + } + for _, declared := range body.Inputs { + producer, declaredKind, ok := splitInput(declared) + if !ok || !chain[producer] { + continue + } + if declaredKind != "*" && declaredKind != kind { + continue + } + lineage[producer] = true + } + } + return lineage +} + // artifactInputs renders artifact rows into the §11.4 bundle's input shape — // the ONE reading of the artifact -> ContextInput mapping (the "ARTIFACT-%d" // name, the producer instance, body beside payload), shared by the diff --git a/internal/engine/dkt491_test.go b/internal/engine/dkt491_test.go index bb07dc0e..c8fea8c1 100644 --- a/internal/engine/dkt491_test.go +++ b/internal/engine/dkt491_test.go @@ -194,14 +194,14 @@ func TestReReviewPacketInlinesTheRevisedDocOnce(t *testing.T) { } // And every input the step legitimately carries, still there and still - // once: the doc, the issue body, and the previous round's four findings - // sets (two judges, the synthesis, the reconciliation). + // once: the doc, the issue body, and the previous round's three findings + // sets from review's own lineage (two judges, the reconciliation the body + // acted on — the synthesis between them is neither, DKT-1055). want := []string{ "== INPUT doc from revise-a@1", "== INPUT issue.body", "== INPUT findings from review@0#0", "== INPUT findings from review@0#1", - "== INPUT findings from synthesize@0", "== INPUT findings from reconcile@0", } headers := inputHeaders(packet.Packet) @@ -246,9 +246,10 @@ func TestReReviewBundleBindsTheRevisedDocOnce(t *testing.T) { if len(docs) != 1 { t.Errorf("the bundle binds %d doc inputs, want 1: %v", len(docs), docs) } - if len(bundle.Inputs) != 6 { - t.Errorf("the bundle carries %d inputs, want 6 (the doc, the issue "+ - "body, and the previous round's four findings sets)", + if len(bundle.Inputs) != 5 { + t.Errorf("the bundle carries %d inputs, want 5 (the doc, the issue "+ + "body, and the previous round's three findings sets from review's "+ + "own lineage — two judges and the reconciliation, DKT-1055)", len(bundle.Inputs)) } } diff --git a/internal/engine/loop_test.go b/internal/engine/loop_test.go index 933e5b63..f97b44bf 100644 --- a/internal/engine/loop_test.go +++ b/internal/engine/loop_test.go @@ -3,6 +3,9 @@ package engine import ( "database/sql" "fmt" + "os" + "slices" + "sort" "strings" "testing" @@ -1033,6 +1036,13 @@ func TestReReviewRoundCarriesThePreviousRoundsFindings(t *testing.T) { t.Errorf("review@1#0 does not carry reconcile@0's reconciled set: %v", producerList(bundle.Inputs)) } + // And nothing of the kind from OUTSIDE review's lineage (DKT-1055): the + // synthesis between the fanout and the reconciled set is neither review's + // own prior round nor the standing set the fix acted on. + if _, ok := byProducer["synthesize@0"]; ok { + t.Errorf("review@1#0 carries synthesize@0's findings — the previous-round "+ + "pass matched on kind rather than lineage: %v", producerList(bundle.Inputs)) + } // The declared inputs are untouched: the fix's own account still binds. if _, ok := byProducer["fix@1"]; !ok { t.Errorf("review@1#0 lost its declared change-summary from fix@1: %v", @@ -1050,6 +1060,122 @@ func TestReReviewRoundCarriesThePreviousRoundsFindings(t *testing.T) { } } +// --------------------------------------------------------------------------- +// DKT-1055: a re-entry round reads its own lineage, not every producer of +// its kind +// --------------------------------------------------------------------------- + +// TestReSynthesisRoundCarriesItsOwnLineageOnly pins DKT-1055: the previous- +// round pass binds the prior round of the step's OWN lineage — its own name +// at ordinal k-1 plus the standing set the loop body acted on — never every +// same-issue producer of its emitted kind. `review`, `synthesize`, and +// `reconcile` all emit `findings`, so the kind-only match handed synthesize@1 +// review@0's whole raw fanout beside synthesize@0 and reconcile@0, on top of +// its declared `review.*` (already scoped to round 1): a security-change +// round-2 synthesis carried a 263KB packet holding four raw judge payloads +// whose digest it was also carrying as reconcile's aggregate. +func TestReSynthesisRoundCarriesItsOwnLineageOnly(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + driveToVerify(t, conn, e, 0) + claimAndComplete(t, conn, e, "verify@0", "the ac report", unmetPayload) + claimAndComplete(t, conn, e, "fix@1", "the fix summary", "") + completeReviewFanout(t, conn, e, 1) + + bundle, err := ReadContext(conn, stepIDByInstance(t, conn, "synthesize@1"), nowMS) + testsupport.Must(t, err, "ReadContext(synthesize@1): %v", err) + + byProducer := map[string]ContextInput{} + for _, in := range bundle.Inputs { + byProducer[in.ProducerStep] = in + } + + // The declared `review.*` still binds the CURRENT round's fanout. + for i := range 4 { + judge := fmt.Sprintf("review@1#%d", i) + if _, ok := byProducer[judge]; !ok { + t.Errorf("synthesize@1 lost its declared %s input: %v", + judge, producerList(bundle.Inputs)) + } + } + // Its own lineage from the round before: the prior synthesis, and the + // reconciled set it was digested into — what fix@1 read and acted on. + for _, prior := range []string{"synthesize@0", "reconcile@0"} { + in, ok := byProducer[prior] + if !ok { + t.Errorf("synthesize@1 does not carry %s's findings — the previous "+ + "round of its own lineage: %v", prior, producerList(bundle.Inputs)) + continue + } + if in.Kind != "findings" { + t.Errorf("%s's prior-round input has kind %q, want findings", prior, in.Kind) + } + } + // And NOT the previous round's raw judge fanout: reconcile@0's aggregate + // is the digest of it, and a match on kind alone is what carried it. + for i := range 4 { + judge := fmt.Sprintf("review@0#%d", i) + if _, ok := byProducer[judge]; ok { + t.Errorf("synthesize@1 carries %s's raw findings from the round before "+ + "— the previous-round pass matched on kind, not lineage: %v", + judge, producerList(bundle.Inputs)) + } + } + if len(bundle.Inputs) != 6 { + t.Errorf("synthesize@1 carries %d inputs, want 6 (review@1's four judges, "+ + "synthesize@0, reconcile@0): %v", len(bundle.Inputs), producerList(bundle.Inputs)) + } +} + +// TestPreviousRoundLineageIsOwnNamePlusTheStandingSet pins the definitional +// half of DKT-1055 against the fixtures directly: a re-entry step's lineage +// is its own name plus the in-chain producers the serving loop body reads of +// the step's kind — and a serves-scoped body whose chain does not contain the +// step contributes nothing to it (DKT-544). +func TestPreviousRoundLineageIsOwnNamePlusTheStandingSet(t *testing.T) { + src, err := os.ReadFile(fixturePath) + testsupport.Must(t, err, "reading fixture: %v", err) + standard, err := workflow.Parse(src) + testsupport.Must(t, err, "parsing fixture: %v", err) + clusters, err := workflow.Parse([]byte(clusterSrc)) + testsupport.Must(t, err, "parsing the cluster fixture: %v", err) + + for _, tc := range []struct { + def *workflow.Definition + step string + kind string + want []string + }{ + // Own name plus the reconciled set `fix` reads: never the raw fanout + // for `synthesize`, never the synthesis for `review`, and never + // `implement`, which is outside the chain and of another kind. + {standard, "review", "findings", []string{"reconcile", "review"}}, + {standard, "synthesize", "findings", []string{"reconcile", "synthesize"}}, + {standard, "verify", "ac-report", []string{"verify"}}, + {standard, "commit", "commit-record", []string{"commit"}}, + // Cluster A's gate: `prd-fix`'s `draft.doc` is outside its chain, and + // cluster B's body never contains the gate at all. + {clusters, "prd-gate", "findings", []string{"prd-gate"}}, + {clusters, "design-gate", "report", []string{"design-gate"}}, + // A step outside every chain has only itself; loopReentryEmitter is + // what keeps the pass from firing there in the first place. + {standard, "implement", "change-summary", []string{"implement"}}, + } { + got := previousRoundLineage(tc.def, tc.step, tc.kind) + names := make([]string, 0, len(got)) + for name := range got { + names = append(names, name) + } + sort.Strings(names) + if !slices.Equal(names, tc.want) { + t.Errorf("previousRoundLineage(%s, %s, %s) = %v, want %v", + tc.def.Pipeline.Name, tc.step, tc.kind, names, tc.want) + } + } +} + // producerList names each input's producer, for failure messages. func producerList(inputs []ContextInput) []string { out := make([]string, 0, len(inputs)) From 839206906a7425a522a9d4daa5ff7539d7703415 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:14:12 -0700 Subject: [PATCH 009/397] feat(engine): report a step's resolved target ref from step show - a vote panel is seated from step show and only probes the context bundle when the row names a target - the bundle carried target_sha and target_worktree but the step view did not, so every panel read its own HEAD instead of the judged tree - the view now resolves the same pair in the same transaction, emitting both keys only when the step consumes issue.diff and a round record exists - steps with no target emit neither key rather than an empty or approximated value --- internal/cli/step.go | 156 +++++++++++++-------- internal/cli/step_show_target_test.go | 188 ++++++++++++++++++++++++++ internal/engine/claim.go | 30 +++- internal/engine/dkt1056_test.go | 110 +++++++++++++++ internal/engine/pregate.go | 29 ++++ 5 files changed, 458 insertions(+), 55 deletions(-) create mode 100644 internal/cli/step_show_target_test.go create mode 100644 internal/engine/dkt1056_test.go diff --git a/internal/cli/step.go b/internal/cli/step.go index 41595471..8efbb0dd 100644 --- a/internal/cli/step.go +++ b/internal/cli/step.go @@ -926,27 +926,61 @@ record, under its holder's token.`, }, } -// heldClusterPayload is `step show`'s shape for a materialized held step: the -// row it always emitted, plus the cluster linkage. +// stepDetailPayload is `step show`'s shape for a step that has something to +// say beyond its row: a materialized held step's cluster linkage (DKT-239), +// and the resolved target ref of a step that consumes `issue.diff` (DKT-1056). // -// The embedded StepRow marshals inline, and `held_cluster` is `omitempty`, so -// EVERY STEP THAT IS NOT A HELD ROW EMITS EXACTLY THE BYTES IT ALWAYS DID — +// The embedded StepRow marshals inline and every added field is `omitempty`, +// so EVERY STEP WITH NOTHING TO ADD EMITS EXACTLY THE BYTES IT ALWAYS DID — // the verb's own promise that "a single id emits the same row object it always -// has, unchanged" holds for all of them. The only steps whose shape moves are -// the ones DKT-239 is about, where the previous shape named neither the -// cluster nor the artifact carrying it. -type heldClusterPayload struct { +// has, unchanged" holds for all of them. +// +// `target_sha` and `target_worktree` are spelled exactly as the context bundle +// spells them (engine.Context), because they ARE the bundle's fields: a reader +// that seats a panel from this row and then re-reads the bundle must not have +// to translate between two names for one fact. +type stepDetailPayload struct { model.StepRow HeldCluster *engine.HeldClusterLink `json:"held_cluster,omitempty"` + // A step whose inputs resolve no `issue.diff` round record emits NEITHER + // key. The wave's pre-check reads their presence as "there is a judged + // tree, spend a probe on the bundle"; emitting an empty string, or a + // stand-in sha, would send it to read a tree nobody recorded. + TargetSHA string `json:"target_sha,omitempty"` + TargetWorktree string `json:"target_worktree,omitempty"` } // stepShowPayload wraps a view only when there is something to add, so the // unchanged case does not even pay for a wrapper type on the wire. func stepShowPayload(view *engine.StepView) any { - if view.HeldCluster == nil { + if view.HeldCluster == nil && view.TargetSHA == "" && view.TargetWorktree == "" { return view.Row } - return heldClusterPayload{StepRow: view.Row, HeldCluster: view.HeldCluster} + return stepDetailPayload{ + StepRow: view.Row, + HeldCluster: view.HeldCluster, + TargetSHA: view.TargetSHA, + TargetWorktree: view.TargetWorktree, + } +} + +// renderStepTarget is the human half: which tree this step's work is about. +// +// Printed only when there is one, for the same reason the JSON keys are +// omitted — a line reading "Target: (none)" on every step that consumes no +// diff is noise on the majority of reads. +func renderStepTarget(view *engine.StepView) string { + if view.TargetSHA == "" && view.TargetWorktree == "" { + return "" + } + out := "\nTarget (the tree under review):\n" + if view.TargetSHA != "" { + out += fmt.Sprintf(" sha: %s\n", view.TargetSHA) + } + if view.TargetWorktree != "" { + out += fmt.Sprintf(" worktree: %s\n", view.TargetWorktree) + } + return out } // renderHeldCluster is the human half: where the question came from. @@ -986,60 +1020,76 @@ including no reap. A single id emits the same row object it always has, unchanged. Two or more ids emit a JSON array of that same shape under data — batch reads are a conductor's most common shape, and a single-id restriction taxes every loop -iteration.`, +iteration. + +A step whose declared inputs resolve an ` + "`issue.diff`" + ` round record also carries +target_sha and target_worktree — the SAME pair its context bundle carries, so +a caller seating a review or a vote panel learns which tree is under review +without reading the whole bundle. Both keys are ABSENT on a step with no such +record; neither is ever approximated from the shared HEAD.`, Args: cobra.MinimumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - w := getWriter(cmd) - conn := getDB(cmd) + return runStepShow(cmd, args, getWriter(cmd)) + }, +} - if len(args) == 1 { - id, err := stepArg(args[0]) - if err != nil { - return err - } - view, err := engine.LoadStepView(conn, id, model.NowMS()) - if err != nil { - return stepErr(err, stepLabel(id)) - } +// runStepShow is `step show`'s body, with the writer injected so tests read +// the envelope rather than process stdout. +func runStepShow(cmd *cobra.Command, args []string, w *output.Writer) error { + conn := getDB(cmd) - var message string - if !w.JSONMode { - message = render.RenderStepDetail( - view.Row, view.Routing, view.SagaStage, view.Owner, view.ExpiresMS) + - render.RenderStepGateSummary(view.Row.Step, gateRows(view.Gates)) + - renderHeldCluster(view.HeldCluster) - } - w.Success(stepShowPayload(view), message) - return nil + if len(args) == 1 { + id, err := stepArg(args[0]) + if err != nil { + return err } - - rows := make([]any, 0, len(args)) - var messages []string - for _, arg := range args { - id, err := stepArg(arg) - if err != nil { - return err - } - view, err := engine.LoadStepView(conn, id, model.NowMS()) - if err != nil { - return stepErr(err, stepLabel(id)) - } - rows = append(rows, stepShowPayload(view)) - if !w.JSONMode { - messages = append(messages, render.RenderStepDetail( - view.Row, view.Routing, view.SagaStage, view.Owner, view.ExpiresMS)+ - render.RenderStepGateSummary(view.Row.Step, gateRows(view.Gates))+ - renderHeldCluster(view.HeldCluster)) - } + view, err := engine.LoadStepView(conn, id, model.NowMS()) + if err != nil { + return stepErr(err, stepLabel(id)) } var message string if !w.JSONMode { - message = strings.Join(messages, "\n\n") + message = stepShowMessage(view) } - w.Success(rows, message) + w.Success(stepShowPayload(view), message) return nil - }, + } + + rows := make([]any, 0, len(args)) + var messages []string + for _, arg := range args { + id, err := stepArg(arg) + if err != nil { + return err + } + view, err := engine.LoadStepView(conn, id, model.NowMS()) + if err != nil { + return stepErr(err, stepLabel(id)) + } + rows = append(rows, stepShowPayload(view)) + if !w.JSONMode { + messages = append(messages, stepShowMessage(view)) + } + } + + var message string + if !w.JSONMode { + message = strings.Join(messages, "\n\n") + } + w.Success(rows, message) + return nil +} + +// stepShowMessage renders one step's human block, identically for the single +// and the batch form — they rendered the same four pieces in two places, and +// the target section is the fifth. +func stepShowMessage(view *engine.StepView) string { + return render.RenderStepDetail( + view.Row, view.Routing, view.SagaStage, view.Owner, view.ExpiresMS) + + render.RenderStepGateSummary(view.Row.Step, gateRows(view.Gates)) + + renderHeldCluster(view.HeldCluster) + + renderStepTarget(view) } // stepContextResult is `step context`'s payload. `--meta` rides as a SIBLING diff --git a/internal/cli/step_show_target_test.go b/internal/cli/step_show_target_test.go new file mode 100644 index 00000000..1ceec77c --- /dev/null +++ b/internal/cli/step_show_target_test.go @@ -0,0 +1,188 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-1056 at the CLI boundary. internal/engine/dkt1056_test.go proves the +// RESOLUTION — that the view's pair is the bundle's pair, and empty when there +// is no round record; what this file asserts is the WIRE SHAPE of `step show +// --json`, which is what a wave's pre-check actually reads: the two keys +// spelled exactly as the bundle spells them, beside the row, and ABSENT on a +// step with no target rather than present-and-empty. + +// showTargetWorkflow is the shape a vote wave sees: a tree-holding implement +// and a judge that reads its issue.diff. +const showTargetWorkflow = ` +[pipeline] +name = "cli-show-target" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "implement" +executor = "implement" +emits = "change-summary" + +[[step]] +name = "review" +after = ["implement"] +holds_tree = false +executor = "judge" +emits = "findings" +inputs = ["issue.diff"] +` + +// showTargetRun activates showTargetWorkflow over one issue and completes +// `implement` with the given recorded head and worktree — head "" and worktree +// "" is the no-round-record case. It returns the two step ids. +func showTargetRun(t *testing.T, conn *sql.DB, head, worktree string) (implementID, reviewID int) { + t.Helper() + now := model.NowMS() + + registerForRun(t, conn, showTargetWorkflow) + issueID, err := db.CreateIssue(conn, &model.Issue{ + Title: "seat the panel on the judged tree", Description: "a body", + Status: model.StatusBacklog, Priority: model.PriorityNone, + Kind: model.IssueKindTask, + }, nil, nil) + testsupport.Must(t, err, "creating issue: %v", err) + + run, err := db.InsertRunWithContext(conn, 1, "", 0, now, db.RunContext{ExecRoot: t.TempDir()}) + testsupport.Must(t, err, "starting run: %v", err) + testsupport.Must(t, db.AddRunIssue(conn, run.ID, issueID), "adding issue to run: %v", nil) + _, err = engine.Activate(conn, run.ID, engine.ActivateOptions{NowMS: now}) + testsupport.Must(t, err, "activate: %v", err) + + for instance, id := range map[string]*int{"implement@0": &implementID, "review@0": &reviewID} { + err := conn.QueryRow(`SELECT id FROM steps WHERE instance = ?`, instance).Scan(id) + testsupport.Must(t, err, "finding %s: %v", instance, err) + } + + // Stubbed rather than shelled out: the subject here is the emitted shape, + // and a real checkout would make the fixture a git test. + e := engine.NewEngine() + e.HeadFn = func(string) string { return head } + e.DiffFn = func(_, _ string, _ []string) (string, error) { return "the diff", nil } + + claim, err := engine.ClaimStep(conn, implementID, engine.ClaimOptions{Owner: "w", NowMS: now}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, implementID, engine.CompleteOptions{ + Token: claim.Token, Artifact: []byte("the change summary"), + WorkDir: worktree, NowMS: now, + }) + testsupport.Must(t, err, "complete implement: %v", err) + return implementID, reviewID +} + +// showTargetEnvelope runs `step show --json` on one step and decodes the two +// fields under test, plus the step id so the row itself is proven still flat. +func showTargetEnvelope(t *testing.T, conn *sql.DB, stepID int) (step, sha, worktree, raw string) { + t.Helper() + w, buf := bufWriter(true) + err := runStepShow(cmdWithDB(conn), []string{model.FormatStepID(stepID)}, w) + testsupport.Must(t, err, "step show: %v\n%s", err, buf.String()) + + var envelope struct { + Data struct { + Step string `json:"step"` + TargetSHA string `json:"target_sha"` + TargetWorktree string `json:"target_worktree"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + return envelope.Data.Step, envelope.Data.TargetSHA, envelope.Data.TargetWorktree, buf.String() +} + +// TestStepShowJSONCarriesTheResolvedTarget: the wave's pre-check reads the +// judged tree off the gate's own row, without a bundle probe. +func TestStepShowJSONCarriesTheResolvedTarget(t *testing.T) { + conn := newTestDB(t) + _, reviewID := showTargetRun(t, conn, "cafe1234cafe1234", "/worktrees/issue-under-test") + + step, sha, worktree, raw := showTargetEnvelope(t, conn, reviewID) + if step != model.FormatStepID(reviewID) { + t.Errorf("data.step = %q; the row must still marshal flat:\n%s", step, raw) + } + if sha != "cafe1234cafe1234" { + t.Errorf("data.target_sha = %q, want the recorded head:\n%s", sha, raw) + } + if worktree != "/worktrees/issue-under-test" { + t.Errorf("data.target_worktree = %q, want the declared worktree:\n%s", worktree, raw) + } + + // The same pair the bundle carries, from the surface that costs no probe. + bundle, err := engine.ReadContext(conn, reviewID, model.NowMS()) + testsupport.Must(t, err, "ReadContext: %v", err) + if bundle.TargetSHA != sha || bundle.TargetWorktree != worktree { + t.Errorf("step show says (%q, %q), the bundle says (%q, %q)", + sha, worktree, bundle.TargetSHA, bundle.TargetWorktree) + } +} + +// TestStepShowJSONOmitsTheTargetWithoutARoundRecord is the fabrication guard: +// a judge whose producer recorded no head and no worktree emits NEITHER key. +// Present-and-empty would read to the wave as "there is a target", and a +// stand-in sha would seat the panel on a tree nobody judged. +func TestStepShowJSONOmitsTheTargetWithoutARoundRecord(t *testing.T) { + conn := newTestDB(t) + _, reviewID := showTargetRun(t, conn, "", "") + + _, sha, worktree, raw := showTargetEnvelope(t, conn, reviewID) + if sha != "" || worktree != "" { + t.Errorf("step show invented a target (%q, %q):\n%s", sha, worktree, raw) + } + if strings.Contains(raw, "target_sha") || strings.Contains(raw, "target_worktree") { + t.Errorf("the envelope carries a target key with no target to name:\n%s", raw) + } +} + +// TestStepShowJSONOmitsTheTargetForAStepThatConsumesNoDiff: `implement` +// declares no inputs, so its row is byte-for-byte what it always was. +func TestStepShowJSONOmitsTheTargetForAStepThatConsumesNoDiff(t *testing.T) { + conn := newTestDB(t) + implementID, _ := showTargetRun(t, conn, "cafe1234cafe1234", "/worktrees/issue-under-test") + + _, sha, worktree, raw := showTargetEnvelope(t, conn, implementID) + if sha != "" || worktree != "" { + t.Errorf("implement@0 reports a target (%q, %q); it consumes no "+ + "issue.diff:\n%s", sha, worktree, raw) + } + if strings.Contains(raw, "target_") { + t.Errorf("a step consuming no issue.diff grew a target key:\n%s", raw) + } +} + +// TestStepShowHumanModeStatesTheTarget: the same fact on the operator's +// channel, and nothing at all where there is no target. +func TestStepShowHumanModeStatesTheTarget(t *testing.T) { + conn := newTestDB(t) + implementID, reviewID := showTargetRun(t, conn, "cafe1234cafe1234", "/worktrees/issue-under-test") + + w, buf := bufWriter(false) + err := runStepShow(cmdWithDB(conn), []string{model.FormatStepID(reviewID)}, w) + testsupport.Must(t, err, "step show: %v", err) + if !strings.Contains(buf.String(), "cafe1234cafe1234") || + !strings.Contains(buf.String(), "/worktrees/issue-under-test") { + t.Errorf("the human block does not state the target:\n%s", buf.String()) + } + + w, buf = bufWriter(false) + err = runStepShow(cmdWithDB(conn), []string{model.FormatStepID(implementID)}, w) + testsupport.Must(t, err, "step show: %v", err) + if strings.Contains(buf.String(), "Target") { + t.Errorf("a step with no target printed a target section:\n%s", buf.String()) + } +} diff --git a/internal/engine/claim.go b/internal/engine/claim.go index 20bdd96e..57fe4cb3 100644 --- a/internal/engine/claim.go +++ b/internal/engine/claim.go @@ -785,6 +785,21 @@ type StepView struct { // give — that verb reports what the STEP produced, and a hold produces // nothing; the payload sits on the routing step's artifact. HeldCluster *HeldClusterLink + // TargetSHA and TargetWorktree are the SAME resolved target ref the step's + // context bundle carries (Context.TargetSHA / TargetWorktree, DKT-24), + // answered without assembling the bundle (DKT-1056). + // + // A vote panel is seated from this read: the wave asks `step show` for the + // gate's row first and probes the (up to 1MiB) bundle only when the row + // names a target, so a target the bundle knows and this view does not is a + // panel reading its own HEAD instead of the judged tree. + // + // BOTH ARE EMPTY, NEVER APPROXIMATE, when the step's inputs resolve no + // `issue.diff` round record. There is no fallback to the shared HEAD here + // and there must not be one — the caller can tell "no target" from "this + // target" only if absence is reported as absence. + TargetSHA string + TargetWorktree string } // LoadStepView reads one step at its effective status. IT WRITES NOTHING — not @@ -846,11 +861,22 @@ func LoadStepView(conn *sql.DB, stepID int, nowMS int64) (*StepView, error) { return nil, err } + // Resolved in the SAME transaction, for the same one-connection reason — + // and from the same snapshot the two reads above used, so the target this + // view reports is the target a bundle assembled at this instant would + // carry (DKT-1056). + targetSHA, targetWorktree, err := stepTargetRef(tx, sched, fresh) + if err != nil { + return nil, err + } + view := &StepView{ Step: fresh, Row: row, Routing: fresh.Routing, SagaStage: fresh.SagaStage, - Gates: gates, - HeldCluster: heldCluster, + Gates: gates, + HeldCluster: heldCluster, + TargetSHA: targetSHA, + TargetWorktree: targetWorktree, } if lease := fresh.Lease(); lease.Live(nowMS) { view.Owner, view.ExpiresMS = lease.Owner, lease.ExpiresMS diff --git a/internal/engine/dkt1056_test.go b/internal/engine/dkt1056_test.go new file mode 100644 index 00000000..dd773caf --- /dev/null +++ b/internal/engine/dkt1056_test.go @@ -0,0 +1,110 @@ +package engine + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-1056 — the step VIEW answers the target question the context bundle +// already answers. +// +// A wave seats a vote panel from `step show`, and probes the (up to 1MiB) +// bundle only when that read names a target. The bundle carried +// `target_sha`/`target_worktree` from DKT-24 onward and the view carried +// neither, so every panel read its own HEAD instead of the judged tree and the +// wave logged "NO target ref on the bundle". These tests pin the view to the +// bundle: same pair, same resolution, no invention. + +// TestStepViewCarriesTheResolvedTarget: a step consuming `issue.diff` reports +// the round record's head and worktree, and reports EXACTLY what the bundle +// reports — the two are resolved by one function and must not drift. +func TestStepViewCarriesTheResolvedTarget(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + e.HeadFn = func(string) string { return "cafe1234cafe1234" } + + implementID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implementID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, implementID, CompleteOptions{ + Token: claim.Token, + Artifact: []byte("the change summary"), + WorkDir: "/worktrees/issue-under-test", + NowMS: nowMS, + }) + testsupport.Must(t, err, "complete implement: %v", err) + + reviewID := stepIDByInstance(t, conn, "review@0#0") + view, err := LoadStepView(conn, reviewID, nowMS) + testsupport.Must(t, err, "LoadStepView(review@0#0): %v", err) + + if view.TargetSHA != "cafe1234cafe1234" { + t.Errorf("view target_sha = %q, want the recorded head", view.TargetSHA) + } + if view.TargetWorktree != "/worktrees/issue-under-test" { + t.Errorf("view target_worktree = %q, want the declared worktree", + view.TargetWorktree) + } + + bundle, err := ReadContext(conn, reviewID, nowMS) + testsupport.Must(t, err, "ReadContext(review@0#0): %v", err) + if view.TargetSHA != bundle.TargetSHA || view.TargetWorktree != bundle.TargetWorktree { + t.Errorf("view target (%q, %q) != bundle target (%q, %q); the cheap read "+ + "exists to save the bundle probe, so a disagreement makes it useless", + view.TargetSHA, view.TargetWorktree, + bundle.TargetSHA, bundle.TargetWorktree) + } +} + +// TestStepViewReportsNoTargetForAStepThatConsumesNoDiff: `implement` declares +// no inputs at all, so there is nothing under review yet. Both halves stay +// empty — the shared checkout's HEAD is NOT a stand-in. +func TestStepViewReportsNoTargetForAStepThatConsumesNoDiff(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + + view, err := LoadStepView(conn, stepIDByInstance(t, conn, "implement@0"), nowMS) + testsupport.Must(t, err, "LoadStepView(implement@0): %v", err) + if view.TargetSHA != "" || view.TargetWorktree != "" { + t.Errorf("implement@0 reports target (%q, %q); a step consuming no "+ + "issue.diff has no tree under review to report", + view.TargetSHA, view.TargetWorktree) + } +} + +// TestStepViewFabricatesNoTargetWithoutARoundRecord is the failure mode the +// issue names: a diff recorded with no resolvable head and no declared +// worktree must yield the EMPTY pair. One fabricated sha in 22 recorded probes +// is why the wave stopped trusting this read at all. +func TestStepViewFabricatesNoTargetWithoutARoundRecord(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + e.HeadFn = func(string) string { return "" } + + implementID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implementID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, implementID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("summary"), NowMS: nowMS, + }) + testsupport.Must(t, err, "complete implement: %v", err) + + reviewID := stepIDByInstance(t, conn, "review@0#0") + view, err := LoadStepView(conn, reviewID, nowMS) + testsupport.Must(t, err, "LoadStepView(review@0#0): %v", err) + if view.TargetSHA != "" || view.TargetWorktree != "" { + t.Errorf("view target = (%q, %q), want both empty: the producer recorded "+ + "no round record and a plausible-looking value here seats a panel "+ + "on a tree nobody judged", view.TargetSHA, view.TargetWorktree) + } + + bundle, err := ReadContext(conn, reviewID, nowMS) + testsupport.Must(t, err, "ReadContext(review@0#0): %v", err) + if bundle.TargetSHA != "" || bundle.TargetWorktree != "" { + t.Fatalf("premise: the bundle itself reports a target (%q, %q)", + bundle.TargetSHA, bundle.TargetWorktree) + } +} diff --git a/internal/engine/pregate.go b/internal/engine/pregate.go index 5d9e9808..12d659cc 100644 --- a/internal/engine/pregate.go +++ b/internal/engine/pregate.go @@ -114,6 +114,35 @@ func resolvedTargetFor( return sha, worktree, nil } +// stepTargetRef is resolvedTargetFor for a reader holding ONLY the step — +// `step show` (DKT-1056). It loads the artifact snapshot itself, because this +// caller answers about one step rather than iterating a manifest. +// +// THE GUARD IS THE POINT. A step whose definition does not declare `issue.diff` +// can carry no round record, so it pays for no input resolution at all and +// reports nothing: `step show` is a conductor's most-repeated read, and making +// every one of them resolve a whole input set to learn "no target" would tax +// the common case for the rare one. The same predicate the dispatch verbs' +// stale-target collector uses (staleTargetCandidates), for the same reason. +// +// It returns the empty pair — never a fabricated one — for a step with no +// resolvable round record. A vote panel seats itself on what this says; a +// plausible-looking sha invented here would seat judges on the wrong tree, +// which is worse than seating them on their own HEAD, because it is silent. +func stepTargetRef( + tx *sql.Tx, sched *Scheduler, step *db.Step, +) (sha, worktree string, err error) { + spec := materializedSpec(sched.defs[step.WorkflowID], step, sched.holdTally) + if spec == nil || !consumesIssueDiff(spec) { + return "", "", nil + } + artifacts, err := db.ListRunArtifactsTx(tx, step.RunID) + if err != nil { + return "", "", err + } + return resolvedTargetFor(tx, sched, step, spec, artifacts) +} + // runPreGates executes the step's pre-gates, one at a time, each result // committing in its own small transaction. // From 1277bd1a19117c1a9da6ff2180a2a2d5e443d370 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:53:15 -0700 Subject: [PATCH 010/397] fix(model): accept related_to as relates_to alias (cherry picked from commit f01b09faa828b3229e413832023f7ffd261c726f) --- internal/model/model_test.go | 2 ++ internal/model/relation.go | 8 +++++++- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/internal/model/model_test.go b/internal/model/model_test.go index 6895b600..773fdb9a 100644 --- a/internal/model/model_test.go +++ b/internal/model/model_test.go @@ -276,6 +276,8 @@ func TestParseRelationType(t *testing.T) { {"depends-on", RelationDependsOn, false}, {"relates_to", RelationRelatesTo, false}, {"relates-to", RelationRelatesTo, false}, + {"related_to", RelationRelatesTo, false}, + {"related-to", RelationRelatesTo, false}, {"duplicates", RelationDuplicates, false}, {"invalid", "", true}, } diff --git a/internal/model/relation.go b/internal/model/relation.go index e0e0b5bc..c62258b4 100644 --- a/internal/model/relation.go +++ b/internal/model/relation.go @@ -35,9 +35,15 @@ func ValidateRelationType(rt RelationType) error { } // ParseRelationType accepts both hyphenated ("depends-on") and underscored ("depends_on") -// forms and returns the canonical underscored RelationType. +// forms and returns the canonical underscored RelationType. "related_to" / +// "related-to" is accepted as an alias of "relates_to": the natural +// mis-typing of the canonical name, and the normalized form the error message +// already surfaces (DKT-1073). func ParseRelationType(input string) (RelationType, error) { normalized := RelationType(strings.ReplaceAll(strings.TrimSpace(input), "-", "_")) + if normalized == "related_to" { + normalized = RelationRelatesTo + } if err := ValidateRelationType(normalized); err != nil { return "", err } From 7ba06113c7853ffc3b7a40c0b2e7fa757da75a0a Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:42:19 -0700 Subject: [PATCH 011/397] test(workflow): tell an intentional label exclusion from a real routing gap - derive the reserved-label set (selected by no workflow, declined by every workflow) from the live corpus instead of hand-listing it, so the sweep tracks corpus changes automatically - fail loudly, naming the label, when one is declined by only some workflows and selected by none (an orphaned exclusion) instead of silently reporting it as an unrouted gap - apply the non-vacuity floor to the routable vocabulary only, after reserved labels are set aside, across all three sweep tests --- internal/workflow/routing_sweep_test.go | 165 +++++++++++++++++++++--- 1 file changed, 148 insertions(+), 17 deletions(-) diff --git a/internal/workflow/routing_sweep_test.go b/internal/workflow/routing_sweep_test.go index 81c470ff..6069d90e 100644 --- a/internal/workflow/routing_sweep_test.go +++ b/internal/workflow/routing_sweep_test.go @@ -4,6 +4,7 @@ import ( "fmt" "os" "path/filepath" + "slices" "sort" "strings" "testing" @@ -37,10 +38,22 @@ import ( // own `.docket/config/workflows` — because that union is what the engine // actually binds against (autoregister.go). Sweeping anything else would // certify a corpus no invocation uses. - -// sweepLabels is the label vocabulary the nine [match] clauses discriminate on: +// +// WHAT "EVERY ISSUE STATE" EXCLUDES. A corpus may RESERVE a label: every +// workflow declines it in `unless_labels` and none selects on it, so an issue +// carrying it binds nothing in every state — on purpose. The shared store did +// exactly that when it dropped its never-run `release` and `retro` pipelines +// and kept each sibling's exclusion "so a stray label fails activation loudly +// instead of silently binding standard-change". That refusal is routing +// working as designed, not a gap, so the sweep sets reserved labels aside — +// derived from the clauses, asserted through the production predicate, and +// logged by name (partitionLabels, routableVocabulary) — and demands +// exactly-one-match over everything that remains. + +// sweepLabels is the label vocabulary the [match] clauses discriminate on: // every label named in a `labels_any`, `labels_all`, or `unless_labels` across -// the set, plus the ones the pipelines are selected by. +// the set, plus the ones the pipelines are selected by. partitionLabels then +// splits it into what the sweep enumerates and what the corpus reserves. // // It is derived from the parsed clauses rather than hand-listed, so a label // introduced by a future workflow enters the sweep automatically. A @@ -67,14 +80,129 @@ func sweepLabels(defs []*Definition) []string { return out } +// labelPartition is sweepLabels split three ways by how the corpus treats +// each label — see partitionLabels. +type labelPartition struct { + routable []string // enumerated by the sweep: selected by some workflow, or orphaned + reserved []string // declined by every workflow, selected by none: set aside + // orphaned maps a label no workflow selects on but only SOME decline to + // the workflows that do not decline it — an exclusion that lost its + // owner without becoming a reservation. + orphaned map[string][]string +} + +// partitionLabels splits the vocabulary into the ROUTABLE labels the sweep +// enumerates and the RESERVED labels it deliberately leaves out. +// +// A label is reserved when NO workflow selects on it (it appears in no +// `labels_any` or `labels_all`) and EVERY workflow declines it (each one has a +// `[match]` table naming it in `unless_labels`). Every state carrying such a +// label binds nothing — each clause's exclusion term fires — and that is the +// corpus authors' decision made once per workflow, not an omission; a +// workflow that starts selecting on the label lifts the reservation on its +// own, because "selected by none" stops holding. +// +// Unanimity is what makes it intent. A label that some workflows decline and +// none selects is an ORPHANED exclusion: an issue carrying it routes on its +// other labels through whichever workflows lack the term and binds nothing +// when those are absent. Orphans stay routable so the sweep surfaces their +// states as the gaps they are, and routableVocabulary names the cause +// alongside. A workflow with no `[match]` table admits everything, so its +// presence makes every unselected label an orphan rather than a reservation. +func partitionLabels(defs []*Definition, labels []string) labelPartition { + selected := make(map[string]struct{}) + for _, d := range defs { + if d.Match == nil { + continue + } + for _, l := range d.Match.LabelsAny { + selected[l] = struct{}{} + } + for _, l := range d.Match.LabelsAll { + selected[l] = struct{}{} + } + } + + p := labelPartition{orphaned: make(map[string][]string)} + for _, l := range labels { + if _, ok := selected[l]; ok { + p.routable = append(p.routable, l) + continue + } + var admits []string + for _, d := range defs { + if d.Match == nil || !slices.Contains(d.Match.UnlessLabels, l) { + admits = append(admits, d.Pipeline.Name) + } + } + if len(admits) == 0 { + p.reserved = append(p.reserved, l) + continue + } + p.routable = append(p.routable, l) + p.orphaned[l] = admits + } + return p +} + +// routableVocabulary is the prologue the three sweeps share: derive the +// vocabulary, set the reserved labels aside, ASSERT what makes them reserved, +// and enforce the non-vacuity floor on what remains. +// +// The reserved set is asserted rather than merely skipped. partitionLabels +// reads `unless_labels` directly — a re-implementation of one of the four +// terms, which is exactly what this file's header warns against — so every +// reserved label is re-checked through the production predicate: for every +// kind, the bare state must bind NO workflow. The set is then logged by name, +// so the test output states which labels the corpus refuses by design. An +// orphaned exclusion fails here with its cause, on top of the states the +// sweep itself reports. +func routableVocabulary(t *testing.T, defs []*Definition, kinds []string) (routable, reserved []string) { + t.Helper() + p := partitionLabels(defs, sweepLabels(defs)) + + orphans := make([]string, 0, len(p.orphaned)) + for l := range p.orphaned { + orphans = append(orphans, l) + } + sort.Strings(orphans) + for _, l := range orphans { + t.Errorf("label %q is selected by no workflow and declined by only some: %v admit it; "+ + "either every workflow must decline it (reserving it, so the label fails "+ + "activation loudly) or one must select on it", l, p.orphaned[l]) + } + + for _, l := range p.reserved { + for _, kind := range kinds { + subject := Subject{Kind: kind, Labels: []string{l}} + for _, d := range defs { + if d.Match.Matches(subject) { + t.Errorf("reserved label %q binds workflow %q at kind=%s; a label every "+ + "clause declines cannot bind, so the partition and the predicate disagree", + l, d.Pipeline.Name, kind) + } + } + } + } + if len(p.reserved) > 0 { + t.Logf("reserved labels, declined by every workflow and selected by none, so every "+ + "state carrying one binds nothing by design and is left out of the sweep: %v", + p.reserved) + } + + requireNonVacuousLabels(t, p.routable) + return p.routable, p.reserved +} + // minShippedWorkflows and minSweepLabels are non-vacuity floors (AC-5): a // corpus that shrinks toward "one workflow with a match clause that // discriminates on nothing" still parses and still sweeps, but sweeps a // near-empty state space and PASSES — a degenerate corpus of exactly that // shape was reproduced logging "swept 5 states ... of 0 labels" against 465 -// for the real one. Both floors sit well below the shipped corpus (9 -// workflows, 8 labels as of this writing) so ordinary corpus edits do not -// trip them, and well above the degenerate case so a real shrink does. +// for the real one. Both floors sit well below the shipped corpus (8 +// workflows; 11 labels named, 9 routable and 2 reserved, as of this writing) +// so ordinary corpus edits do not trip them, and well above the degenerate +// case so a real shrink does. const ( minShippedWorkflows = 6 minSweepLabels = 4 @@ -166,11 +294,12 @@ func loadShippedWorkflows(t *testing.T) []*Definition { // requireNonVacuousLabels enforces the label half of the non-vacuity floor // (see minSweepLabels): a corpus whose match clauses stop discriminating on // labels still sweeps and still passes, over a state space too small to mean -// anything. +// anything. It is measured over the ROUTABLE vocabulary — a corpus that +// reserved every label it names would sweep nothing and must trip it. func requireNonVacuousLabels(t *testing.T, labels []string) { t.Helper() if len(labels) < minSweepLabels { - t.Fatalf("only %d labels discriminated across the swept corpus, want at least %d; "+ + t.Fatalf("only %d routable labels discriminated across the swept corpus, want at least %d; "+ "a corpus whose match clauses stopped discriminating on labels would sweep a "+ "near-empty state space and pass", len(labels), minSweepLabels) } @@ -247,14 +376,17 @@ func kindStrings() []string { // activate at all; multi-match means activation refuses with exactly-one-match. // The multi count was already 0 before this check existed and must stay there — // a sweep that only watched the zero side would let a precedence bug in. +// +// "Every issue state" is every state over the ROUTABLE vocabulary: a state +// carrying a reserved label binds nothing by the corpus's own design and is +// asserted as such in routableVocabulary rather than counted as a gap here. func TestShippedWorkflowsRouteEveryIssueState(t *testing.T) { defs := loadShippedWorkflows(t) - labels := sweepLabels(defs) - requireNonVacuousLabels(t, labels) + labels, reserved := routableVocabulary(t, defs, kindStrings()) res := sweep(defs, kindStrings(), labels, 3) - t.Logf("swept %d states over %d kinds x label subsets<=3 of %d labels", - res.total, len(kindStrings()), len(labels)) + t.Logf("swept %d states over %d kinds x label subsets<=3 of %d routable labels (%d reserved)", + res.total, len(kindStrings()), len(labels), len(reserved)) if len(res.zero) > 0 { t.Errorf("%d issue states bind NO workflow and cannot activate; first 10:\n %s", @@ -272,11 +404,11 @@ func TestShippedWorkflowsRouteEveryIssueState(t *testing.T) { // simultaneous labels is still a gap. func TestShippedWorkflowsRouteFullPowerset(t *testing.T) { defs := loadShippedWorkflows(t) - labels := sweepLabels(defs) - requireNonVacuousLabels(t, labels) + labels, reserved := routableVocabulary(t, defs, kindStrings()) res := sweep(defs, kindStrings(), labels, len(labels)) - t.Logf("swept %d states over the full 2^%d powerset", res.total, len(labels)) + t.Logf("swept %d states over the full 2^%d powerset of routable labels (%d reserved)", + res.total, len(labels), len(reserved)) if len(res.zero) > 0 { t.Errorf("%d issue states bind NO workflow; first 10:\n %s", @@ -379,9 +511,8 @@ func TestNoKindListEnumeratesTheClosedSet(t *testing.T) { // nothing, which is the regression this pins. func TestAddingASixthKindKeepsEveryStateRoutable(t *testing.T) { defs := loadShippedWorkflows(t) - labels := sweepLabels(defs) - kinds := append(kindStrings(), "spike") + labels, _ := routableVocabulary(t, defs, kinds) res := sweep(defs, kinds, labels, 3) if len(res.zero) > 0 { From ce514947b07374365876912e6fdf3b0a5b9e09d8 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:48:05 -0700 Subject: [PATCH 012/397] fix(model): keep related_to typo tolerance a CLI-only courtesy - ParseRelationType no longer accepts related_to/related-to; it backs the wire format and the workflow-spec vocabulary, both of which are meant to stay exactly the documented set - move the typo tolerance to a new parseRelationTypeArg helper shared by link add and link remove, where it only ever affects a human typing at a terminal - add tests pinning the parser, the JSON decode, and the CLI helper so a future alias can't widen the wire format silently again --- internal/cli/issue_link.go | 19 +++++++++++-- internal/cli/issue_link_test.go | 42 +++++++++++++++++++++++++++++ internal/model/model_test.go | 48 +++++++++++++++++++++++++++++++-- internal/model/relation.go | 13 +++++---- 4 files changed, 111 insertions(+), 11 deletions(-) create mode 100644 internal/cli/issue_link_test.go diff --git a/internal/cli/issue_link.go b/internal/cli/issue_link.go index 208bcda8..93c2da28 100644 --- a/internal/cli/issue_link.go +++ b/internal/cli/issue_link.go @@ -29,6 +29,21 @@ type unlinkResult struct { RelationType string `json:"relation_type"` } +// parseRelationTypeArg parses the argument typed at `link add` / +// `link remove`. It is model.ParseRelationType plus one CLI-only courtesy: +// "related_to" / "related-to" — the natural mis-typing of "relates_to", and the +// normalized form the refusal message already echoes back — resolves to the +// canonical type (DKT-1073). The tolerance stops at the command line on +// purpose: the JSON wire format (model.Relation.UnmarshalJSON) and the workflow +// `issue.linked..` vocabulary enumerated in +// docs/design/engine-spec.md stay exactly the canonical token set (DKT-1077). +func parseRelationTypeArg(arg string) (model.RelationType, error) { + if strings.ReplaceAll(strings.TrimSpace(arg), "-", "_") == "related_to" { + return model.RelationRelatesTo, nil + } + return model.ParseRelationType(arg) +} + var linkCmd = &cobra.Command{ Use: "link", Short: "Manage issue relations", @@ -47,7 +62,7 @@ var linkAddCmd = &cobra.Command{ return err } - relType, err := model.ParseRelationType(args[1]) + relType, err := parseRelationTypeArg(args[1]) if err != nil { return cmdErr(fmt.Errorf("%w", err), output.ErrValidation) } @@ -101,7 +116,7 @@ var linkRemoveCmd = &cobra.Command{ return err } - relType, err := model.ParseRelationType(args[1]) + relType, err := parseRelationTypeArg(args[1]) if err != nil { return cmdErr(fmt.Errorf("%w", err), output.ErrValidation) } diff --git a/internal/cli/issue_link_test.go b/internal/cli/issue_link_test.go new file mode 100644 index 00000000..59c60caf --- /dev/null +++ b/internal/cli/issue_link_test.go @@ -0,0 +1,42 @@ +package cli + +import ( + "testing" + + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// TestParseRelationTypeArg pins DKT-1073's typo courtesy to the command +// boundary it was asked for (DKT-1077): `issue link add/remove` accepts +// "related_to" / "related-to" for "relates_to" — the natural mis-typing of the +// canonical name — on top of everything model.ParseRelationType accepts, and +// nothing else. The model parser stays canonical so the JSON wire format and +// the workflow `issue.linked.` vocabulary are unchanged. +func TestParseRelationTypeArg(t *testing.T) { + tests := []struct { + input string + want model.RelationType + wantErr bool + }{ + {"relates_to", model.RelationRelatesTo, false}, + {"relates-to", model.RelationRelatesTo, false}, + {"related_to", model.RelationRelatesTo, false}, + {"related-to", model.RelationRelatesTo, false}, + {" related_to ", model.RelationRelatesTo, false}, + {"blocks", model.RelationBlocks, false}, + {"depends-on", model.RelationDependsOn, false}, + {"duplicates", model.RelationDuplicates, false}, + {"relate_to", "", true}, + {"invalid", "", true}, + } + for _, tt := range tests { + got, err := parseRelationTypeArg(tt.input) + if (err != nil) != tt.wantErr { + t.Errorf("parseRelationTypeArg(%q) error = %v, wantErr %v", tt.input, err, tt.wantErr) + continue + } + if got != tt.want { + t.Errorf("parseRelationTypeArg(%q) = %q, want %q", tt.input, got, tt.want) + } + } +} diff --git a/internal/model/model_test.go b/internal/model/model_test.go index 773fdb9a..21fce83d 100644 --- a/internal/model/model_test.go +++ b/internal/model/model_test.go @@ -2,6 +2,7 @@ package model import ( "encoding/json" + "fmt" "testing" "time" @@ -276,10 +277,15 @@ func TestParseRelationType(t *testing.T) { {"depends-on", RelationDependsOn, false}, {"relates_to", RelationRelatesTo, false}, {"relates-to", RelationRelatesTo, false}, - {"related_to", RelationRelatesTo, false}, - {"related-to", RelationRelatesTo, false}, {"duplicates", RelationDuplicates, false}, {"invalid", "", true}, + // DKT-1077: "related_to" is a CLI typo courtesy + // (cli.parseRelationTypeArg), NOT a model-level alias — this parser + // also decodes the JSON wire format and backs the workflow + // `issue.linked.` vocabulary, both of which are exactly the + // canonical token set. + {"related_to", "", true}, + {"related-to", "", true}, } for _, tt := range tests { @@ -316,6 +322,12 @@ func TestParseRelationDirection(t *testing.T) { {"duplicate_of", RelationDuplicates, true, false}, {"specified_by", "", false, true}, {"", "", false, true}, + // DKT-1077: the vocabulary this parser accepts must stay the one + // RelationDirectionTokens() names and docs/design/engine-spec.md + // enumerates, so a workflow's `issue.linked.related_to.` is + // refused rather than silently resolving to relates_to. + {"related_to", "", false, true}, + {"related-to", "", false, true}, } for _, tt := range tests { got, inverse, err := ParseRelationDirection(tt.input) @@ -331,6 +343,38 @@ func TestParseRelationDirection(t *testing.T) { } } +// TestRelationUnmarshalJSONVocabulary pins the JSON wire format's relation +// vocabulary (DKT-1077): Relation.UnmarshalJSON decodes exactly what +// MarshalJSON emits — the canonical types in either spelling — and refuses a +// near-miss like "related_to", whose tolerance is a CLI-argument courtesy that +// must not reach an exported/imported database or any other producer's JSON. +func TestRelationUnmarshalJSONVocabulary(t *testing.T) { + const shell = `{"id":1,"source_issue_id":"DKT-1","target_issue_id":"DKT-2",` + + `"relation_type":%q,"created_at":"2026-09-02T00:00:00Z"}` + tests := []struct { + token string + want RelationType + wantErr bool + }{ + {"relates_to", RelationRelatesTo, false}, + {"relates-to", RelationRelatesTo, false}, + {"depends_on", RelationDependsOn, false}, + {"related_to", "", true}, + {"related-to", "", true}, + } + for _, tt := range tests { + var rel Relation + err := json.Unmarshal([]byte(fmt.Sprintf(shell, tt.token)), &rel) + if (err != nil) != tt.wantErr { + t.Errorf("Unmarshal relation_type %q error = %v, wantErr %v", tt.token, err, tt.wantErr) + continue + } + if err == nil && rel.RelationType != tt.want { + t.Errorf("Unmarshal relation_type %q = %q, want %q", tt.token, rel.RelationType, tt.want) + } + } +} + func TestRelationTypeInverse(t *testing.T) { tests := []struct { rt RelationType diff --git a/internal/model/relation.go b/internal/model/relation.go index c62258b4..68852248 100644 --- a/internal/model/relation.go +++ b/internal/model/relation.go @@ -35,15 +35,14 @@ func ValidateRelationType(rt RelationType) error { } // ParseRelationType accepts both hyphenated ("depends-on") and underscored ("depends_on") -// forms and returns the canonical underscored RelationType. "related_to" / -// "related-to" is accepted as an alias of "relates_to": the natural -// mis-typing of the canonical name, and the normalized form the error message -// already surfaces (DKT-1073). +// forms and returns the canonical underscored RelationType. The accepted set is +// exactly validRelationTypes in either spelling — this function backs the JSON +// wire format (Relation.UnmarshalJSON) and, through ParseRelationDirection, the +// workflow `issue.linked..` vocabulary, so no near-miss spelling +// is tolerated here. The CLI's own typo courtesy for "related_to" lives at the +// command boundary instead (internal/cli/issue_link.go, DKT-1073/DKT-1077). func ParseRelationType(input string) (RelationType, error) { normalized := RelationType(strings.ReplaceAll(strings.TrimSpace(input), "-", "_")) - if normalized == "related_to" { - normalized = RelationRelatesTo - } if err := ValidateRelationType(normalized); err != nil { return "", err } From 9d38c3bc0c28fec5e56dc9ce2d2b8aaadfb4f77c Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:11:50 -0700 Subject: [PATCH 013/397] feat(engine): give a run a note channel every packet renders - new run_notes table (schema v26): append-only, run-scoped, capped at 16 KiB per note, dead with the run - docket run note add RUN-N --text|--file records one; refuses an empty or oversized note, an unknown run, or a done/abandoned run - docket run note list RUN-N renders them back, human and --json - context assembly reads notes as a sixth source alongside the existing five, so step context/claim/render all carry them - the default packet template renders each note verbatim as its own section right after the request body, so an operator ruling made before dispatch reaches every step of the run instead of getting rediscovered per step - documented in the engine spec and reliability delta --- docs/design/engine-spec.md | 31 ++- docs/tdd/reliability-delta.md | 34 +++ internal/cli/run_note.go | 219 +++++++++++++++++++ internal/cli/run_note_test.go | 270 +++++++++++++++++++++++ internal/cli/step.go | 25 ++- internal/db/migrate_v10_test.go | 20 +- internal/db/migrate_v26_test.go | 162 ++++++++++++++ internal/db/run_notes.go | 83 +++++++ internal/db/schema.go | 68 +++++- internal/engine/context.go | 72 ++++++- internal/engine/event.go | 14 ++ internal/engine/events_read.go | 7 + internal/engine/events_test.go | 16 +- internal/engine/note.go | 141 ++++++++++++ internal/engine/note_test.go | 311 +++++++++++++++++++++++++++ internal/engine/packets/default.tmpl | 5 + internal/engine/render.go | 20 +- 17 files changed, 1459 insertions(+), 39 deletions(-) create mode 100644 internal/cli/run_note.go create mode 100644 internal/cli/run_note_test.go create mode 100644 internal/db/migrate_v26_test.go create mode 100644 internal/db/run_notes.go create mode 100644 internal/engine/note.go create mode 100644 internal/engine/note_test.go diff --git a/docs/design/engine-spec.md b/docs/design/engine-spec.md index d059426b..f5541697 100644 --- a/docs/design/engine-spec.md +++ b/docs/design/engine-spec.md @@ -52,6 +52,11 @@ docket dispatch open|close|verify|abandon --run RUN-N # batch manifest (TTL'd, one open per run, # CAS); next refuses while open or # discrepancies exist; abandon unsticks +docket run note add|list RUN-N # a standing statement recorded once + # against the run, rendered as + # `== RUN NOTE N` in EVERY packet the + # run renders from then on and carried + # by `context.notes` (DKT-1079) docket step claim STEP-N # atomic; mints a capability token and # returns token + the step CONTEXT bundle docket step heartbeat|complete|fail # token via DOCKET_TOKEN env or stdin @@ -177,6 +182,22 @@ or a pinned instance template file. Packet *layout* is thereby core mechanics wh instance's harness hands the packet to an LLM as its prompt; core neither knows nor cares.) +**Run notes.** `docket run note add RUN-N --text "…"` (or `--file F`, `-` for stdin) +records a standing statement against a run — once — and every packet the run +renders from then on carries it verbatim as a `== RUN NOTE N` section directly +after `== REQUEST`, for every step of every issue; `step context` carries it as +`context.notes`, so a contract can name it. It is the one steering channel that is +run-wide: `step resolve -m` reaches the step it rules on (and the round it +authorizes), which is right for a ruling about one step's work and useless for a +fact about the whole run — a required gate known to fail on clean HEAD, the issue +tracking it, and the disposition already given, which a dispatcher learns before +the first dispatch and which every worker otherwise rediscovers and re-files +(DKT-1079). Issue comments remain an audit surface and never render. Notes are +append-only (a changed ruling is a second note, rendered after the first), capped +at 16 KiB because each rides every packet, refused on a terminal run, and each is +recorded as a `run-note-added` event carrying its text, attributed human. +`docket run note list RUN-N` reads them back in render order. + **Guards.** `docket guard spawn|record|stop|gate` are deterministic allow/deny predicates over engine state for harness enforcement points (exit 0 allow / exit 2 deny with reason): `spawn` — proposed rows byte-match the open dispatch and no @@ -575,7 +596,8 @@ next row { step, instance, issue, run, executor, class, attempt, claim response { step, token, lease_expires_ms, context } context { step: , issue: {id, title, body_snapshot, kind, labels, scope}, inputs: [{artifact, kind, producer_step, body, payload?}], - pins: [{path, sha256}], loop_entry, metadata, pre_gates? } + pins: [{path, sha256}], loop_entry, metadata, pre_gates?, + notes?: [{id, text, recorded_at_ms}] } # notes: DKT-1079 dispatch { dispatch, run, opened_seq, rows: […] } # verify = byte-equality on rows complete args --artifact-file F [--payload-file F] [--usage '{"unit":n,…}'] [--metadata '{…}'] (token via DOCKET_TOKEN env or stdin — §4) @@ -594,6 +616,11 @@ DKT-15)*. `pre_gates?` is an array of §11.4-shaped gate results for the step's a verdict is `unmatched` or timed out, null on ordinary pass/fail *(added 2026-08-03, DKT-19 / DKT-20)*. +`notes?` is the run's recorded notes (`docket run note add`, DKT-1079) in insertion +order, present only when the run has at least one, so every bundle of a run without +notes is byte-identical to before the member existed. `run note list RUN-N` and +`run note add`'s answer carry the same `{id, text, recorded_at_ms}` shape. + `docket step context STEP-N` re-emits `context` read-only (no token required; local inspection). `--meta` on it reports per-section byte counts — the closure-size record -(engine-core §8). +(engine-core §8) — `notes_bytes` among them, since a note rides every packet. diff --git a/docs/tdd/reliability-delta.md b/docs/tdd/reliability-delta.md index 50601585..47e3a004 100644 --- a/docs/tdd/reliability-delta.md +++ b/docs/tdd/reliability-delta.md @@ -661,6 +661,40 @@ NOTHING — no operator waived a stale-target warning before the verb for waiving one existed — and it is dormant: a run that never records a waiver reads byte-identically to v24 on every verb. +### AMENDMENT — the span extends to v26 (DKT-1079, 2026-09-02) + +**What changed.** v26 adds ONE table, `run_notes`: a standing statement +whoever drives a run records ONCE against it, and every packet the run +renders from then on carries — for every step of every issue, verbatim, as a +`== RUN NOTE N` section directly after `== REQUEST`, and in `step context` as +`context.notes`. A note is minted by `run note add` (one row, event-logged as +`run-note-added` carrying the text, attributed human) and read by context +assembly as its sixth source, in the claim's own transaction beside the other +five. It is append-only — no edit, no delete — because a packet is the record +of what a worker was told, and two renders of one step that disagreed about +that with nothing in the ledger between them would be exactly the drift the +snapshot discipline exists to prevent; a changed ruling is a second note. The +note dies with its run — the `run_id` FK is the whole scope rule. + +**What it fixes.** A packet's every writable source was step-scoped. RUN-70's +conductor gate-probed before dispatch, found `tests` failing on clean HEAD, +got the operator's disposition ("file issue, override-pass"), filed DKT-1075 +— and had nowhere to put any of it that a packet reads: issue comments are an +audit surface (§6.6), the body froze at activation, and no step had a routing +record for `step resolve -m` to reach. The executor stashed, re-ran the +suite, re-derived the failure, and filed DKT-1076, a duplicate the conductor +then spent three more calls closing. + +**Why the ratified arithmetic is untouched.** Like v11–v25, v26 is an +amendment, not a stage: one additive table, `CREATE TABLE IF NOT EXISTS` +throughout so the migration is idempotent and re-runnable, and a rewind guard +that probes the TABLE (the v24/v25 form, since v26 adds no column). It +BACK-FILLS NOTHING — no dispatcher recorded a note before the verb for +recording one existed — and it is dormant: a run that never records a note +reads byte-identically to v25 on every verb, every bundle, and every packet +(`notes` is `omitempty`, and the template's section renders only over a +non-empty list). + ### 2.1 The never-mutate rule engine-spec.md §3 requires v4 DBs open unchanged and existing verbs stay diff --git a/internal/cli/run_note.go b/internal/cli/run_note.go new file mode 100644 index 00000000..bfd85a33 --- /dev/null +++ b/internal/cli/run_note.go @@ -0,0 +1,219 @@ +package cli + +import ( + "fmt" + "io" + "os" + "strings" + "time" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/spf13/cobra" +) + +// `docket run note add|list` — DKT-1079. The dispatcher's channel into every +// packet of a run: a statement recorded once against the run, rendered as +// `== RUN NOTE N` beside the request in every packet the run renders from +// then on, and carried by `step context` as `notes`. + +var runNoteCmd = &cobra.Command{ + Use: "note", + Short: "Record standing statements that render in every packet of a run", + Long: `Record and list a run's notes. + +A packet draws from frozen and recorded state only: the issue as snapshotted at +activation, the recorded artifacts, the pins, the step's packet files, and the +step's own routing record. Issue comments never render, a mid-run description +edit never renders, and a ` + "`step resolve -m`" + ` note reaches only the step it rules +on. A fact about the WHOLE RUN had no channel — so a dispatcher that learned +before dispatch that a required gate fails on clean HEAD, got the operator's +disposition, and filed the tracking issue, could tell none of it to the workers, +and each one rediscovered the failure and filed a duplicate. + +A run note is that channel. ` + "`run note add`" + ` records the statement once; every +packet the run renders afterwards — every step, every issue, every round — +carries it verbatim as a ` + "`== RUN NOTE N`" + ` section right after ` + "`== REQUEST`" + `, +and ` + "`step context`" + ` exposes it as ` + "`notes`" + ` so a contract can name it. Notes are +append-only and each is recorded as a ` + "`run-note-added`" + ` event carrying its text, +so the feed says what every later worker was told. + +The shipped packet template renders notes; a custom ` + "`step render --template F`" + ` +renders them only where it ranges over ` + "`.Notes`" + ` (each has ` + "`.ID`" + ` and ` + "`.Text`" + `).`, +} + +var runNoteAddCmd = &cobra.Command{ + Use: "add RUN-N (--text T | --file F)", + Short: "Record a note that every later packet of the run carries", + Long: `Record one note against a run. + +Put in it what a worker would otherwise rediscover: the gate, why its failure +is pre-existing, the issue tracking it, and the disposition already given — +for example: + + docket run note add RUN-70 --text "Gate tests fails on clean HEAD \ + (routing_sweep_test.go), pre-existing and tracked as DKT-1075; \ + disposition: override-pass. Do not re-derive it and do not file a gap." + +--file reads the note from a file, or from stdin with "-", for a note that +does not fit a shell argument. One trailing newline is dropped so a file-fed +note renders exactly as the same words passed inline; nothing else is touched. + +The note is capped at 16 KiB because it rides every packet of the run; keep +the detail on the issue the note cites. Legal while the run is planning, +active, or parked — the motivating note is written BEFORE the first dispatch — +and refused on a done or abandoned run, which renders no more packets. A note +cannot be edited or removed: a packet is the record of what a worker was +told, so a changed ruling is recorded as a second note, which renders after +the first.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runRunNoteAdd(cmd, args[0], getWriter(cmd)) + }, +} + +var runNoteListCmd = &cobra.Command{ + Use: "list RUN-N", + Short: "List a run's notes in the order they render", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runRunNoteList(cmd, args[0], getWriter(cmd)) + }, +} + +// runNoteOutcome is `add`'s answer: the run and the note as recorded — the +// SAME shape the bundle's `notes` element carries, so a caller that reads one +// can read the other. +type runNoteOutcome struct { + Run string `json:"run"` + Note engine.RunNote `json:"note"` +} + +// runNoteListing is `list`'s answer. The notes ride under a named key rather +// than as a bare array so the envelope can gain a sibling field later without +// changing the shape a caller already parses. +type runNoteListing struct { + Run string `json:"run"` + Notes []engine.RunNote `json:"notes"` +} + +func runRunNoteAdd(cmd *cobra.Command, ref string, w *output.Writer) error { + conn := getDB(cmd) + + runID, err := model.ParseRunID(ref) + if err != nil { + return cmdErr(err, output.ErrValidation) + } + text, err := runNoteText(cmd) + if err != nil { + return err + } + + note, err := engine.AddRunNote(conn, runID, text, model.NowMS()) + if err != nil { + return runErr(err) + } + + outcome := runNoteOutcome{Run: model.FormatRunID(runID), Note: *note} + var message string + if !w.JSONMode { + message = fmt.Sprintf( + "Recorded note %d on %s (%d bytes); every packet the run renders "+ + "from now on carries it", note.ID, outcome.Run, len(note.Text)) + } + w.Success(outcome, message) + return nil +} + +// runNoteText resolves `--text` or `--file`, refusing both and neither: a +// note has exactly one source, and a verb that silently preferred one flag +// over the other would record something the caller did not write. +func runNoteText(cmd *cobra.Command) (string, error) { + text, _ := cmd.Flags().GetString("text") + path, _ := cmd.Flags().GetString("file") + hasText := cmd.Flags().Changed("text") + hasFile := cmd.Flags().Changed("file") + + switch { + case hasText && hasFile: + return "", cmdErr(fmt.Errorf("pass --text or --file, not both"), output.ErrValidation) + case !hasText && !hasFile: + return "", cmdErr(fmt.Errorf( + "a note needs its text: pass --text \"...\" or --file F (\"-\" reads stdin)"), + output.ErrValidation) + case hasText: + return text, nil + } + + // One byte over the cap is enough to let the engine refuse with the cap + // named; reading the rest of an oversized stream buys nothing. + var raw []byte + var err error + if path == "-" { + raw, err = io.ReadAll(io.LimitReader(os.Stdin, engine.RunNoteMaxBytes+1)) + if err != nil { + return "", cmdErr(fmt.Errorf("reading the note from stdin: %w", err), + output.ErrGeneral) + } + } else { + raw, err = os.ReadFile(path) + if err != nil { + if os.IsNotExist(err) { + return "", cmdErr(fmt.Errorf("note file %s not found", path), output.ErrNotFound) + } + return "", cmdErr(fmt.Errorf("reading the note file %s: %w", path, err), + output.ErrGeneral) + } + } + return string(raw), nil +} + +func runRunNoteList(cmd *cobra.Command, ref string, w *output.Writer) error { + conn := getDB(cmd) + + runID, err := model.ParseRunID(ref) + if err != nil { + return cmdErr(err, output.ErrValidation) + } + notes, err := engine.ListRunNotes(conn, runID) + if err != nil { + return runErr(err) + } + + listing := runNoteListing{Run: model.FormatRunID(runID), Notes: notes} + var message string + if !w.JSONMode { + message = renderRunNotes(listing) + } + w.Success(listing, message) + return nil +} + +// renderRunNotes is the listing's human form: one block per note, dated, the +// text indented under its header so a multi-line note reads as one entry. +func renderRunNotes(l runNoteListing) string { + if len(l.Notes) == 0 { + return fmt.Sprintf("%s has no notes", l.Run) + } + var b strings.Builder + fmt.Fprintf(&b, "%s has %d note(s)", l.Run, len(l.Notes)) + for _, n := range l.Notes { + when := time.UnixMilli(n.RecordedAtMS).UTC().Format(time.RFC3339) + fmt.Fprintf(&b, "\n [%d] %s", n.ID, when) + for _, line := range strings.Split(n.Text, "\n") { + fmt.Fprintf(&b, "\n %s", line) + } + } + return b.String() +} + +func init() { + runNoteAddCmd.Flags().String("text", "", + "The note, inline (exactly one of --text or --file)") + runNoteAddCmd.Flags().String("file", "", + "Read the note from this file; \"-\" reads stdin") + + runNoteCmd.AddCommand(runNoteAddCmd, runNoteListCmd) + runCmd.AddCommand(runNoteCmd) +} diff --git a/internal/cli/run_note_test.go b/internal/cli/run_note_test.go new file mode 100644 index 00000000..757844f8 --- /dev/null +++ b/internal/cli/run_note_test.go @@ -0,0 +1,270 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/output" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/spf13/cobra" +) + +// DKT-1079 at the CLI boundary: `run note add` records the note and reports +// it on both channels, `run note list` reads it back, and the verb refuses a +// note with no source (or two) naming the flags. WHICH packets a note then +// renders in is the engine's to prove (internal/engine/note_test.go); what +// this file asserts is the verb. + +func runNoteCmdWithDB(conn *sql.DB) *cobra.Command { + cmd := cmdWithDB(conn) + cmd.Flags().String("text", "", "") + cmd.Flags().String("file", "", "") + return cmd +} + +func TestRunNoteAddCLIRecordsAndReports(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + runRef := model.FormatRunID(runID) + + w, buf := bufWriter(true) + cmd := runNoteCmdWithDB(conn) + err := cmd.Flags().Set("text", "tests fails on clean HEAD; tracked as DKT-1075; override-pass") + testsupport.Must(t, err, "setting --text: %v", err) + err = runRunNoteAdd(cmd, runRef, w) + testsupport.Must(t, err, "run note add: %v", err) + + var env struct { + Data struct { + Run string `json:"run"` + Note engine.RunNote `json:"note"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Run != runRef || env.Data.Note.ID == 0 || + env.Data.Note.Text != "tests fails on clean HEAD; tracked as DKT-1075; override-pass" || + env.Data.Note.RecordedAtMS == 0 { + t.Fatalf("payload = %+v, want the note on %s with a real id and timestamp", env.Data, runRef) + } + + // The row really landed, run-scoped, and its event beside it. + var count int + err = conn.QueryRow(`SELECT COUNT(*) FROM run_notes WHERE run_id = ?`, runID).Scan(&count) + testsupport.Must(t, err, "counting notes: %v", err) + if count != 1 { + t.Errorf("run_notes rows = %d, want 1", count) + } + err = conn.QueryRow( + `SELECT COUNT(*) FROM events WHERE run_id = ? AND kind = 'run-note-added'`, runID, + ).Scan(&count) + testsupport.Must(t, err, "counting events: %v", err) + if count != 1 { + t.Errorf("run-note-added events = %d, want 1", count) + } +} + +func TestRunNoteAddCLIHumanMessageNamesTheNote(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + runRef := model.FormatRunID(runID) + + w, buf := bufWriter(false) + cmd := runNoteCmdWithDB(conn) + err := cmd.Flags().Set("text", "a ruling") + testsupport.Must(t, err, "setting --text: %v", err) + err = runRunNoteAdd(cmd, runRef, w) + testsupport.Must(t, err, "run note add: %v", err) + + out := buf.String() + if !strings.Contains(out, runRef) || !strings.Contains(out, "every packet") { + t.Errorf("the human message does not name the run and what the note does:\n%s", out) + } +} + +func TestRunNoteAddCLIReadsAFile(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + + path := filepath.Join(t.TempDir(), "note.txt") + err := os.WriteFile(path, []byte("line one\nline two\n"), 0o644) + testsupport.Must(t, err, "writing the note file: %v", err) + + w, buf := bufWriter(true) + cmd := runNoteCmdWithDB(conn) + err = cmd.Flags().Set("file", path) + testsupport.Must(t, err, "setting --file: %v", err) + err = runRunNoteAdd(cmd, model.FormatRunID(runID), w) + testsupport.Must(t, err, "run note add --file: %v", err) + + var env struct { + Data struct { + Note engine.RunNote `json:"note"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + // One trailing newline dropped, the rest verbatim. + if env.Data.Note.Text != "line one\nline two" { + t.Errorf("note text = %q, want the file's contents less one trailing newline", + env.Data.Note.Text) + } +} + +func TestRunNoteAddCLIRefusesNoSourceOrTwo(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + runRef := model.FormatRunID(runID) + + w, _ := bufWriter(true) + err := runRunNoteAdd(runNoteCmdWithDB(conn), runRef, w) + if err == nil { + t.Fatal("a note with no --text and no --file was accepted") + } + if !strings.Contains(err.Error(), "--text") || !strings.Contains(err.Error(), "--file") { + t.Errorf("the refusal %q does not name both flags", err) + } + var ce *CmdError + if !asCmdErr(err, &ce) || ce.Code != output.ErrValidation { + t.Errorf("the refusal is not VALIDATION_ERROR: %v", err) + } + + cmd := runNoteCmdWithDB(conn) + _ = cmd.Flags().Set("text", "x") + _ = cmd.Flags().Set("file", "y") + err = runRunNoteAdd(cmd, runRef, w) + if err == nil || !strings.Contains(err.Error(), "not both") { + t.Errorf("both flags at once: err = %v, want a refusal naming the conflict", err) + } + + var count int + err = conn.QueryRow(`SELECT COUNT(*) FROM run_notes`).Scan(&count) + testsupport.Must(t, err, "counting notes: %v", err) + if count != 0 { + t.Errorf("run_notes rows after two refusals = %d, want 0", count) + } +} + +func TestRunNoteAddCLIMapsEngineCodes(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + + // Unknown run: NOT_FOUND. + w, _ := bufWriter(true) + cmd := runNoteCmdWithDB(conn) + _ = cmd.Flags().Set("text", "a ruling") + err := runRunNoteAdd(cmd, model.FormatRunID(runID+99), w) + var ce *CmdError + if !asCmdErr(err, &ce) || ce.Code != output.ErrNotFound { + t.Errorf("unknown run: err = %v, want NOT_FOUND", err) + } + + // Empty text: VALIDATION_ERROR from the engine, mapped. + cmd = runNoteCmdWithDB(conn) + _ = cmd.Flags().Set("text", " ") + err = runRunNoteAdd(cmd, model.FormatRunID(runID), w) + if !asCmdErr(err, &ce) || ce.Code != output.ErrValidation { + t.Errorf("blank note: err = %v, want VALIDATION_ERROR", err) + } +} + +func TestRunNoteListCLI(t *testing.T) { + conn := newTestDB(t) + runID, _ := seedRun(t, conn) + runRef := model.FormatRunID(runID) + + // Empty first: the human line says so rather than printing nothing. + w, buf := bufWriter(false) + err := runRunNoteList(cmdWithDB(conn), runRef, w) + testsupport.Must(t, err, "run note list (empty): %v", err) + if !strings.Contains(buf.String(), "no notes") { + t.Errorf("the empty listing does not say so:\n%s", buf.String()) + } + + for _, text := range []string{"first ruling", "second ruling\nwith a second line"} { + cmd := runNoteCmdWithDB(conn) + _ = cmd.Flags().Set("text", text) + wa, _ := bufWriter(true) + err := runRunNoteAdd(cmd, runRef, wa) + testsupport.Must(t, err, "run note add: %v", err) + } + + w, buf = bufWriter(true) + err = runRunNoteList(cmdWithDB(conn), runRef, w) + testsupport.Must(t, err, "run note list: %v", err) + var env struct { + Data struct { + Run string `json:"run"` + Notes []engine.RunNote `json:"notes"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &env); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + if env.Data.Run != runRef || len(env.Data.Notes) != 2 || + env.Data.Notes[0].Text != "first ruling" || + env.Data.Notes[1].Text != "second ruling\nwith a second line" || + env.Data.Notes[0].ID >= env.Data.Notes[1].ID { + t.Errorf("listing = %+v, want both notes in insertion order", env.Data) + } + + w, buf = bufWriter(false) + err = runRunNoteList(cmdWithDB(conn), runRef, w) + testsupport.Must(t, err, "run note list (human): %v", err) + out := buf.String() + for _, want := range []string{"2 note(s)", "first ruling", "second ruling", " with a second line"} { + if !strings.Contains(out, want) { + t.Errorf("the human listing lacks %q:\n%s", want, out) + } + } +} + +// TestRunNoteVerbsAreRegistered: `run note add|list` hang off `run`, so +// `run --help` and the derived unknown-verb list both name them. +func TestRunNoteVerbsAreRegistered(t *testing.T) { + var note *cobra.Command + for _, sub := range runCmd.Commands() { + if sub.Name() == "note" { + note = sub + } + } + if note == nil { + t.Fatalf("run has no `note` subcommand; registered: %s", runVerbList(runCmd)) + } + got := map[string]bool{} + for _, sub := range note.Commands() { + got[sub.Name()] = true + } + for _, want := range []string{"add", "list"} { + if !got[want] { + t.Errorf("run note has no %q subcommand", want) + } + } + for _, flag := range []string{"text", "file"} { + if runNoteAddCmd.Flags().Lookup(flag) == nil { + t.Errorf("run note add has no --%s flag", flag) + } + } +} + +// asCmdErr unwraps to the CLI's coded error, so a test asserts the exit code a +// script would see rather than a message. +func asCmdErr(err error, target **CmdError) bool { + if err == nil { + return false + } + ce, ok := err.(*CmdError) + if !ok { + return false + } + *target = ce + return true +} diff --git a/internal/cli/step.go b/internal/cli/step.go index 8efbb0dd..279531e1 100644 --- a/internal/cli/step.go +++ b/internal/cli/step.go @@ -1106,10 +1106,11 @@ var stepContextCmd = &cobra.Command{ Long: `Re-emit a step's context bundle. No token required. The bundle is assembled from the run's PINNED and SNAPSHOTTED state only: the -issue as it read at activation, the recorded input artifacts, and the pin list. -It never reads the live issue, never reads the working tree, and never opens a -pinned file. Two calls at the same run state are byte-identical, whatever has -been edited in between. +issue as it read at activation, the recorded input artifacts, the pin list, +and the run's recorded notes (` + "`notes`" + `, absent when the run has none — see +` + "`docket run note`" + `). It never reads the live issue, never reads the working +tree, and never opens a pinned file. Two calls at the same run state are +byte-identical, whatever has been edited in between. --meta reports per-section byte counts alongside the bundle, and says whether the template a render would use is pinned.`, @@ -1154,12 +1155,16 @@ var stepRenderCmd = &cobra.Command{ THE PACKET DRAWS FROM EXACTLY THESE SURFACES: the step row and its pinned workflow definition, the issue's description and title/kind/labels/scope AS SNAPSHOTTED AT ACTIVATION, the recorded input artifacts the step declares, the -pin list, the step's declared packet files, and the step's own routing record -(rendered as == RESOLUTION). Issue COMMENTS are never rendered, and a mid-run -edit to the issue's description never renders either — the packet reads the -activation-time snapshot. To steer a step's next execution, put the words in -` + "`step resolve ... -m`" + `: a retry note renders on the same step, and a -fix-round note renders in every packet of the round it authorizes. +pin list, the run's recorded notes (rendered as == RUN NOTE N, right after +== REQUEST), the step's declared packet files, and the step's own routing +record (rendered as == RESOLUTION). Issue COMMENTS are never rendered, and a +mid-run edit to the issue's description never renders either — the packet +reads the activation-time snapshot. To steer a step's next execution, put the +words in ` + "`step resolve ... -m`" + `: a retry note renders on the same step, and +a fix-round note renders in every packet of the round it authorizes. To tell +EVERY worker of the run something — a gate known to fail on clean HEAD, the +issue tracking it, the disposition already given — put it in +` + "`docket run note add RUN-N`" + `; it renders in every packet from then on. Without --template the shipped default is used; it ships in the binary, so it cannot drift. With --template F, if the run PINNED that path, the file's bytes diff --git a/internal/db/migrate_v10_test.go b/internal/db/migrate_v10_test.go index da009634..c6e4dba5 100644 --- a/internal/db/migrate_v10_test.go +++ b/internal/db/migrate_v10_test.go @@ -92,7 +92,7 @@ func TestMigrateToV10(t *testing.T) { // GAPS, because a missing entry turns Migrate's loop into a runtime error on a // user's database rather than a build-time failure on ours. func TestSchemaSpanIsComplete(t *testing.T) { - // The span now ends at v25. v11 through v25 are AMENDMENTS, not stages — + // The span now ends at v26. v11 through v26 are AMENDMENTS, not stages — // workflow retirement (DKT-21), the projects dimension (operator request, // 2026-08-09), vote provenance (DKT-71), per-seat vote spend (DKT-95), // artifact revisions (DKT-70), the retry-budget base (DKT-86/DKT-90), @@ -100,21 +100,21 @@ func TestSchemaSpanIsComplete(t *testing.T) { // measured-usage cap (DKT-238), operator loop grants (DKT-237), the // hollow-assurance marker (DKT-265), the pause origin (DKT-305), the // attempt-outcome breakdown (DKT-490), the batch gate-override grant - // (DKT-546), and the stale-target waiver (DKT-742) — each recorded in - // docs/tdd/reliability-delta.md §2 under its own heading, with the reason - // it needed a version and the argument that it leaves the ratified v5-v10 - // arithmetic untouched. + // (DKT-546), the stale-target waiver (DKT-742), and the run note + // (DKT-1079) — each recorded in docs/tdd/reliability-delta.md §2 under + // its own heading, with the reason it needed a version and the argument + // that it leaves the ratified v5-v10 arithmetic untouched. // // The tripwire itself is UNCHANGED IN PURPOSE. It still fails the next // author who reaches for a version without filing an amendment first — // this test firing is exactly how v11 through v23 came to be documented // rather than discovered afterwards. Raising the number without editing // that section is the move it exists to stop. - if currentSchemaVersion != 25 { - t.Errorf("currentSchemaVersion = %d, want 25 — the span of "+ - "docs/tdd/reliability-delta.md §2 ends at v25 (the stale-target "+ - "waiver). Moving past 25 needs an amendment against that "+ - "section, per docs/design/amendments.md", currentSchemaVersion) + if currentSchemaVersion != 26 { + t.Errorf("currentSchemaVersion = %d, want 26 — the span of "+ + "docs/tdd/reliability-delta.md §2 ends at v26 (the run note). "+ + "Moving past 26 needs an amendment against that section, per "+ + "docs/design/amendments.md", currentSchemaVersion) } for v := 2; v <= currentSchemaVersion; v++ { diff --git a/internal/db/migrate_v26_test.go b/internal/db/migrate_v26_test.go new file mode 100644 index 00000000..650c41e1 --- /dev/null +++ b/internal/db/migrate_v26_test.go @@ -0,0 +1,162 @@ +package db + +import ( + "regexp" + "sort" + "testing" +) + +// v26 — the run-note table (DKT-1079). The same two obligations as v24 and +// v25: the table arrives on a migrated store, and the rewind guard converges a +// store stamped 26 without it. There is deliberately no back-fill to test — no +// dispatcher recorded a note before the verb for recording one existed. The +// behavior the table exists for — a note minted at `run note add`, carried by +// every later context bundle and packet of the run, dead with its run — is +// the engine's to prove. + +// v26Tables is stated independently of v26Sentinels so the sentinel test +// cannot pass by both lists drifting together — the v7/v8 discipline. +var v26Tables = []string{ + "run_notes", +} + +func TestMigrateToV26(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + v, err := SchemaVersion(db) + if err != nil { + t.Fatalf("SchemaVersion: %v", err) + } + if v != currentSchemaVersion { + t.Errorf("schema_version = %d, want %d", v, currentSchemaVersion) + } + for _, table := range v26Tables { + if !hasTable(t, db, table) { + t.Errorf("%s missing after migration", table) + } + } + if !hasIndex(t, db, "idx_run_notes_run") { + t.Error("idx_run_notes_run missing after migration") + } +} + +// TestRewindGuardProbesEveryV26Sentinel derives the table list from the DDL +// itself: a later edit that adds a CREATE TABLE to v26DDL and forgets the +// sentinel fails HERE rather than shipping a database the guard silently +// declines to repair. +func TestRewindGuardProbesEveryV26Sentinel(t *testing.T) { + re := regexp.MustCompile(`(?i)CREATE TABLE IF NOT EXISTS\s+(\w+)`) + var created []string + for _, m := range re.FindAllStringSubmatch(v26DDL, -1) { + created = append(created, m[1]) + } + if len(created) == 0 { + t.Fatal("no CREATE TABLE statements found in v26DDL") + } + sort.Strings(created) + + for _, pair := range []struct { + name string + list []string + }{ + {"v26Sentinels", append([]string(nil), v26Sentinels...)}, + {"v26Tables", append([]string(nil), v26Tables...)}, + } { + got := pair.list + sort.Strings(got) + if len(got) != len(created) { + t.Fatalf("%s = %v, but v26DDL creates %v", pair.name, got, created) + } + for i := range created { + if got[i] != created[i] { + t.Errorf("%s = %v, but v26DDL creates %v", pair.name, got, created) + break + } + } + } +} + +// TestV26RewindGuardConvergesAStampedStore drops the table while leaving the +// stamp at 26 — the mid-change-binary database the guard comments describe — +// and asserts Migrate converges it. The TABLE form matters: v26 adds no +// column, so every v25 sentinel is present on such a store and a column probe +// would never fire. +func TestV26RewindGuardConvergesAStampedStore(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + mustExec(t, db, `DROP TABLE run_notes`) + if hasTable(t, db, "run_notes") { + t.Fatal("the fixture did not remove the table it is testing the recovery of") + } + + if err := Migrate(db); err != nil { + t.Fatalf("re-running Migrate on the stamped store: %v", err) + } + for _, table := range v26Sentinels { + if !hasTable(t, db, table) { + t.Fatalf("the rewind guard did not converge %s back", table) + } + } +} + +// TestRunNotesDieWithTheirRun pins the scope rule at the schema level: the +// run_id foreign key cascades, so a deleted run takes its notes with it and +// nothing can render a note for a run that no longer exists. +func TestRunNotesDieWithTheirRun(t *testing.T) { + db := mustOpen(t) + if err := Initialize(db); err != nil { + t.Fatalf("Initialize: %v", err) + } + if err := Migrate(db); err != nil { + t.Fatalf("Migrate: %v", err) + } + + run, err := InsertRun(db, 1, "a run", 0, 1_000) + if err != nil { + t.Fatalf("InsertRun: %v", err) + } + tx, err := db.Begin() + if err != nil { + t.Fatalf("Begin: %v", err) + } + id, err := InsertRunNoteTx(tx, RunNote{RunID: run.ID, Text: "standing ruling", CreatedAtMS: 2_000}) + if err != nil { + t.Fatalf("InsertRunNoteTx: %v", err) + } + if err := tx.Commit(); err != nil { + t.Fatalf("Commit: %v", err) + } + if id == 0 { + t.Fatal("InsertRunNoteTx returned no id") + } + + notes, err := ListRunNotes(db, run.ID) + if err != nil { + t.Fatalf("ListRunNotes: %v", err) + } + if len(notes) != 1 || notes[0].ID != id || notes[0].Text != "standing ruling" || + notes[0].RunID != run.ID || notes[0].CreatedAtMS != 2_000 { + t.Fatalf("ListRunNotes = %+v, want the one note back verbatim", notes) + } + + mustExec(t, db, `DELETE FROM runs WHERE id = ?`, run.ID) + var left int + if err := db.QueryRow(`SELECT COUNT(*) FROM run_notes`).Scan(&left); err != nil { + t.Fatalf("counting notes: %v", err) + } + if left != 0 { + t.Errorf("run_notes rows after deleting the run = %d, want 0 (ON DELETE CASCADE)", left) + } +} diff --git a/internal/db/run_notes.go b/internal/db/run_notes.go new file mode 100644 index 00000000..380b81c6 --- /dev/null +++ b/internal/db/run_notes.go @@ -0,0 +1,83 @@ +package db + +import ( + "database/sql" + "fmt" +) + +// RunNote is one standing statement recorded against a run by whoever drives +// it (DKT-1079): a fact the run's workers need that no other packet source +// can carry — a gate known to fail on clean HEAD, the issue tracking it, and +// the disposition the operator already gave. Every packet the run renders +// after the note lands carries it, for every step of every issue in the run. +// +// The note is RUN-SCOPED BY CONSTRUCTION — the run_id foreign key is the +// whole scope rule, exactly gate_override_grants' and stale_target_waivers' +// shape. It is also APPEND-ONLY: no edit, no delete. A packet is the record +// of what a worker was told, and a note that could be rewritten after the +// fact would make two renders of one step disagree about that with nothing +// in the ledger explaining the difference. A ruling that changes is recorded +// as a second note, which renders after the first. +type RunNote struct { + ID int + RunID int + Text string + CreatedAtMS int64 +} + +// InsertRunNoteTx records one note and returns its id. +// +// It takes a transaction because the verb records the note and its +// `run-note-added` event together: a note with no event would be a packet +// change no feed reader can attribute, and an event with no note would name a +// ruling that renders nowhere. +func InsertRunNoteTx(tx *sql.Tx, n RunNote) (int, error) { + res, err := tx.Exec( + `INSERT INTO run_notes (run_id, text, created_at_ms) VALUES (?, ?, ?)`, + n.RunID, n.Text, n.CreatedAtMS) + if err != nil { + return 0, fmt.Errorf("recording the run note: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, fmt.Errorf("reading the run note id: %w", err) + } + return int(id), nil +} + +// ListRunNotesTx returns every note recorded for one run, in insertion order +// — the order they render in, since a later note may qualify an earlier one. +// +// It takes a transaction because its main consumer is context assembly, which +// reads every source inside the claim's own transaction so a claimant sees +// one consistent run state (engine-core §8's "one atomic mediation"). +func ListRunNotesTx(tx *sql.Tx, runID int) ([]RunNote, error) { + rows, err := tx.Query( + `SELECT id, run_id, text, created_at_ms + FROM run_notes WHERE run_id = ? ORDER BY id`, runID) + if err != nil { + return nil, fmt.Errorf("reading run notes: %w", err) + } + return scanRunNotes(rows) +} + +// ListRunNotes is ListRunNotesTx on the pooled connection, for the read verb. +func ListRunNotes(conn *sql.DB, runID int) ([]RunNote, error) { + rows, err := conn.Query( + `SELECT id, run_id, text, created_at_ms + FROM run_notes WHERE run_id = ? ORDER BY id`, runID) + if err != nil { + return nil, fmt.Errorf("reading run notes: %w", err) + } + return scanRunNotes(rows) +} + +func scanRunNotes(rows *sql.Rows) ([]RunNote, error) { + return scanRows(rows, "run notes", func(r *sql.Rows) (RunNote, error) { + var n RunNote + if err := r.Scan(&n.ID, &n.RunID, &n.Text, &n.CreatedAtMS); err != nil { + return RunNote{}, fmt.Errorf("reading a run note: %w", err) + } + return n, nil + }) +} diff --git a/internal/db/schema.go b/internal/db/schema.go index 034c999d..fe1cf10b 100644 --- a/internal/db/schema.go +++ b/internal/db/schema.go @@ -10,7 +10,7 @@ import ( "github.com/ALT-F4-LLC/docket/internal/schema" ) -const currentSchemaVersion = 25 +const currentSchemaVersion = 26 // schemaDDL contains the CREATE TABLE statements for the initial schema. // @@ -194,6 +194,7 @@ var migrations = map[int]func(tx *sql.Tx) error{ 23: migrateV22ToV23, 24: migrateV23ToV24, 25: migrateV24ToV25, + 26: migrateV25ToV26, } // migrationsNeedingFKOff names the migrations that REBUILD tables and so must @@ -2298,6 +2299,53 @@ func migrateV24ToV25(tx *sql.Tx) error { return nil } +// v26Sentinels is the table the v26 DDL creates, probed by the rewind guard in +// the TABLE form v24 and v25 use. TestRewindGuardProbesEveryV26Sentinel derives +// the list from the DDL, so an added table cannot ship without its sentinel. +var v26Sentinels = []string{ + "run_notes", +} + +// v26DDL is the run-scoped note (DKT-1079): a standing statement whoever +// drives a run records ONCE, and every packet the run renders from then on +// carries — the channel a dispatcher had no way to reach a worker through. +// RUN-70's conductor learned before dispatch that a required gate fails on +// clean HEAD, got an operator disposition, and filed a tracking issue; nothing +// it could write reached the executor's packet (issue comments are an audit +// surface, never a context source, and the issue body froze at activation), so +// the executor re-derived the failure and filed a duplicate. +// +// It is a TABLE beside gate_override_grants and stale_target_waivers rather +// than a column because it is the same shape — standing operator context, +// run-scoped by the run_id foreign key, dead with its run. A note is +// append-only: there is no edit and no delete, because a packet that rendered +// a note and a packet that did not must be distinguishable in the record. +const v26DDL = ` +CREATE TABLE IF NOT EXISTS run_notes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER NOT NULL REFERENCES runs(id) ON DELETE CASCADE, + text TEXT NOT NULL, + created_at_ms INTEGER NOT NULL +); + +CREATE INDEX IF NOT EXISTS idx_run_notes_run + ON run_notes(run_id); +` + +// migrateV25ToV26 creates the run-note table (DKT-1079). +// +// Additive and dormant, exactly as v24 and v25 were: it creates one new table +// and touches no existing one, so a run that never records a note reads +// byte-identically to v25 — every context bundle and every packet. There is +// nothing to back-fill: no dispatcher recorded a note before the verb for +// recording one existed. +func migrateV25ToV26(tx *sql.Tx) error { + if _, err := tx.Exec(v26DDL); err != nil { + return fmt.Errorf("migrating v25 to v26: %w", err) + } + return nil +} + // migrateV19ToV20 adds the operator loop-grant column. // // It BACK-FILLS NOTHING, and zero is the correct value for every existing row: @@ -2851,6 +2899,24 @@ func Migrate(db *sql.DB) error { } } + // The v26 guard, same TABLE form as v24 and v25 and for their reason: v26 + // adds a table and no columns, so a database stamped 26 by a binary built + // mid-change carries every v25 sentinel and the note table never arrives. + // The v26 migration is CREATE TABLE IF NOT EXISTS throughout, so re-running + // it against such a store is safe. + if version >= 26 { + for _, table := range v26Sentinels { + exists, err := tableExists(db, table) + if err != nil { + break + } + if !exists { + version = 25 + break + } + } + } + if version == currentSchemaVersion { return nil } diff --git a/internal/engine/context.go b/internal/engine/context.go index 3e617032..5f784243 100644 --- a/internal/engine/context.go +++ b/internal/engine/context.go @@ -15,18 +15,23 @@ import ( // The context bundle — engine-spec §11.4's `context`, and the determinism // engine-core §8 and §9 item 5 require of it. // -// ASSEMBLY IS PURE AND SNAPSHOT-PINNED. It reads, and may read, exactly five -// sources (TDD §6.6): +// ASSEMBLY IS PURE AND SNAPSHOT-PINNED. It reads, and may read, exactly six +// sources (TDD §6.6, plus DKT-1079's sixth): // // 1. the step row and its definition, from the PINNED workflow's `parsed` // 2. `run_issues.body_snapshot` — never the live issue body // 3. `run_issues.issue_snapshot` — title/kind/labels/scope as of activation // 4. recorded `artifacts` bodies, resolved per §6.7's input rule // 5. `pins` rows (path + hash) — the LIST, not the file contents +// 6. `run_notes` rows — the standing statements `run note add` recorded +// against the run, every one, in insertion order // // It never reads the working tree, never re-reads a file a pin names, and reads // NO LIVE ISSUE FIELD AT ALL: `id` is the run_issues key and every other member -// of §11.4's `context.issue` comes from a snapshot column. +// of §11.4's `context.issue` comes from a snapshot column. The sixth source is +// in the artifact category, not the issue one: a note is RECORDED RUN STATE, +// written once by a verb and event-logged, exactly as an artifact is written +// once at completion — never a live field an `issue edit` can move. // // The scheduler's live read of `issues.scope_globs` (§6.3 R4) is NOT context // assembly and is deliberately outside this file. Two different questions @@ -103,6 +108,24 @@ type Context struct { // nil when the step carries no routing record, so a first-round packet is // byte-identical to what it always was. Resolution *ContextResolution `json:"resolution,omitempty"` + // Notes are the run's standing statements (DKT-1079): every `run note + // add` recorded against THIS RUN, in insertion order, whichever issue or + // step the bundle is for. + // + // It is the one steering channel that is run-wide rather than + // step-scoped. `Resolution` reaches the step a ruling was issued against + // (and the round it authorizes), which is right for a ruling about one + // step's work and useless for a fact about the whole run — RUN-70's + // conductor learned BEFORE DISPATCH that a required gate fails on clean + // HEAD, got the operator's disposition, filed the tracking issue, and had + // nowhere to put any of it that a packet reads: issue comments are an + // audit surface (§6.6's source rule, by design), the body froze at + // activation, and no step had a routing record yet. The executor + // re-derived the failure and filed a duplicate. + // + // omitempty, so a run with no notes renders every bundle byte-identically + // to what it always was. + Notes []RunNote `json:"notes,omitempty"` } // ContextResolution is the routing that sent a step back, and its note. @@ -304,6 +327,14 @@ func AssembleContext( return nil, err } + // Source 6: the run's notes (DKT-1079), read in the same transaction as + // everything else so a claimant sees the notes that stood at the moment + // its token was minted. + notes, err := contextNotes(tx, step.RunID) + if err != nil { + return nil, err + } + metadata, err := decodeMetadata(step.Metadata) if err != nil { return nil, err @@ -320,10 +351,30 @@ func AssembleContext( Step: row, Issue: *issue, Inputs: inputs, Pins: pins, LoopEntry: loopEntryOf(step), Metadata: metadata, TargetSHA: sha, TargetWorktree: worktree, - Resolution: resolution, + Resolution: resolution, Notes: notes, }, nil } +// contextNotes reads the run's notes into the bundle's shape (DKT-1079). +// +// nil, not an empty slice, when the run has none: `omitempty` elides either, +// but a nil keeps the in-memory bundle a test compares field-by-field equal to +// one assembled before the field existed. +func contextNotes(tx *sql.Tx, runID int) ([]RunNote, error) { + rows, err := db.ListRunNotesTx(tx, runID) + if err != nil { + return nil, err + } + if len(rows) == 0 { + return nil, nil + } + out := make([]RunNote, 0, len(rows)) + for _, n := range rows { + out = append(out, RunNote{ID: n.ID, Text: n.Text, RecordedAtMS: n.CreatedAtMS}) + } + return out, nil +} + // attachGateOutcomes fills a resolution's failing-gate list (DKT-261). // // It is a SEPARATE pass rather than part of resolutionOf, because that function @@ -1521,7 +1572,12 @@ type ContextMeta struct { InputsBytes int `json:"inputs_bytes"` PinsBytes int `json:"pins_bytes"` MetadataBytes int `json:"metadata_bytes"` - TotalBytes int `json:"total_bytes"` + // NotesBytes is the run notes' share (DKT-1079). Counted, because a note + // rides EVERY packet of the run: a cap computed without it would let a + // bundle exceed its declared limit by exactly the words a dispatcher + // added to all of them. + NotesBytes int `json:"notes_bytes"` + TotalBytes int `json:"total_bytes"` // TemplatePinned reports whether the template a render would use is pinned. // An UNPINNED template is reported so the reproducibility gap is visible // rather than assumed (§6.11.1) — a packet rendered through an unpinned @@ -1552,6 +1608,10 @@ func (c *Context) Meta() ContextMeta { for k, v := range c.Metadata { meta.MetadataBytes += len(k) + len(fmt.Sprint(v)) } - meta.TotalBytes = meta.IssueBytes + meta.InputsBytes + meta.PinsBytes + meta.MetadataBytes + for _, n := range c.Notes { + meta.NotesBytes += len(n.Text) + } + meta.TotalBytes = meta.IssueBytes + meta.InputsBytes + meta.PinsBytes + + meta.MetadataBytes + meta.NotesBytes return meta } diff --git a/internal/engine/event.go b/internal/engine/event.go index 66933ec3..d4dc2af2 100644 --- a/internal/engine/event.go +++ b/internal/engine/event.go @@ -305,6 +305,19 @@ const ( // carried e642afb would be indistinguishable in the record from a packet // that rendered correctly. EventIssueDiffRepinned = "issue-diff-repinned" + + // A run note being recorded (DKT-1079): `run note add` put a standing + // statement into every packet the run renders from that moment on. + // + // It earns its place on the `run-repinned` argument, in yet another + // column: a note is the third thing that changes what a step's packet + // says while the run is live, beside a repin and a scope refresh. Two + // renders of one step that differ by a `== RUN NOTE` section would + // otherwise be indistinguishable in the record from the packet drift the + // snapshot discipline exists to prevent. It carries the note's id and its + // text verbatim, as `step-annotated` carries its annotation, so what + // every later worker was told survives in the feed itself. + EventRunNoteAdded = "run-note-added" ) // eventKinds is the closed set, as a set. The writer checks membership here, so @@ -341,6 +354,7 @@ var eventKinds = map[string]bool{ EventStaleTargetWaived: true, EventIssueScopeRefreshed: true, EventIssueDiffRepinned: true, + EventRunNoteAdded: true, } // recordEvent writes one event in the caller's transaction. diff --git a/internal/engine/events_read.go b/internal/engine/events_read.go index fc2f3c84..097be44a 100644 --- a/internal/engine/events_read.go +++ b/internal/engine/events_read.go @@ -555,6 +555,13 @@ var eventActors = map[string]Actor{ // the checkout the patched tree stands in, which is precisely what // `human` means in this table. EventIssueDiffRepinned: ActorHuman, + + // The run note (DKT-1079): nothing in the engine writes a note — the + // packet's other sources are frozen at activation or recorded by the + // steps themselves, and a note exists only because a person (or the + // dispatcher relaying one) ran `run note add` naming the run it should + // reach, which is precisely what `human` means in this table. + EventRunNoteAdded: ActorHuman, } // ActorFor reports which of the four causes an event kind is attributable to, diff --git a/internal/engine/events_test.go b/internal/engine/events_test.go index b558b7ee..882cba30 100644 --- a/internal/engine/events_test.go +++ b/internal/engine/events_test.go @@ -204,20 +204,30 @@ func TestEventKindsAreAClosedSet(t *testing.T) { // sha, so a packet that judged the wrong commit is distinguishable // in the record from one that rendered correctly. "issue-diff-repinned", + + // The run-note kind (DKT-1079) — ONE, the `run-repinned` argument + // in its fourth column: a note is the third thing that changes what + // a live run's packets say (beside a repin and a scope refresh), and + // two renders of one step that differ by a `== RUN NOTE` section + // must be distinguishable in the record from the drift the snapshot + // discipline exists to prevent. No "spent" counterpart: rendering a + // note changes no step's state, and every later render reads it the + // way every render reads the frozen body. + "run-note-added", } { if !eventKinds[kind] { t.Errorf("the spec names %q but eventKinds does not contain it", kind) } } - if len(eventKinds) != 48 { + if len(eventKinds) != 49 { t.Errorf("eventKinds has %d entries; §7.6 plus gates-trust §6.4/§8.1, "+ "payloads-thresholds §7.7, runs-dispatch §5/§6, events-follow "+ "§6/§7.3, DKT-35's annotation kind, DKT-61's tenancy kind, "+ "DKT-236's spawn carve-out, DKT-294's live-status mirror, "+ "DKT-408's repin kind, DKT-546's batch-override pair, "+ "DKT-742's stale-target waiver kind, DKT-869's scope-refresh "+ - "kind, and DKT-1034's issue.diff re-pin kind enumerate 48 "+ - "(see DKT-21)", + "kind, DKT-1034's issue.diff re-pin kind, and DKT-1079's "+ + "run-note kind enumerate 49 (see DKT-21)", len(eventKinds)) } } diff --git a/internal/engine/note.go b/internal/engine/note.go new file mode 100644 index 00000000..25024cb6 --- /dev/null +++ b/internal/engine/note.go @@ -0,0 +1,141 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "errors" + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" +) + +// DKT-1079 — the run note: the dispatcher's channel into every packet. +// +// A packet draws from frozen and recorded state only (render.go's exhaustive +// list), and until this verb every writable source in it was STEP-SCOPED: +// `step resolve -m` reaches the step it rules on and the round it authorizes. +// A fact about the WHOLE RUN had no channel at all. RUN-70's conductor +// gate-probed before dispatch, found `tests` failing on clean HEAD, asked the +// operator, got "file issue, override-pass", filed DKT-1075 — and dispatched +// packets that carried none of it, because issue comments never render +// (§6.6), the issue body froze at activation, and no step had a routing +// record yet. The executor stashed, re-ran the suite, re-derived the same +// failure, and filed DKT-1076: a duplicate the conductor then spent three more +// calls closing. +// +// `run note add` records the statement ONCE against the run. Context assembly +// reads it as its sixth source, the default template renders it as +// `== RUN NOTE N` beside the request, and `step context` carries it as +// `notes` so a contract can name it. It is append-only and event-logged, +// like every other post-activation move of a packet's premises (repin, scope +// refresh): what a worker was told is part of the record. + +// RunNoteMaxBytes caps one note. A note rides EVERY packet of the run, so an +// oversized one is paid for by every step; 16 KiB — MetadataMaxBytes' figure, +// for the same reason — is three orders of magnitude above the motivating use +// (a gate name, why it is pre-existing, an issue id, and a disposition: a few +// hundred bytes) and small enough that a note stays a note. Bulk detail goes +// in the issue the note cites. +const RunNoteMaxBytes = 16 << 10 + +// RunNote is one recorded note, in the ONE shape it has everywhere: the verb's +// answer, the `run note list` row, and the bundle's `notes` element are all +// this, so a consumer parses one shape wherever a note appears. +type RunNote struct { + ID int `json:"id"` + Text string `json:"text"` + // RecordedAtMS dates the note, so a reader of a packet can tell a note + // recorded before dispatch from one added mid-run. + RecordedAtMS int64 `json:"recorded_at_ms"` +} + +// AddRunNote records one note against a run, event-logged, in one +// transaction. +// +// It refuses an empty note (a note with no content steers nothing), one over +// the cap, an unknown run, and a TERMINAL run: a run that is done or abandoned +// renders no more packets, so a note against it would be a statement nobody +// will ever be told, recorded as if somebody would. A `planning` run is fine — +// the motivating note is written BEFORE the first dispatch, and the verb must +// not force a dispatcher to activate first. +// +// One trailing newline is dropped, so a note fed from a file renders exactly +// as the same words passed inline; nothing else about the text is touched. +func AddRunNote(conn *sql.DB, runID int, text string, nowMS int64) (*RunNote, error) { + text = strings.TrimSuffix(text, "\n") + if strings.TrimSpace(text) == "" { + return nil, validationErr( + "the note is empty; a note with no content records nothing") + } + if len(text) > RunNoteMaxBytes { + return nil, validationErr( + "the note is %d bytes, over the %d-byte cap; a note rides every packet "+ + "of the run, so record the detail on the issue it cites and keep the "+ + "note to the ruling", len(text), RunNoteMaxBytes) + } + + run, err := db.GetRun(conn, runID) + if errors.Is(err, db.ErrRunNotFound) { + return nil, notFoundErr(err, "run %s not found", model.FormatRunID(runID)) + } + if err != nil { + return nil, err + } + if run.Status.Terminal() { + return nil, conflictErr( + "run %s is %s; it renders no more packets, so a note against it "+ + "would reach nobody", run.Ref(), run.Status) + } + + tx, err := conn.Begin() + if err != nil { + return nil, fmt.Errorf("recording the run note: %w", err) + } + defer tx.Rollback() + + id, err := db.InsertRunNoteTx(tx, db.RunNote{ + RunID: runID, Text: text, CreatedAtMS: nowMS, + }) + if err != nil { + return nil, err + } + + // The event carries the note VERBATIM, as `step-annotated` carries its + // annotation: the feed must be able to say what every later worker was + // told without a join against the note table. + data, err := json.Marshal(map[string]any{"note": id, "text": text}) + if err != nil { + return nil, fmt.Errorf("recording the run note: %w", err) + } + if err := recordEvent(tx, eventRecord{ + Kind: EventRunNoteAdded, RunID: runID, Data: string(data), AtMS: nowMS, + }); err != nil { + return nil, err + } + + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("recording the run note: %w", err) + } + return &RunNote{ID: id, Text: text, RecordedAtMS: nowMS}, nil +} + +// ListRunNotes returns a run's notes in the order they render. +func ListRunNotes(conn *sql.DB, runID int) ([]RunNote, error) { + if _, err := db.GetRun(conn, runID); err != nil { + if errors.Is(err, db.ErrRunNotFound) { + return nil, notFoundErr(err, "run %s not found", model.FormatRunID(runID)) + } + return nil, err + } + rows, err := db.ListRunNotes(conn, runID) + if err != nil { + return nil, err + } + out := make([]RunNote, 0, len(rows)) + for _, n := range rows { + out = append(out, RunNote{ID: n.ID, Text: n.Text, RecordedAtMS: n.CreatedAtMS}) + } + return out, nil +} diff --git a/internal/engine/note_test.go b/internal/engine/note_test.go new file mode 100644 index 00000000..a877c905 --- /dev/null +++ b/internal/engine/note_test.go @@ -0,0 +1,311 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "strconv" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-1079: the run note — a statement recorded once against a run that every +// packet of the run renders from then on. +// +// The acceptance criteria, each falsified separately: a note added before +// dispatch appears VERBATIM in `step render` for EVERY step of the run; `step +// context` carries it as `notes`; and a run with no notes renders every bundle +// and packet byte-identically to before the field existed. + +const noteText = "Gate `tests` fails on clean HEAD (routing_sweep_test.go): " + + "pre-existing, tracked as DKT-1075; disposition override-pass.\n" + + "Do not re-derive it and do not file a gap for it." + +// noteRunSteps lists every step of a run, id and instance, so an assertion over +// "every packet of the run" enumerates the run rather than a hand-picked pair. +func noteRunSteps(t *testing.T, conn *sql.DB, runID int) map[int]string { + t.Helper() + rows, err := conn.Query(`SELECT id, instance FROM steps WHERE run_id = ? ORDER BY id`, runID) + testsupport.Must(t, err, "listing steps: %v", err) + defer rows.Close() + out := map[int]string{} + for rows.Next() { + var id int + var instance string + err := rows.Scan(&id, &instance) + testsupport.Must(t, err, "scanning a step: %v", err) + out[id] = instance + } + testsupport.Must(t, rows.Err(), "iterating steps: %v", rows.Err()) + return out +} + +func TestRunNoteRendersInEveryPacketOfTheRun(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + steps := noteRunSteps(t, conn, run.ID) + if len(steps) < 2 { + t.Fatalf("the fixture expanded %d step(s); this test needs several", len(steps)) + } + + // BEFORE: no `notes` member at all, so every bundle that predates the + // field is byte-identical — the omitempty half of the acceptance. + for id, instance := range steps { + if strings.Contains(bundleJSON(t, conn, id), `"notes"`) { + t.Errorf("%s's bundle carries a notes member before any note exists", instance) + } + } + + first, err := AddRunNote(conn, run.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + second, err := AddRunNote(conn, run.ID, "Second ruling: also pre-existing.", nowMS+1) + testsupport.Must(t, err, "AddRunNote (second): %v", err) + if second.ID <= first.ID { + t.Fatalf("note ids = %d then %d, want ascending", first.ID, second.ID) + } + + for id, instance := range steps { + // `step context`: the bundle carries both notes, in order, verbatim. + bundle, err := ReadContext(conn, id, nowMS) + testsupport.Must(t, err, "ReadContext(%s): %v", instance, err) + if len(bundle.Notes) != 2 { + t.Fatalf("%s's bundle carries %d note(s), want 2: %+v", + instance, len(bundle.Notes), bundle.Notes) + } + if bundle.Notes[0] != *first || bundle.Notes[1] != *second { + t.Errorf("%s's bundle notes = %+v, want [%+v %+v] verbatim and in order", + instance, bundle.Notes, *first, *second) + } + + // `step render`: the packet carries each note as its own section, + // verbatim, beside the request. + result, err := RenderStep(conn, id, "", nowMS) + testsupport.Must(t, err, "RenderStep(%s): %v", instance, err) + packet := result.Packet + for _, n := range []*RunNote{first, second} { + section := "\n== RUN NOTE " + itoa(n.ID) + "\n" + n.Text + "\n" + if !strings.Contains(packet, section) { + t.Errorf("%s's packet lacks note %d verbatim as its own section:\n%s", + instance, n.ID, packet) + } + } + request := strings.Index(packet, "== REQUEST") + note := strings.Index(packet, "== RUN NOTE ") + if request < 0 || note < request { + t.Errorf("%s's packet renders the notes before the request (%d vs %d)", + instance, note, request) + } + if strings.Contains(packet, "== FILE") && note > strings.Index(packet, "== FILE") { + t.Errorf("%s's packet renders the notes after the packet files; they belong "+ + "beside the request", instance) + } + } +} + +// TestRunNoteIsRunScoped: a note on one run reaches none of another run's +// packets — the run_id foreign key is the whole scope rule. +func TestRunNoteIsRunScoped(t *testing.T) { + conn := mustDB(t) + first, _ := activatedRun(t, conn) + + other := createIssue(t, conn, "another issue", "another body", "task", nil) + second := startRun(t, conn, other) + _, err := activate(conn, second.ID) + testsupport.Must(t, err, "activating the second run: %v", err) + + _, err = AddRunNote(conn, first.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + + for id, instance := range noteRunSteps(t, conn, second.ID) { + if strings.Contains(bundleJSON(t, conn, id), `"notes"`) { + t.Errorf("%s (run %s) carries a note recorded against %s", + instance, second.Ref(), first.Ref()) + } + } + for id := range noteRunSteps(t, conn, first.ID) { + bundle, err := ReadContext(conn, id, nowMS) + testsupport.Must(t, err, "ReadContext: %v", err) + if len(bundle.Notes) != 1 { + t.Fatalf("the noted run's own bundle carries %d note(s), want 1", len(bundle.Notes)) + } + } +} + +// TestRunNoteIsDeterministicAtFixedState: two reads at the same run state are +// byte-identical, and a note is RECORDED run state — the mid-run edits the +// snapshot discipline immunizes against still move nothing. +func TestRunNoteIsDeterministicAtFixedState(t *testing.T) { + conn := mustDB(t) + run, issue := activatedRun(t, conn) + _, err := AddRunNote(conn, run.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + stepID := stepIDByInstance(t, conn, "implement@0") + + before := bundleJSON(t, conn, stepID) + execSQL(t, conn, `UPDATE issues SET description = ? WHERE id = ?`, "EDITED BODY", issue) + after := bundleJSON(t, conn, stepID) + if before != after { + t.Errorf("a live issue edit moved a bundle carrying a note:\nbefore: %s\nafter: %s", + before, after) + } + + r1, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "RenderStep: %v", err) + r2, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "RenderStep: %v", err) + if r1.Packet != r2.Packet { + t.Error("two renders at one run state differ") + } +} + +func TestRunNoteRefusals(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + cases := []struct { + name string + run int + text string + code ErrorCode + want string + }{ + {"empty", run.ID, "", CodeValidation, "empty"}, + {"blank", run.ID, " \n\t\n", CodeValidation, "empty"}, + {"oversized", run.ID, strings.Repeat("x", RunNoteMaxBytes+1), CodeValidation, "cap"}, + {"unknown run", run.ID + 99, noteText, CodeNotFound, "not found"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, err := AddRunNote(conn, tc.run, tc.text, nowMS) + if err == nil { + t.Fatal("the note was accepted") + } + if code, ok := CodeOf(err); !ok || code != tc.code { + t.Errorf("code = %v (%v), want %s: %v", code, ok, tc.code, err) + } + if !strings.Contains(err.Error(), tc.want) { + t.Errorf("refusal %q does not say %q", err, tc.want) + } + }) + } + + // Exactly at the cap is fine: the cap is a ceiling, not a strict bound. + if _, err := AddRunNote(conn, run.ID, strings.Repeat("y", RunNoteMaxBytes), nowMS); err != nil { + t.Errorf("a note exactly at the cap was refused: %v", err) + } + + // A terminal run renders no more packets, so a note against it is CONFLICT. + err := db.SetRunStatus(conn, run.ID, model.RunAbandoned, "test", nowMS) + testsupport.Must(t, err, "abandoning the run: %v", err) + _, err = AddRunNote(conn, run.ID, noteText, nowMS) + if code, ok := CodeOf(err); !ok || code != CodeConflict { + t.Errorf("a note on an abandoned run: code = %v (%v), want CONFLICT: %v", code, ok, err) + } + if err == nil || !strings.Contains(err.Error(), "abandoned") { + t.Errorf("the refusal does not name the run's status: %v", err) + } + + // Nothing above landed: the refusals left the note table as it was. + notes, err := ListRunNotes(conn, run.ID) + testsupport.Must(t, err, "ListRunNotes: %v", err) + if len(notes) != 1 { + t.Errorf("run_notes holds %d note(s) after the refusals, want the one at-cap note", + len(notes)) + } +} + +// TestRunNoteOnAPlanningRun: the motivating note is written BEFORE the first +// dispatch, and a planning run — not yet activated — must accept it and then +// render it once activated. +func TestRunNoteOnAPlanningRun(t *testing.T) { + conn := mustDB(t) + registerFixture(t, conn) + issue := createIssue(t, conn, "do the thing", "a body", "task", nil) + run := startRun(t, conn, issue) + + note, err := AddRunNote(conn, run.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote on a planning run: %v", err) + + _, err = activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + stepID := stepIDByInstance(t, conn, "implement@0") + result, err := RenderStep(conn, stepID, "", nowMS) + testsupport.Must(t, err, "RenderStep: %v", err) + if !strings.Contains(result.Packet, "== RUN NOTE "+itoa(note.ID)+"\n"+noteText) { + t.Errorf("a note recorded before activation does not render after it:\n%s", + result.Packet) + } +} + +func TestRunNoteIsEventLogged(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + note, err := AddRunNote(conn, run.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + + var runID int + var data string + err = conn.QueryRow( + `SELECT run_id, data FROM events WHERE kind = ?`, EventRunNoteAdded, + ).Scan(&runID, &data) + testsupport.Must(t, err, "reading the run-note-added event: %v", err) + if runID != run.ID { + t.Errorf("event run_id = %d, want %d", runID, run.ID) + } + var payload struct { + Note int `json:"note"` + Text string `json:"text"` + } + testsupport.Must(t, json.Unmarshal([]byte(data), &payload), "event data %q: %v", data, err) + if payload.Note != note.ID || payload.Text != noteText { + t.Errorf("event data = %s, want note %d with the text verbatim", data, note.ID) + } + + actor, ok := ActorFor(EventRunNoteAdded) + if !ok || actor != ActorHuman { + t.Errorf("ActorFor(run-note-added) = %v/%v, want ActorHuman", actor, ok) + } +} + +func TestRunNoteMetaCountsNotes(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + stepID := stepIDByInstance(t, conn, "implement@0") + + without, err := ReadContext(conn, stepID, nowMS) + testsupport.Must(t, err, "ReadContext: %v", err) + _, err = AddRunNote(conn, run.ID, noteText, nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + with, err := ReadContext(conn, stepID, nowMS) + testsupport.Must(t, err, "ReadContext: %v", err) + + before, after := without.Meta(), with.Meta() + if before.NotesBytes != 0 { + t.Errorf("notes_bytes before any note = %d, want 0", before.NotesBytes) + } + if after.NotesBytes != len(noteText) { + t.Errorf("notes_bytes = %d, want %d", after.NotesBytes, len(noteText)) + } + if after.TotalBytes != before.TotalBytes+len(noteText) { + t.Errorf("total_bytes grew by %d, want %d — a note rides every packet and "+ + "the cap must see it", after.TotalBytes-before.TotalBytes, len(noteText)) + } +} + +// TestRunNoteDropsOneTrailingNewline: a file-fed note renders exactly as the +// same words passed inline, and nothing else about the text is touched. +func TestRunNoteDropsOneTrailingNewline(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + + note, err := AddRunNote(conn, run.ID, " keep my leading spaces\nand my blank line\n\n", nowMS) + testsupport.Must(t, err, "AddRunNote: %v", err) + if note.Text != " keep my leading spaces\nand my blank line\n" { + t.Errorf("stored text = %q, want exactly one trailing newline dropped", note.Text) + } +} + +func itoa(n int) string { return strconv.Itoa(n) } diff --git a/internal/engine/packets/default.tmpl b/internal/engine/packets/default.tmpl index 0258417d..937eb052 100644 --- a/internal/engine/packets/default.tmpl +++ b/internal/engine/packets/default.tmpl @@ -18,6 +18,11 @@ target_worktree: {{.TargetWorktree}} == REQUEST {{.Issue.BodySnapshot}} +{{- range .Notes}} + +== RUN NOTE {{.ID}} +{{.Text}} +{{- end}} {{range .Files}} == FILE {{.Path}} {{.SHA256}} {{.Body}} diff --git a/internal/engine/render.go b/internal/engine/render.go index d7518b18..8f673cfe 100644 --- a/internal/engine/render.go +++ b/internal/engine/render.go @@ -29,9 +29,11 @@ import ( // to put words they need a worker to read (DKT-725). A packet is the context // bundle (§6.6's five sources: the pinned step definition, the issue's // ACTIVATION-FROZEN body_snapshot and issue_snapshot, recorded input -// artifacts, and the pin list) plus the step's declared packet files and the -// step's OWN routing record, rendered as `== RESOLUTION`. Two consequences -// operators repeatedly discover the hard way: +// artifacts, and the pin list — plus DKT-1079's sixth, the run's recorded +// notes, rendered as `== RUN NOTE N` beside the request) plus the step's +// declared packet files and the step's OWN routing record, rendered as +// `== RESOLUTION`. Two consequences operators repeatedly discover the hard +// way: // // - issue COMMENTS never render. They are an audit surface, not a context // source, and no template can reach them. @@ -44,10 +46,14 @@ import ( // the issue out of the run and re-planning it, and `issue edit --scope` // says so when it lands on an issue with live steps in a live run. // -// The sanctioned steering channel is the resolve note: `step resolve --as -// retry|rerun-gates -m` renders on the same step's re-execution, and `--as -// fix-round -m` is stamped onto the new round's rows (stampEntryRouting) so -// the authorization's remedy reaches the round it paid for. +// The sanctioned steering channels are two. Per step, the resolve note: `step +// resolve --as retry|rerun-gates -m` renders on the same step's re-execution, +// and `--as fix-round -m` is stamped onto the new round's rows +// (stampEntryRouting) so the authorization's remedy reaches the round it paid +// for. Per run, the run note (DKT-1079): `run note add RUN-N` renders in every +// packet of the run from that moment on, for a fact about the run rather than +// about one step — a gate known to fail on clean HEAD, the issue tracking it, +// the disposition already given — so no worker rediscovers it per step. // defaultPacket is the shipped template. It lives under internal/ because the // Vorpal build's include list requires embeds there — the same constraint the From 4bfcf337af5be7c8d36e4f9702a95995cc6a9496 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:18:22 -0700 Subject: [PATCH 014/397] feat(engine): rank a gap issue from its own header instead of dropping it - a gap file's leading header block (Severity/Priority/Kind/Labels lines right after the title, ending at the first line that isn't one) now sets the filed issue's priority, kind, and labels instead of every gap landing at priority none, kind task, no labels - Severity maps to priority (blocker->critical, high->high, medium->medium, else none); an explicit Priority line wins either way it's ordered against Severity - unrecognized keys are skipped without ending the block, so a Home: line composes with the header; invalid values fall back to the default rather than refusing the whole issue - the body is stored verbatim, header included - documented in step complete/record's help text --- internal/cli/step.go | 18 +++- internal/db/issues.go | 53 +++++++++- internal/engine/gap_test.go | 196 ++++++++++++++++++++++++++++++++++++ internal/engine/saga.go | 119 +++++++++++++++++++++- 4 files changed, 381 insertions(+), 5 deletions(-) diff --git a/internal/cli/step.go b/internal/cli/step.go index 279531e1..b93640ab 100644 --- a/internal/cli/step.go +++ b/internal/cli/step.go @@ -395,7 +395,23 @@ all passed prints the one-line success it always has. auxiliary artifact of kind ` + "`gap`" + ` beside the step's declared emit, plus a backlog issue related to the step's own — same transaction, so the residue cannot evaporate. No workflow declaration is needed; the channel is always -open.`, +open. + +A GAP FILE MAY RANK THE ISSUE IT FILES. Its first line is the issue title; +immediately after it, consecutive ` + "`Key: value`" + ` lines are read as a header +block, ending at the first line that is not one (a blank line included). The +body is stored verbatim either way — the header is read, never stripped. + + Severity: blocker|high|medium|low priority critical|high|medium|none + Priority: critical|high|medium|low|none wins over Severity when both appear + Kind: bug|feature|task|epic|chore defaults to task + Labels: a, b, c comma-separated, at most 16 + +Unrecognized keys are skipped WITHOUT ending the block, so a convention line +like ` + "`Home: THIS repository`" + ` on line two composes with a ` + "`Severity:`" + ` line +on line three. An unrecognized value for a recognized key is ignored rather +than refused. A gap file with no header block files exactly what it always +did: priority none, kind task, no labels.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { w, conn, id, token, err := tokenedStep(cmd, args) diff --git a/internal/db/issues.go b/internal/db/issues.go index b7e08deb..d87dd93c 100644 --- a/internal/db/issues.go +++ b/internal/db/issues.go @@ -1291,6 +1291,22 @@ func ClearProjectDataTx(tx *sql.Tx, projectID int) error { return nil } +// GapIssue is the shape one recorded gap materializes into: the mechanically +// derived title and the verbatim body, plus whatever ranking the gap's own +// header block declared. +// +// Priority and Kind carry the CALLER's decision, not a parse: an empty +// Priority means "the gap declared none" and lands `none`, an empty Kind lands +// `task`. Both defaults are applied here, once, so a caller that declares +// nothing gets exactly the row this insert always wrote (DKT-1082). +type GapIssue struct { + Title string + Description string + Priority model.Priority + Kind model.IssueKind + Labels []string +} + // InsertGapIssueTx materializes a backlog issue from a recorded gap artifact // (DKT-72), inside the completion saga's transaction, and relates it to the // issue whose step recorded the gap. @@ -1301,13 +1317,29 @@ func ClearProjectDataTx(tx *sql.Tx, projectID int) error { // claim with no record behind it. `relates_to` rather than a directional // relation: a gap is out-of-scope BY DEFINITION, so it must not block the // issue that surfaced it. -func InsertGapIssueTx(tx *sql.Tx, projectID int, title, description string, relatedIssueID int) (int, error) { +// +// The ranking rides in the same transaction as the row it ranks (DKT-1082): +// a drained high-severity cluster that landed `priority none` was invisible to +// every priority-ordered planning pass, which is the same "residue nothing +// re-reads" failure one layer up. +func InsertGapIssueTx(tx *sql.Tx, projectID int, gap GapIssue, relatedIssueID int) (int, error) { + projectID = projectOrDefault(projectID) + + priority := gap.Priority + if priority == "" { + priority = model.PriorityNone + } + kind := gap.Kind + if kind == "" { + kind = model.IssueKindTask + } + now := time.Now().UTC().Format(time.RFC3339) res, err := tx.Exec( `INSERT INTO issues (project_id, title, description, status, priority, kind, created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, - projectOrDefault(projectID), title, description, - string(model.StatusBacklog), string(model.PriorityNone), string(model.IssueKindTask), + projectID, gap.Title, gap.Description, + string(model.StatusBacklog), string(priority), string(kind), now, now, ) if err != nil { @@ -1319,6 +1351,21 @@ func InsertGapIssueTx(tx *sql.Tx, projectID int, title, description string, rela } id := int(id64) + // Labels are a per-project namespace, and the gap issue's own project is + // the one they live in — the same rule `issue label add` follows. + for _, name := range gap.Labels { + labelID, err := findOrCreateLabel(tx, projectID, name) + if err != nil { + return 0, fmt.Errorf("labelling gap issue %d: %w", id, err) + } + if _, err := tx.Exec( + `INSERT OR IGNORE INTO issue_labels (issue_id, label_id) VALUES (?, ?)`, + id, labelID, + ); err != nil { + return 0, fmt.Errorf("labelling gap issue %d: %w", id, err) + } + } + if _, err := tx.Exec( `INSERT INTO issue_relations (source_issue_id, target_issue_id, relation_type, created_at) VALUES (?, ?, ?, ?)`, diff --git a/internal/engine/gap_test.go b/internal/engine/gap_test.go index 0972406e..9042b8b2 100644 --- a/internal/engine/gap_test.go +++ b/internal/engine/gap_test.go @@ -99,6 +99,202 @@ func TestGapRecordsArtifactAndIssue(t *testing.T) { } } +// TestGapHeaderRanksTheFiledIssue is DKT-1082: drain-highs filed every +// high-severity cluster it drained at priority `none`, unranked and unlabelled +// — invisible to every priority-ordered planning pass, while the severity was +// already written into the body the step handed over. A gap body's leading +// header block now ranks the issue it materializes, and the body still records +// verbatim. +func TestGapHeaderRanksTheFiledIssue(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + + // The shape drain-highs writes: title, the `Home:` convention line the + // contract already puts on line two, then the severity. + body := "The retry loop swallows the cancel\n" + + "Home: THIS repository\n" + + "Severity: high\n" + + "Labels: review-drain, reliability\n" + + "Kind: bug\n" + + "\n" + + "Cluster SYN-1, round 1.\n" + + var gapIssues []string + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, + Artifact: []byte("the change summary"), + Gaps: [][]byte{[]byte(body)}, + GapIssues: &gapIssues, + NowMS: nowMS, + }) + testsupport.Must(t, err, "complete with a ranked gap: %v", err) + + if len(gapIssues) != 1 { + t.Fatalf("gap issues = %v, want exactly one ref", gapIssues) + } + gapID, err := model.ParseID(gapIssues[0]) + testsupport.Must(t, err, "parsing %s: %v", gapIssues[0], err) + + issue, err := db.GetIssue(conn, gapID) + testsupport.Must(t, err, "reading the gap issue: %v", err) + if issue.Priority != model.PriorityHigh { + t.Errorf("gap issue priority = %q, want %q — a drained high that lands "+ + "unranked is invisible to priority-ordered planning", + issue.Priority, model.PriorityHigh) + } + if issue.Kind != model.IssueKindBug { + t.Errorf("gap issue kind = %q, want %q", issue.Kind, model.IssueKindBug) + } + if issue.Title != "The retry loop swallows the cancel" { + t.Errorf("gap issue title = %q; the header must not displace the title", + issue.Title) + } + // The body is stored VERBATIM: the header is read, never stripped, so the + // artifact and the issue keep saying the same thing. + if issue.Description != body { + t.Errorf("gap issue body = %q, want the gap verbatim %q", + issue.Description, body) + } + + labels, err := db.GetIssueLabels(conn, gapID) + testsupport.Must(t, err, "reading labels: %v", err) + for _, want := range []string{"reliability", "review-drain"} { + if !slices.Contains(labels, want) { + t.Errorf("gap issue labels = %v, missing %q", labels, want) + } + } +} + +// TestGapWithoutHeaderMaterializesAsBefore pins the other half of DKT-1082: +// the header block is OPTIONAL, and a gap that declares nothing files exactly +// the row it always filed. +func TestGapWithoutHeaderMaterializesAsBefore(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + stepID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, stepID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim: %v", err) + + var gapIssues []string + err = e.CompleteStep(conn, stepID, CompleteOptions{ + Token: claim.Token, + Artifact: []byte("the change summary"), + Gaps: [][]byte{[]byte( + "# The helper is duplicated in three packages\n\nSeverity: high\n")}, + GapIssues: &gapIssues, + NowMS: nowMS, + }) + testsupport.Must(t, err, "complete with a headerless gap: %v", err) + + if len(gapIssues) != 1 { + t.Fatalf("gap issues = %v, want exactly one ref", gapIssues) + } + gapID, err := model.ParseID(gapIssues[0]) + testsupport.Must(t, err, "parsing %s: %v", gapIssues[0], err) + + issue, err := db.GetIssue(conn, gapID) + testsupport.Must(t, err, "reading the gap issue: %v", err) + // The blank line after the title ENDS the header block: a `Severity:` line + // in the prose below is body, not a declaration. + if issue.Priority != model.PriorityNone { + t.Errorf("gap issue priority = %q, want %q — a gap that declared no "+ + "header must materialize exactly as before", issue.Priority, model.PriorityNone) + } + if issue.Kind != model.IssueKindTask { + t.Errorf("gap issue kind = %q, want %q", issue.Kind, model.IssueKindTask) + } + labels, err := db.GetIssueLabels(conn, gapID) + testsupport.Must(t, err, "reading labels: %v", err) + if len(labels) != 0 { + t.Errorf("gap issue labels = %v, want none", labels) + } +} + +// TestParseGapHeader covers the mapping table and the block's boundaries +// directly, where the round trip through a completion cannot reach every case. +func TestParseGapHeader(t *testing.T) { + cases := []struct { + name string + body string + priority model.Priority + kind model.IssueKind + labels []string + }{ + { + name: "blocker maps to critical", + body: "title\nSeverity: blocker\n", + priority: model.PriorityCritical, + }, + { + name: "medium maps to medium", + body: "title\nSeverity: MEDIUM\n", + priority: model.PriorityMedium, + }, + { + name: "low sits below every gate and ranks none", + body: "title\nSeverity: low\n", + }, + { + name: "an unknown severity is ignored, not refused", + body: "title\nSeverity: spicy\n", + }, + { + name: "an explicit priority wins over the mapped severity", + body: "title\nSeverity: high\nPriority: low\n", + priority: model.PriorityLow, + }, + { + name: "and wins in either order", + body: "title\nPriority: low\nSeverity: high\n", + priority: model.PriorityLow, + }, + { + name: "prose ends the block", + body: "title\nFound while implementing.\nSeverity: high\n", + }, + { + name: "an unknown key does not end the block", + body: "title\nHome: THIS repository\nSeverity: high\n", + priority: model.PriorityHigh, + }, + { + name: "labels split, trim and dedupe", + body: "title\nLabels: a , b,, a ,c\n", + labels: []string{"a", "b", "c"}, + }, + { + name: "an unknown kind is ignored", + body: "title\nKind: catastrophe\n", + }, + { + name: "a header before the title is body, not a declaration", + body: "Severity: high\n", + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := parseGapHeader([]byte(tc.body)) + if got.Priority != tc.priority { + t.Errorf("priority = %q, want %q", got.Priority, tc.priority) + } + if got.Kind != tc.kind { + t.Errorf("kind = %q, want %q", got.Kind, tc.kind) + } + if !slices.Equal(got.Labels, tc.labels) { + t.Errorf("labels = %v, want %v", got.Labels, tc.labels) + } + }) + } +} + // TestGapOnlyCompletionParksForTheOperator is DKT-25: a completion whose only // product is gap artifacts must not feed the issue's remaining pipeline. The // measured failure: an unimplementable step recorded gap-only, its gates ran diff --git a/internal/engine/saga.go b/internal/engine/saga.go index 693cec8e..a6537a6f 100644 --- a/internal/engine/saga.go +++ b/internal/engine/saga.go @@ -8,6 +8,8 @@ import ( "errors" "fmt" "os/exec" + "regexp" + "slices" "sort" "strings" @@ -452,7 +454,18 @@ func (e *Engine) stageZero(conn *sql.DB, step *db.Step, opts CompleteOptions) er if err != nil { return err } - issueID, err := db.InsertGapIssueTx(tx, gapProject, gapTitle(gap), string(gap), step.IssueID) + // The gap's own leading header block ranks the issue it materializes + // (DKT-1082). The body is stored VERBATIM either way — the header is + // read, never stripped, so the artifact and the issue keep saying the + // same thing. + attrs := parseGapHeader(gap) + issueID, err := db.InsertGapIssueTx(tx, gapProject, db.GapIssue{ + Title: gapTitle(gap), + Description: string(gap), + Priority: attrs.Priority, + Kind: attrs.Kind, + Labels: attrs.Labels, + }, step.IssueID) if err != nil { return err } @@ -532,6 +545,110 @@ func gapTitle(gap []byte) string { return "gap" } +// gapAttrs is what a gap body's leading header block declared about the issue +// it materializes. A zero value means "declared nothing", and the insert +// applies its own defaults — priority `none`, kind `task`, no labels. +type gapAttrs struct { + Priority model.Priority + Kind model.IssueKind + Labels []string +} + +// gapHeaderLine matches one `Key: value` line of a gap's leading header block. +// The key is a SINGLE token — a header block is a block, not prose that +// happens to contain a colon. +var gapHeaderLine = regexp.MustCompile(`^([A-Za-z][A-Za-z0-9_-]*):[ \t]*(.*)$`) + +// severityPriority maps a review severity onto the backlog priority a drained +// cluster must land at (DKT-1082). Below `medium` a finding sits under every +// gate by design, so it ranks `none` and the map simply has no entry. +var severityPriority = map[string]model.Priority{ + "blocker": model.PriorityCritical, + "high": model.PriorityHigh, + "medium": model.PriorityMedium, +} + +// parseGapHeader reads the OPTIONAL header block a gap body may carry +// immediately after its title line: consecutive `Key: value` lines, ending at +// the first line that is not one (a blank line included). +// +// It exists because drain-highs filed every high-severity cluster it drained +// at priority `none`, where no priority-ordered planning pass would ever look +// (DKT-1082) — while the severity was already written into the body the step +// handed over. Reading it is MECHANICAL: a key, a value, and a closed set of +// recognized keys. Docket still never interprets what the gap SAYS; it reads +// only what the gap DECLARES about itself, in a shape the writer chose. +// +// Unrecognized keys are skipped without ending the block, so the `Home:` line +// the drain-highs contract already puts on line two composes with a +// `Severity:` line on line three. An unrecognized VALUE for a recognized key +// is ignored the same way — a typo'd severity must not cost a worker its +// completion, and the insert's default is the honest answer for "nothing +// legible was declared". +// +// The body is never modified: callers store it verbatim. +func parseGapHeader(gap []byte) gapAttrs { + var attrs gapAttrs + var sawTitle bool + var severity model.Priority + + for _, line := range strings.Split(string(gap), "\n") { + line = strings.TrimRight(line, "\r") + if !sawTitle { + // The title is the first non-empty line — the same line gapTitle + // takes — and the header block starts immediately after it. + if strings.TrimSpace(line) == "" { + continue + } + sawTitle = true + continue + } + match := gapHeaderLine.FindStringSubmatch(line) + if match == nil { + break + } + key := strings.ToLower(match[1]) + value := strings.TrimSpace(match[2]) + switch key { + case "severity": + severity = severityPriority[strings.ToLower(value)] + case "priority": + if p := model.Priority(strings.ToLower(value)); model.ValidatePriority(p) == nil { + attrs.Priority = p + } + case "kind": + if k := model.IssueKind(strings.ToLower(value)); model.ValidateIssueKind(k) == nil { + attrs.Kind = k + } + case "labels": + attrs.Labels = parseGapLabels(value, attrs.Labels) + } + } + + // An explicit `Priority:` wins over a mapped `Severity:`: the specific + // statement beats the derived one, whichever order the two lines appear in. + if attrs.Priority == "" { + attrs.Priority = severity + } + return attrs +} + +// parseGapLabels appends the comma-separated label names in value to seen, +// trimming each, dropping empties, and de-duplicating case-sensitively (label +// names are a per-project namespace and `bug` is not `Bug`). Capped, because a +// header line is a declaration and not a bulk import. +func parseGapLabels(value string, seen []string) []string { + const maxGapLabels = 16 + for _, name := range strings.Split(value, ",") { + name = strings.TrimSpace(name) + if name == "" || slices.Contains(seen, name) || len(seen) >= maxGapLabels { + continue + } + seen = append(seen, name) + } + return seen +} + // gapOnlyCompletion reports whether a step's recorded product is gap // artifacts alone: at least one gap recorded while the declared emit's body is // empty (DKT-25). From fd82446f0f2baee4f80a0a0d3252941063d1d5bc Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:40:18 -0700 Subject: [PATCH 015/397] feat(engine): skip a step's successors when it never fires A gate step that ends up skipped (routing resolved elsewhere, an on_fail=skip rejection, a false `when`, or a quorum miss) used to leave its downstream `after` successors ready to run anyway, since a skipped predecessor still counts as terminal. Successors had no way to say "only run me if that predecessor actually fired." Add a step-level `after_fired` list: naming a predecessor there means this step is skipped in the same transaction the moment that predecessor is skipped, cascading transitively through the graph. Every `after_fired` entry must also appear in `after`, so the step still waits for the predecessor to reach a terminal state before the skip (or the run) is decided. --- docs/design/engine-spec.md | 1 + internal/cli/workflow_show.go | 3 + internal/db/steps.go | 9 + internal/engine/activate.go | 10 + internal/engine/after_fired.go | 169 ++++++++++ internal/engine/dkt1085_test.go | 392 ++++++++++++++++++++++++ internal/engine/human.go | 12 +- internal/engine/loop.go | 14 + internal/engine/mirror_coverage_test.go | 2 +- internal/engine/ready.go | 24 +- internal/engine/reconcile.go | 16 +- internal/engine/saga.go | 4 +- internal/engine/saga_resume.go | 13 +- internal/engine/vote.go | 2 +- internal/workflow/parse.go | 21 ++ internal/workflow/validate.go | 39 ++- internal/workflow/validate_test.go | 79 +++++ 17 files changed, 791 insertions(+), 19 deletions(-) create mode 100644 internal/engine/after_fired.go create mode 100644 internal/engine/dkt1085_test.go diff --git a/docs/design/engine-spec.md b/docs/design/engine-spec.md index f5541697..301647dd 100644 --- a/docs/design/engine-spec.md +++ b/docs/design/engine-spec.md @@ -480,6 +480,7 @@ is exactly `"write" = { max = 1, lease_ttl = "45m", max_step_duration = "2h" }`. | `payload` | `schema@ver`, optional | payload validated at `complete`; threshold fields check against it at register time; required on `action = "aggregate"` steps *(amended 2026-08-03, DKT-25)* | | `voters`, `vote_rule` | [executor hints], proposal-config name | required on `type="vote"` steps — who casts, which existing Docket threshold config tallies | | `after` | [step names], **required** except the first step and `loop = true` steps (whose ordering comes from loop entry, §11.3) | intra-workflow predecessors; `[]` = root (implicit topology was a footgun) | +| `after_fired` | [step names], optional; every entry must also appear in `after` | predecessors this step runs ONLY IF THEY FIRED: when every instance of a named step ends `skipped` — an interposed gate its threshold routed elsewhere (§11.2), a false `when`, an `on_fail = "skip"` routing, an operator's `--as skip` — this step is terminalized `skipped` in the same transaction, and the skip cascades through every step declaring `after_fired` on it in turn. Additive beside `after`, whose meaning is unchanged: `skipped` still releases an `after` join (J1), and a skipped `after_fired` step contributes no input (J3). The corpus case is a `drain-highs` executor that runs only on rounds `security-vote` actually decided *(added 2026-09-02, DKT-1085)* | | `inputs` | [`"."` \| `".*"` \| `".vote-record"` \| `"issue.body"` \| `"issue.diff"` \| `"issue.linked.."`] | artifacts inlined into the context bundle, in order. `issue.diff` = the engine-computed VCS diff for the issue's scope, snapshotted and fingerprinted when its producing step completed (git in v1 — the one declared VCS coupling, §7). `.vote-record` = the named `type="vote"` step's recorded proposal — tally outcome, weighted score, and every cast with its rationale — engine-served from the existing vote machinery; the named step must be a vote step, and the `vote-record` kind is reserved from `emits` *(amended 2026-08-22, DKT-545)*. `issue.linked..` = a CROSS-ISSUE input: the latest recorded artifact of `` held by each issue this issue is linked to by `` (a relation type or its inverse form — `depends_on`, `dependency_of`, `blocks`, `blocked_by`, `relates_to`, `duplicates`, `duplicate_of`), resolved and pinned by artifact id at activation inside the fat transaction; activation fails loudly when the relation is missing or no linked issue holds the kind, so the binding is enforced rather than an issue-body citation. V11's produced-kind table deliberately does not apply — the producer is another issue's run — and the `issue.linked` name is reserved from step names as `issue.latest` is *(amended 2026-08-22, DKT-547)* | | `gates` | [trusted gate names \| `{name, source="fence:", pre=bool}`] | `pre = true` gates run at claim with results included in the context bundle (measure-then-judge steps); the rest run in order inside `complete` (§2, §4) | | `params` | opaque KV table | arguments to `action` steps (e.g. the builtin `aggregate`) | diff --git a/internal/cli/workflow_show.go b/internal/cli/workflow_show.go index a662634d..d295204c 100644 --- a/internal/cli/workflow_show.go +++ b/internal/cli/workflow_show.go @@ -134,6 +134,9 @@ func renderWorkflowShow(wf *model.Workflow) string { if len(step.After) > 0 { fmt.Fprintf(&b, " after=[%s]", strings.Join(step.After, ", ")) } + if len(step.AfterFired) > 0 { + fmt.Fprintf(&b, " after_fired=[%s]", strings.Join(step.AfterFired, ", ")) + } if step.Loop { b.WriteString(" loop") } diff --git a/internal/db/steps.go b/internal/db/steps.go index e6d68650..78f9aae8 100644 --- a/internal/db/steps.go +++ b/internal/db/steps.go @@ -259,6 +259,15 @@ func ListRunStepsTx(tx *sql.Tx, runID int) ([]*Step, error) { return scanSteps(tx.Query(stepFullSelect+` WHERE run_id = ? ORDER BY issue_id, id`, runID)) } +// ListIssueStepsTx reads every step of ONE issue in a run, inside a +// transaction — the `after_fired` cascade's reader (DKT-1085), which must see +// the rows the open routing transaction just terminalized and needs no other +// issue's steps to answer its question. +func ListIssueStepsTx(tx *sql.Tx, runID, issueID int) ([]*Step, error) { + return scanSteps(tx.Query( + stepFullSelect+` WHERE run_id = ? AND issue_id = ? ORDER BY id`, runID, issueID)) +} + // ListActiveRunSteps reads the steps of every non-terminal run — `guard stop`'s // reader (§6.12) and the scope-conflict check's, since a claimed step in ANOTHER // active run excludes just as surely as one in this run. diff --git a/internal/engine/activate.go b/internal/engine/activate.go index 22ccab59..d2032997 100644 --- a/internal/engine/activate.go +++ b/internal/engine/activate.go @@ -1677,6 +1677,16 @@ func expandIssue( } } + // A step whose `after_fired` predecessor was just CREATED `skipped` by its + // `when` is created skipped with it (DKT-1085), in the same fat + // transaction: expansion is the one skip no routing transaction follows, + // so the cascade every routing runs (reconcileIssueAndRun) runs here too. + if _, err := propagateAfterFiredSkips( + tx, run.ID, issue.ID, bound.definition, nowMS, + ); err != nil { + return 0, nil, err + } + return len(rows), warnings, nil } diff --git a/internal/engine/after_fired.go b/internal/engine/after_fired.go new file mode 100644 index 00000000..5c9ee8dc --- /dev/null +++ b/internal/engine/after_fired.go @@ -0,0 +1,169 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// `after_fired` (DKT-1085): a step that runs ONLY IF a named predecessor fired. +// +// The gap it closes. An interposed gate — a step some `threshold` routes to by +// name (§11.2) — ends `skipped` on every round its routing step resolved +// elsewhere (skipUnroutedTargets), and `skipped` is terminal, so an ordinary +// `after = ["security-vote"]` successor is released by the skip exactly as by +// an approval (J1). `when` reads issue kind and labels only (V22) and cannot +// say "the gate fired"; a second step-name routing beside the vote on the same +// threshold would be evaluated first (ThresholdOrder sorts step targets) and +// skip the vote itself. The corpus's `drain-highs` therefore ran on every +// skipped-vote round: 3 of 81 measured, one sonnet spawn each for an empty +// report. +// +// The rule. A step declaring `after_fired = [g, ...]` is terminalized +// `skipped` when any named predecessor DID NOT FIRE — every instance of it, at +// the ordinal R3 resolves (predecessorInstancesIn), ended `skipped` — and it +// is skipped IN THE SAME TRANSACTION as the predecessor's own skip. The +// propagation is transitive by construction: the sweep runs to a fixed point, +// so a step declaring `after_fired` on a step this pass skipped is skipped by +// the next pass, in the same transaction, however deep the chain. +// +// WHY the predecessor's whole instance set. For a single-instance predecessor +// the two readings coincide. For a fanout, "the step fired" means some +// sibling ran: a predecessor with one `done` sibling produced a result its +// `after_fired` successor has something to run over; only a predecessor NONE +// of whose siblings ran has not fired. And `skipped` LITERALLY: a predecessor +// that ended `failed-routed` or `superseded` was decided, not skipped, and the +// issue's own acceptance names `skipped` alone. +// +// WHERE it runs — every place the engine can end a step `skipped`: +// +// - reconcileIssueAndRun, the tail of EVERY routing transaction: after +// skipUnroutedTargets (the interposed-gate skip) and before the completion +// check, so an issue whose last live steps were just skipped completes in +// the same transaction. This also covers an `on_fail = "skip"` routing and +// an operator's `step resolve --as skip`, both of which reconcile; +// - resolveQuorumMisses (loop.go), the one routing path that terminalizes a +// threshold-bearing step without reconciling; +// - expandIssue (activate.go), where a false `when` CREATES a step skipped +// (§5.3.1) and its `after_fired` successors must be created skipped with +// it. Loop re-instantiation needs no fourth call: every loop entry commits +// inside a routing transaction that ends in one of the first two. +// +// WHAT it does not do. It never touches a step that has left `pending` — a +// claimed, running, gated, parked, or terminal step has already answered the +// question — and it consults no readiness clause: V39a requires every +// `after_fired` entry to also appear in `after`, so R3 already holds the step +// behind its gate and the sweep only ever decides a step whose predecessor is +// terminal. `after` itself is untouched (done OR skipped still releases a +// join); this is an additive second list, not a new reading of the first. +// +// DORMANT UNLESS DECLARED. A definition with no `after_fired` anywhere costs +// no query at all — the same discipline the budget and reap-hold snapshots +// keep (§4.8 B29) — so a routing on a workflow that never heard of the key is +// byte-identical to what it was. + +// propagateAfterFiredSkips terminalizes `skipped`, to a fixed point, every +// `pending` step of one issue whose `after_fired` names a predecessor that did +// not fire. It returns the rows it skipped, so a caller holding a loaded +// snapshot (resolveQuorumMisses) can mirror the change into it. +func propagateAfterFiredSkips( + tx *sql.Tx, runID, issueID int, def *workflow.Definition, nowMS int64, +) ([]*db.Step, error) { + if !declaresAfterFired(def) { + return nil, nil + } + steps, err := db.ListIssueStepsTx(tx, runID, issueID) + if err != nil { + return nil, err + } + + var skipped []*db.Step + // Each pass either skips at least one more row or stops, so the bound is + // the step count — the same argument downstreamClosure (loop.go) makes. + for { + changed := false + for _, step := range steps { + if step.Status != db.StepPending || step.Materialized { + continue + } + spec := workflow.StepByName(def, step.StepName) + if spec == nil || len(spec.AfterFired) == 0 { + continue + } + unfired := unfiredPredecessor(steps, step, spec.AfterFired) + if unfired == "" { + continue + } + if err := db.SetStepStatusTx(tx, step.ID, db.StepSkipped, nowMS, nowMS); err != nil { + return nil, err + } + if err := recordEvent(tx, eventRecord{ + Kind: EventStepSkipped, RunID: runID, + Instance: step.Instance, IssueID: issueID, + Data: fmt.Sprintf("after_fired predecessor did not fire: %s", unfired), + AtMS: nowMS, + }); err != nil { + return nil, err + } + // Reflected in the loaded rows so the next pass — and the caller's + // snapshot, which shares nothing with these rows but is repaired + // from the returned set — sees the skip this pass just wrote. + step.Status = db.StepSkipped + skipped = append(skipped, step) + changed = true + } + if !changed { + return skipped, nil + } + } +} + +// unfiredPredecessor names the first `after_fired` predecessor of a step that +// did not fire — every instance R3 resolves for it ended `skipped` — rendered +// as its instances for the event, or "" when each named predecessor fired, is +// still open, or has no instance yet. +// +// A predecessor with no instance at or below the step's ordinal has not been +// expanded (a loop body not yet entered): nothing is decided, so nothing is +// skipped. An instance in any non-`skipped` status — including a pending or +// running one — means the predecessor fired or may still fire, and the step +// waits exactly as R3 makes it wait. +func unfiredPredecessor(steps []*db.Step, step *db.Step, names []string) string { + for _, name := range names { + instances := predecessorInstancesIn(steps, step, name) + if len(instances) == 0 { + continue + } + fired := false + rendered := make([]string, 0, len(instances)) + for _, inst := range instances { + if inst.Status != db.StepSkipped { + fired = true + break + } + rendered = append(rendered, inst.Instance) + } + if !fired { + return strings.Join(rendered, ", ") + } + } + return "" +} + +// declaresAfterFired reports whether any step of a definition declares +// `after_fired` — the sweep's dormancy test. A nil definition (a caller with +// none, such as the interrupted-gate park) declares nothing. +func declaresAfterFired(def *workflow.Definition) bool { + if def == nil { + return false + } + for _, step := range def.Steps { + if len(step.AfterFired) > 0 { + return true + } + } + return false +} diff --git a/internal/engine/dkt1085_test.go b/internal/engine/dkt1085_test.go new file mode 100644 index 00000000..eb78ef60 --- /dev/null +++ b/internal/engine/dkt1085_test.go @@ -0,0 +1,392 @@ +package engine + +import ( + "database/sql" + "fmt" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" +) + +// DKT-1085: `after_fired` — "run me only if the gate fired". +// +// An interposed gate ends `skipped` on every round its routing step resolved +// elsewhere, `skipped` is terminal, and R3 releases an `after` successor on +// ANY terminal predecessor — so `after = ["security-vote"]` ran the corpus's +// `drain-highs` on skipped-vote rounds too (3 of 81 measured, one spawn each +// for an empty report). `after_fired = ["security-vote"]` is the declaration +// the engine lacked: the successor is skipped in the same transaction as the +// gate, transitively, while `after` keeps its exact meaning. + +// afterFiredSrc is the corpus shape: an interposed vote behind a routing +// step, a `drain-highs` that runs only if the vote fired, a `summarize` +// behind THAT (the transitive case), a plain-`after` `verify` that runs on +// every round, and a `finish` whose inputs name the conditional chain (J3). +const afterFiredSrc = ` +[pipeline] +name = "after-fired" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "reconcile" +executor = "reconcile" +emits = "report" +threshold = { "security-vote" = "any(status == blocked)" } + +[[step]] +name = "security-vote" +after = ["reconcile"] +type = "human" +on_fail = "skip" + +[[step]] +name = "drain-highs" +after = ["security-vote"] +after_fired = ["security-vote"] +executor = "drain" +emits = "drain-report" + +[[step]] +name = "summarize" +after = ["drain-highs"] +after_fired = ["drain-highs"] +executor = "summarize" +emits = "summary" + +[[step]] +name = "verify" +after = ["security-vote"] +executor = "verify" +emits = "verification" + +[[step]] +name = "finish" +after = ["verify", "summarize"] +executor = "finish" +emits = "record" +inputs = ["summarize.summary", "verify.verification"] +` + +// skipEventData reads the `step-skipped` event recorded against one instance. +func skipEventData(t *testing.T, conn *sql.DB, runID int, instance string) string { + t.Helper() + var data string + err := conn.QueryRow( + `SELECT e.data FROM events e JOIN steps s ON s.id = e.step_id + WHERE e.run_id = ? AND e.kind = 'step-skipped' AND s.instance = ? + ORDER BY e.seq DESC LIMIT 1`, runID, instance).Scan(&data) + testsupport.Must(t, err, "reading the step-skipped event for %s: %v", instance, err) + return data +} + +// TestAfterFiredSuccessorIsSkippedWithItsGate is the issue's exact scenario, +// fixed: the routing resolves `pass`, the gate is skipped, and the ONE +// transaction that skipped it also skips its `after_fired` successor and, +// transitively, the successor's own — while the plain-`after` successor is +// untouched and the issue still completes over the skipped chain. +func TestAfterFiredSuccessorIsSkippedWithItsGate(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, afterFiredSrc) + e := testEngine() + + // Nothing is decided at activation: no `when`, so every row is pending. + for _, inst := range []string{"security-vote@0", "drain-highs@0", "summarize@0", "verify@0"} { + if got := stepStatus(t, conn, inst); got != db.StepPending { + t.Fatalf("%s = %q at activation, want pending", inst, got) + } + } + + // No `next`, no lifecycle drive in between: the statuses below are what + // the routing transaction itself committed. + claimAndComplete(t, conn, e, "reconcile@0", "all clear", `[{"status":"ok"}]`) + + if got := stepStatus(t, conn, "security-vote@0"); got != db.StepSkipped { + t.Fatalf("security-vote@0 = %q after a pass routing, want %q", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "drain-highs@0"); got != db.StepSkipped { + t.Errorf("drain-highs@0 = %q after its gate was skipped, want %q — an "+ + "after_fired successor must be skipped in the same transaction as "+ + "its gate, not left pending or run unconditionally", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "summarize@0"); got != db.StepSkipped { + t.Errorf("summarize@0 = %q, want %q — the skip must cascade through a "+ + "step whose after_fired predecessor was itself skipped by propagation", + got, db.StepSkipped) + } + + // `after` keeps its meaning: the plain successor is pending and, over the + // skipped gate, ready (J1: skipped is terminal). + if got := stepStatus(t, conn, "verify@0"); got != db.StepPending { + t.Errorf("verify@0 = %q, want pending — an after-only successor is "+ + "unaffected by its gate's skip", got) + } + loadScheduler(t, conn, runID, nowMS, func(sched *Scheduler) { + if ok, cond := sched.Ready(stepNamed(t, sched, "verify@0")); !ok { + t.Errorf("verify@0 not ready over the skipped gate: %q", cond) + } + }) + + // Each cascade skip is event-logged naming the predecessor that did not + // fire, so the ledger explains the row. + if data := skipEventData(t, conn, runID, "drain-highs@0"); !strings.Contains(data, "security-vote@0") { + t.Errorf("drain-highs@0 step-skipped data = %q, want it to name security-vote@0", data) + } + if data := skipEventData(t, conn, runID, "summarize@0"); !strings.Contains(data, "drain-highs@0") { + t.Errorf("summarize@0 step-skipped data = %q, want it to name drain-highs@0", data) + } + + // J3: `finish` binds no input from the skipped chain, and J1 releases it + // over the skipped `summarize`; the issue and run complete. + claimAndComplete(t, conn, e, "verify@0", "verified", `[]`) + loadScheduler(t, conn, runID, nowMS, func(sched *Scheduler) { + if ok, cond := sched.Ready(stepNamed(t, sched, "finish@0")); !ok { + t.Fatalf("finish@0 not ready behind a skipped after_fired chain: %q", cond) + } + }) + for _, in := range contextInputs(t, conn, runID, "finish@0") { + if in.ProducerStep == "summarize@0" { + t.Error("finish@0 bound an input from a skipped producer; J3 resolves " + + "over recorded producers only") + } + } + claimAndComplete(t, conn, e, "finish@0", "the record", `[]`) + + if got := issueStatusOf(t, conn, issue); got != "done" { + t.Errorf("issue = %q after every live step finished, want done", got) + } + if got := runStatusOf(t, conn, runID); got != "done" { + t.Errorf("run = %q, want done", got) + } +} + +// TestAfterFiredSuccessorRunsWhenTheGateFired is the other half: when the +// threshold routes TO the gate and the gate approves, the chain runs exactly +// as an ordinary `after` chain would — nothing is skipped. +func TestAfterFiredSuccessorRunsWhenTheGateFired(t *testing.T) { + conn := mustDB(t) + runID, issue := activateInterposed(t, conn, afterFiredSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "reconcile@0", "blocked finding", `[{"status":"blocked"}]`) + + if got := stepRouting(t, conn, "reconcile@0"); got != "security-vote" { + t.Fatalf("reconcile@0 routing = %q, want the interposed step name", got) + } + for _, inst := range []string{"security-vote@0", "drain-highs@0", "summarize@0"} { + if got := stepStatus(t, conn, inst); got != db.StepPending { + t.Fatalf("%s = %q while the gate is open, want pending", inst, got) + } + } + // Ordered behind the gate by its `after` edge, as V39a guarantees. + loadScheduler(t, conn, runID, nowMS, func(sched *Scheduler) { + ok, cond := sched.Ready(stepNamed(t, sched, "drain-highs@0")) + if ok || cond != CondPredecessors { + t.Fatalf("drain-highs@0 with its gate open: ready=%v cond=%q, want a "+ + "CondPredecessors wait", ok, cond) + } + }) + + err := e.DecideStep(conn, stepIDByInstance(t, conn, "security-vote@0"), true, "lgtm", nowMS) + testsupport.Must(t, err, "approving the gate: %v", err) + + if got := stepStatus(t, conn, "drain-highs@0"); got != db.StepPending { + t.Fatalf("drain-highs@0 = %q after its gate fired, want pending", got) + } + loadScheduler(t, conn, runID, nowMS, func(sched *Scheduler) { + if ok, cond := sched.Ready(stepNamed(t, sched, "drain-highs@0")); !ok { + t.Fatalf("drain-highs@0 not ready after its gate approved: %q", cond) + } + }) + + claimAndComplete(t, conn, e, "drain-highs@0", "drained", `[]`) + claimAndComplete(t, conn, e, "summarize@0", "the summary", `[]`) + claimAndComplete(t, conn, e, "verify@0", "verified", `[]`) + claimAndComplete(t, conn, e, "finish@0", "the record", `[]`) + + if got := issueStatusOf(t, conn, issue); got != "done" { + t.Errorf("issue = %q, want done", got) + } +} + +// TestAfterFiredFollowsAnOnFailSkip: the gate fired but was REJECTED, and its +// `on_fail = "skip"` ends it `skipped` — "skipped for any other existing +// reason". The cascade follows that skip too, in the rejection's own +// transaction. +func TestAfterFiredFollowsAnOnFailSkip(t *testing.T) { + conn := mustDB(t) + runID, _ := activateInterposed(t, conn, afterFiredSrc) + e := testEngine() + + claimAndComplete(t, conn, e, "reconcile@0", "blocked finding", `[{"status":"blocked"}]`) + err := e.DecideStep(conn, stepIDByInstance(t, conn, "security-vote@0"), false, "no", nowMS) + testsupport.Must(t, err, "rejecting the gate: %v", err) + + if got := stepStatus(t, conn, "security-vote@0"); got != db.StepSkipped { + t.Fatalf("security-vote@0 = %q after a rejection routed skip, want %q", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "drain-highs@0"); got != db.StepSkipped { + t.Errorf("drain-highs@0 = %q after its gate ended skipped, want %q", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "summarize@0"); got != db.StepSkipped { + t.Errorf("summarize@0 = %q, want %q (transitive)", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "verify@0"); got != db.StepPending { + t.Errorf("verify@0 = %q, want pending — after-only, unaffected", got) + } + loadScheduler(t, conn, runID, nowMS, func(sched *Scheduler) { + if ok, cond := sched.Ready(stepNamed(t, sched, "verify@0")); !ok { + t.Errorf("verify@0 not ready over the skipped gate: %q", cond) + } + }) +} + +// afterFiredWhenSrc skips the gate at EXPANSION: its `when` is false for an +// issue without the label, so activation creates it `skipped` (§5.3.1) — and +// must create its `after_fired` successor skipped with it, in the fat +// transaction, before any step has run. +const afterFiredWhenSrc = ` +[pipeline] +name = "after-fired-when" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "verify" +executor = "verify" +emits = "report" + +[[step]] +name = "vote" +after = ["verify"] +type = "human" +on_fail = "skip" +when = "labels contains needs-vote" + +[[step]] +name = "drain" +after = ["vote"] +after_fired = ["vote"] +executor = "drain" +emits = "drain-report" + +[[step]] +name = "finish" +after = ["vote"] +executor = "finish" +emits = "record" +` + +func TestAfterFiredFollowsAWhenSkipAtExpansion(t *testing.T) { + conn := mustDB(t) + registerSource(t, conn, []byte(afterFiredWhenSrc), "after-fired-when.toml") + + // Without the label the gate's `when` is false: gate and successor are + // created skipped, the after-only successor pending. + without := createIssue(t, conn, "without", "a body", "task", nil) + run := startRun(t, conn, without) + _, err := activate(conn, run.ID) + testsupport.Must(t, err, "activate: %v", err) + + if got := stepStatus(t, conn, "vote@0"); got != db.StepSkipped { + t.Fatalf("vote@0 = %q at activation with its when false, want %q", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "drain@0"); got != db.StepSkipped { + t.Errorf("drain@0 = %q at activation, want %q — a when-skipped gate's "+ + "after_fired successor is created skipped with it", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "finish@0"); got != db.StepPending { + t.Errorf("finish@0 = %q at activation, want pending", got) + } + if data := skipEventData(t, conn, run.ID, "drain@0"); !strings.Contains(data, "vote@0") { + t.Errorf("drain@0 step-skipped data = %q, want it to name vote@0", data) + } + + // With the label the gate is live, and so is its successor. + conn2 := mustDB(t) + registerSource(t, conn2, []byte(afterFiredWhenSrc), "after-fired-when.toml") + with := createIssue(t, conn2, "with", "a body", "task", []string{"needs-vote"}) + run2 := startRun(t, conn2, with) + _, err = activate(conn2, run2.ID) + testsupport.Must(t, err, "activate: %v", err) + + for _, inst := range []string{"vote@0", "drain@0", "finish@0"} { + if got := stepStatus(t, conn2, inst); got != db.StepPending { + t.Errorf("%s = %q at activation with the label, want pending", inst, got) + } + } +} + +// afterFiredQuorumSrc puts the gate behind a fanned routing step with a real +// quorum, so the skip arrives by the ONE routing path that terminalizes a +// step without reconciling — resolveQuorumMisses, inside `next`. +const afterFiredQuorumSrc = ` +[pipeline] +name = "after-fired-quorum" +version = 1 + +[match] +kind = ["task"] + +[[step]] +name = "spread" +after = [] +fanout = ["a", "b", "c", "d"] +min_siblings = 2 +emits = "findings" +on_fail = "skip" +threshold = { "gate" = "any(status == blocked)" } + +[[step]] +name = "gate" +after = ["spread"] +type = "human" +on_fail = "skip" + +[[step]] +name = "drain" +after = ["gate"] +after_fired = ["gate"] +executor = "drain" +emits = "drain-report" +` + +func TestAfterFiredCascadesOnTheQuorumMissPath(t *testing.T) { + conn := mustDB(t) + runID, _ := activateInterposed(t, conn, afterFiredQuorumSrc) + e := testEngine() + + // One `done` sibling routing `pass`, three that ended without a result: + // the join completes below the quorum, and the skip of the gate — and the + // cascade behind it — happens inside `next`'s resolution. + claimAndComplete(t, conn, e, "spread@0#0", "clear", `[{"status":"ok"}]`) + if got := stepStatus(t, conn, "gate@0"); got != db.StepPending { + t.Fatalf("gate@0 = %q with siblings still open, want pending", got) + } + for i := 1; i < 4; i++ { + execSQL(t, conn, `UPDATE steps SET status = ? WHERE instance = ?`, + db.StepFailedRouted, fmt.Sprintf("spread@0#%d", i)) + } + + answer, err := e.NextSteps(conn, runID, 0, nowMS) + testsupport.Must(t, err, "NextSteps: %v", err) + + if got := stepStatus(t, conn, "gate@0"); got != db.StepSkipped { + t.Fatalf("gate@0 = %q after the missed quorum routed, want %q", got, db.StepSkipped) + } + if got := stepStatus(t, conn, "drain@0"); got != db.StepSkipped { + t.Errorf("drain@0 = %q, want %q — the cascade must run on the quorum-miss "+ + "path, which reconciles nothing", got, db.StepSkipped) + } + for _, row := range answer.Steps { + if row.Instance == "drain@0" { + t.Errorf("drain@0 offered by the same `next` that skipped it: %v", instancesIn(answer)) + } + } +} diff --git a/internal/engine/human.go b/internal/engine/human.go index 8c1cbbab..1ee7e885 100644 --- a/internal/engine/human.go +++ b/internal/engine/human.go @@ -274,7 +274,9 @@ func (e *Engine) DecideStepValue( }); err != nil { return err } - if err := reconcileIssueAndRun(tx, step, spec, routing, nowMS); err != nil { + if err := reconcileIssueAndRun( + tx, step, defs[step.WorkflowID], spec, routing, nowMS, + ); err != nil { return err } return tx.Commit() @@ -815,7 +817,9 @@ func (e *Engine) resolveStep( return err } } - if err := reconcileIssueAndRun(tx, step, spec, routing, nowMS); err != nil { + if err := reconcileIssueAndRun( + tx, step, defs[step.WorkflowID], spec, routing, nowMS, + ); err != nil { return err } if err := tx.Commit(); err != nil { @@ -1080,7 +1084,9 @@ func (e *Engine) FailStep(conn *sql.DB, stepID int, token, note, metadata string failureNote(step.Instance, note, attempt, max), nowMS); err != nil { return err } - if err := reconcileIssueAndRun(tx, step, spec, routing, nowMS); err != nil { + if err := reconcileIssueAndRun( + tx, step, defs[step.WorkflowID], spec, routing, nowMS, + ); err != nil { return err } return tx.Commit() diff --git a/internal/engine/loop.go b/internal/engine/loop.go index e36c180d..475a9413 100644 --- a/internal/engine/loop.go +++ b/internal/engine/loop.go @@ -1384,6 +1384,20 @@ func resolveQuorumMisses(tx *sql.Tx, sched *Scheduler, nowMS int64) error { if err := skipUnroutedTargets(tx, step, spec, routing, nowMS); err != nil { return err } + // ...and the `after_fired` cascade behind those targets, and behind a + // `skip` routing this miss just recorded (DKT-1085), for the same + // reason: nothing else on this path would run it. Mirrored into the + // snapshot exactly as the routed step is below, so this call's + // readiness pass cannot offer a row the transaction just skipped. + cascade, err := propagateAfterFiredSkips(tx, step.RunID, step.IssueID, def, nowMS) + if err != nil { + return err + } + for _, row := range cascade { + if loaded := sched.stepByID[row.ID]; loaded != nil { + loaded.Status = db.StepSkipped + } + } // Reflect it in the loaded snapshot, so this call's readiness pass sees // the routed step rather than the row it read a moment ago — the same diff --git a/internal/engine/mirror_coverage_test.go b/internal/engine/mirror_coverage_test.go index 6ba94caf..3bcc24db 100644 --- a/internal/engine/mirror_coverage_test.go +++ b/internal/engine/mirror_coverage_test.go @@ -164,7 +164,7 @@ on_fail = "waiting-human" tx, err := conn.Begin() testsupport.Must(t, err, "Begin: %v", err) - testsupport.Must(t, reconcileIssueAndRun(tx, step, nil, workflow.OnFailWaitingHuman, nowMS), + testsupport.Must(t, reconcileIssueAndRun(tx, step, nil, nil, workflow.OnFailWaitingHuman, nowMS), "reconcile: %v", err) testsupport.Must(t, tx.Commit(), "Commit") diff --git a/internal/engine/ready.go b/internal/engine/ready.go index 5bfb2f58..041643d3 100644 --- a/internal/engine/ready.go +++ b/internal/engine/ready.go @@ -715,12 +715,22 @@ func (s *Scheduler) predecessorsDone(step *db.Step) bool { // has instances, and the fallback applies only where re-instantiation did not // reach. func (s *Scheduler) predecessorInstances(step *db.Step, predName string) []*db.Step { - if at := s.instancesOf(step.IssueID, predName, step.Ordinal); len(at) > 0 { + return predecessorInstancesIn(s.steps, step, predName) +} + +// predecessorInstancesIn is predecessorInstances over an explicit step set, so +// the `after_fired` cascade (after_fired.go) — which runs inside a routing +// transaction with no Scheduler loaded — resolves a predecessor by the SAME +// ordinal rule R3 uses. Two readings of "which instances of g does S wait on" +// would disagree at the first loop that re-instantiated one of them and not +// the other. +func predecessorInstancesIn(steps []*db.Step, step *db.Step, predName string) []*db.Step { + if at := instancesAt(steps, step.IssueID, predName, step.Ordinal); len(at) > 0 { return at } best := -1 - for _, other := range s.steps { + for _, other := range steps { if other.IssueID != step.IssueID || other.StepName != predName { continue } @@ -731,7 +741,7 @@ func (s *Scheduler) predecessorInstances(step *db.Step, predName string) []*db.S if best < 0 { return nil } - return s.instancesOf(step.IssueID, predName, best) + return instancesAt(steps, step.IssueID, predName, best) } // routedTo is R3's interposition clause (DKT-38): for a step some `threshold` @@ -908,8 +918,14 @@ func (s *Scheduler) quorumMet(predName string, def *workflow.Definition, sibling // instancesOf returns every instance of a named step for one issue at an // ordinal — the fanout siblings when there are any, or the single instance. func (s *Scheduler) instancesOf(issueID int, name string, ordinal int) []*db.Step { + return instancesAt(s.steps, issueID, name, ordinal) +} + +// instancesAt is instancesOf over an explicit step set — see +// predecessorInstancesIn for why the rule is shared rather than copied. +func instancesAt(steps []*db.Step, issueID int, name string, ordinal int) []*db.Step { var out []*db.Step - for _, step := range s.steps { + for _, step := range steps { if step.IssueID == issueID && step.StepName == name && step.Ordinal == ordinal { out = append(out, step) } diff --git a/internal/engine/reconcile.go b/internal/engine/reconcile.go index 487d97c1..c9ffa11c 100644 --- a/internal/engine/reconcile.go +++ b/internal/engine/reconcile.go @@ -38,13 +38,25 @@ import ( // reconcileIssueAndRun updates the issue and the run after a step routes, // INSIDE the caller's routing transaction. `spec` is the routed step's pinned // spec (nil when the caller has none), consulted for its threshold's -// interposed targets and nothing else. -func reconcileIssueAndRun(tx *sql.Tx, step *db.Step, spec *workflow.Step, routing string, nowMS int64) error { +// interposed targets and nothing else; `def` is the issue's pinned definition +// (nil likewise), consulted for the `after_fired` cascade and nothing else. +func reconcileIssueAndRun( + tx *sql.Tx, step *db.Step, def *workflow.Definition, spec *workflow.Step, + routing string, nowMS int64, +) error { // BEFORE the completion check: an unrouted gate left `pending` is exactly // what issueStepsComplete must not still be counting (DKT-38). if err := skipUnroutedTargets(tx, step, spec, routing, nowMS); err != nil { return err } + // And the `after_fired` cascade behind whatever this transaction skipped + // (DKT-1085) — the gate above, an `on_fail = "skip"` routing, an operator's + // `--as skip` — for the same reason and in the same place: a successor left + // `pending` is exactly what the completion check must not still count, and + // "skipped in the same transaction as the gate" is the whole contract. + if _, err := propagateAfterFiredSkips(tx, step.RunID, step.IssueID, def, nowMS); err != nil { + return err + } // Abandonment and the live-status mirror are MUTUALLY EXCLUSIVE, not // sequential. Abandonment decides the issue's fate for this run on its own diff --git a/internal/engine/saga.go b/internal/engine/saga.go index a6537a6f..696ab07c 100644 --- a/internal/engine/saga.go +++ b/internal/engine/saga.go @@ -1437,7 +1437,9 @@ func (e *Engine) runRoutingStage( } // The issue mirror and the run rollup, in the same transaction. - if err := reconcileIssueAndRun(tx, step, spec, routing, nowMS); err != nil { + if err := reconcileIssueAndRun( + tx, step, defs[step.WorkflowID], spec, routing, nowMS, + ); err != nil { return err } diff --git a/internal/engine/saga_resume.go b/internal/engine/saga_resume.go index 4495b1a8..271d42ca 100644 --- a/internal/engine/saga_resume.go +++ b/internal/engine/saga_resume.go @@ -246,15 +246,16 @@ func (e *Engine) parkInterruptedGate( // alone, and this was the one park with nothing downstream left to call it. // The run had the same hole, staying `active` with a parked step in it. // - // `spec` is nil, which is exactly right here rather than a shortcut: - // skipUnroutedTargets is threshold-target bookkeeping and an interrupted - // gate decided no threshold. The completion check inside is false by - // construction — the step this call is about is `waiting-human`, which is - // not terminal. + // `spec` and `def` are nil, which is exactly right here rather than a + // shortcut: skipUnroutedTargets is threshold-target bookkeeping and an + // interrupted gate decided no threshold, and the `after_fired` cascade + // follows a skip this park does not perform. The completion check inside + // is false by construction — the step this call is about is + // `waiting-human`, which is not terminal. // // This is NOT the saga advancing; see below. if err := reconcileIssueAndRun( - tx, step, nil, workflow.OnFailWaitingHuman, nowMS, + tx, step, nil, nil, workflow.OnFailWaitingHuman, nowMS, ); err != nil { return err } diff --git a/internal/engine/vote.go b/internal/engine/vote.go index 00f154a3..98028c98 100644 --- a/internal/engine/vote.go +++ b/internal/engine/vote.go @@ -579,7 +579,7 @@ func routeVoteStep( }); err != nil { return err } - if err := reconcileIssueAndRun(tx, step, spec, routing, nowMS); err != nil { + if err := reconcileIssueAndRun(tx, step, def, spec, routing, nowMS); err != nil { return err } return tx.Commit() diff --git a/internal/workflow/parse.go b/internal/workflow/parse.go index ca166519..4111c94f 100644 --- a/internal/workflow/parse.go +++ b/internal/workflow/parse.go @@ -89,6 +89,27 @@ type Step struct { Params map[string]any `toml:"params" json:"params,omitempty"` MinSiblings *int `toml:"min_siblings" json:"min_siblings,omitempty"` Threshold map[string]string `toml:"threshold" json:"threshold,omitempty"` + // AfterFired names predecessors this step runs ONLY IF THEY FIRED + // (DKT-1085). When every instance of a named step ends `skipped` — an + // interposed gate its threshold routed elsewhere (§11.2), a false `when`, + // an `on_fail = "skip"` routing, an operator's `--as skip` — this step is + // terminalized `skipped` in the SAME transaction, and the skip cascades + // through every step declaring `after_fired` on it in turn. + // + // It is a SECOND predecessor list beside `after`, never a replacement: + // `after` keeps its exact meaning (done OR skipped releases the join, J1), + // and every entry here must also appear in `after` (V39a), so R3 already + // orders the step behind the gate and "did it fire?" is settled before the + // step could ever be ready. The corpus case: a `drain-highs` executor that + // should run after `security-vote` APPROVED a round, and not on a round + // where reconcile never routed to the vote at all. `when` cannot say it — + // it reads issue kind and labels only (V22) — and `threshold` cannot: a + // second step-name routing beside the vote would fire first and skip it. + // + // `omitempty`, so the pinned form of every definition that never declares + // it is byte-identical to what it was — an idempotent re-register must not + // read as a CONFLICT (canonical.go). + AfterFired []string `toml:"after_fired" json:"after_fired,omitempty"` // PassFloor refuses a `pass` routing that would exit with declared-floor // work still standing (DKT-870): when this step's routing resolves to // `pass` but its recorded payload holds an element whose `field` value diff --git a/internal/workflow/validate.go b/internal/workflow/validate.go index a9d0adcb..9c27a1ab 100644 --- a/internal/workflow/validate.go +++ b/internal/workflow/validate.go @@ -29,7 +29,7 @@ var RuleIDs = []string{ "V20", "V21", "V21a", "V21b", "V21c", "V21d", "V22", "V23", "V24", "V25", "V25a", "V26", "V27", "V28", "V28a", "V29", "V30", "V31", - "V32", "V33", "V34", "V35", "V36", "V37", "V37a", "V38", + "V32", "V33", "V34", "V35", "V36", "V37", "V37a", "V38", "V39", "V39a", } // VoteRuleResolver reports whether a named vote rule is registered, and lists @@ -390,6 +390,43 @@ func validateStep(def *Definition, step *Step, index int, byName map[string]*Ste } } + // V39: every `after_fired` entry names a step in this workflow (DKT-1085) + // — V9 for the second predecessor list, and for the same reason: a name + // the definition never declares can never end `skipped`, so the successor + // would run unconditionally, which is the exact defect the key exists to + // end. + for _, pred := range step.AfterFired { + if _, ok := byName[pred]; !ok { + return &Error{ + Rule: "V39", Step: step.Name, Field: "after_fired", + Message: fmt.Sprintf( + "step %q: `after_fired` names %q, which is not a step in this workflow", + step.Name, pred), + } + } + } + + // V39a: every `after_fired` entry also appears in `after`. `after_fired` + // asks whether a predecessor FIRED, and the answer exists only once that + // predecessor is terminal — which is exactly what an `after` edge waits + // for (R3). An entry `after` did not name would let the step become ready + // while the gate was still open, run, and be too late to skip. A + // `loop = true` step declares no `after` (V18) and so no `after_fired` + // either: its ordering comes from loop entry. + for _, pred := range step.AfterFired { + if !slices.Contains(step.After, pred) { + return &Error{ + Rule: "V39a", Step: step.Name, Field: "after_fired", + Message: fmt.Sprintf( + "step %q: `after_fired` names %q, which is not in its `after`; "+ + "an `after_fired` predecessor must also be an `after` "+ + "predecessor, so the step waits for it to resolve before "+ + "asking whether it fired", + step.Name, pred), + } + } + } + // V11: inputs shape, existence, and artifact-kind resolution. if err := validateInputs(step, byName); err != nil { return err diff --git a/internal/workflow/validate_test.go b/internal/workflow/validate_test.go index 0afbdebc..68b7c972 100644 --- a/internal/workflow/validate_test.go +++ b/internal/workflow/validate_test.go @@ -217,6 +217,85 @@ executor = "x" src: twoStepPipeline("after = [\"ghost\"]\nexecutor = \"y\"\nemits = \"k\"\n"), wants: []string{`"b"`, "`after`", `"ghost"`, "not a step in this workflow"}, }, + { + rule: "V39", name: "after_fired names an unknown step", + src: twoStepPipeline("after = [\"a\"]\nafter_fired = [\"ghost\"]\nexecutor = \"y\"\nemits = \"k\"\n"), + wants: []string{`"b"`, "`after_fired`", `"ghost"`, "not a step in this workflow"}, + }, + { + rule: "V39a", name: "after_fired names a step not in after", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +[[step]] +name = "gate" +after = ["a"] +type = "human" +on_fail = "skip" +[[step]] +name = "b" +after = ["a"] +after_fired = ["gate"] +executor = "y" +emits = "k" +`, + wants: []string{`"b"`, "`after_fired`", `"gate"`, "not in its `after`"}, + }, + { + rule: "V39a", name: "after_fired on a loop step, which has no after", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "a" +executor = "x" +emits = "k" +[[step]] +name = "check" +after = ["a"] +executor = "y" +emits = "report" +threshold = { "fix-loop" = "any(status == unmet)" } +[[step]] +name = "fix" +executor = "z" +emits = "k" +loop = true +after_loop = "a" +after_fired = ["check"] +`, + wants: []string{`"fix"`, "`after_fired`", `"check"`, "not in its `after`"}, + }, + { + rule: "V39", name: "after_fired beside after registers clean", + src: ` +[pipeline] +name = "w" +version = 1 +[[step]] +name = "reconcile" +executor = "x" +emits = "report" +threshold = { "security-vote" = "any(status == blocked)" } +[[step]] +name = "security-vote" +after = ["reconcile"] +type = "human" +on_fail = "skip" +[[step]] +name = "drain-highs" +after = ["security-vote"] +after_fired = ["security-vote"] +executor = "y" +emits = "k" +`, + }, { rule: "V10", name: "loop step declaring an empty after", src: twoStepPipeline("after = []\nloop = true\nexecutor = \"y\"\nemits = \"k\"\nafter_loop = \"a\"\n"), From 49ed4e060d827670c6531edc1ce4bf027cbb8351 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Wed, 2 Sep 2026 20:14:52 -0700 Subject: [PATCH 016/397] fix(engine): replay a claimed step's recorded inputs on read-back - the claim wrote step_inputs but no read verb used it, so step context and step show re-resolved a handed-out step over the run's current artifacts and reported inputs and a target sha the worker never saw - a claimed step now reads back the bindings its claim recorded; a pending or never-claimed step still resolves live - a re-claim clears the last attempt's bindings before recording its own - the claim records its bindings before pre-gates run so a read that lands mid-claim finds them - step context --live keeps the current-state resolution reachable --- docs/tdd/engine-spine.md | 15 + internal/cli/step.go | 77 +++-- internal/cli/step_context_live_test.go | 102 +++++++ internal/db/steps.go | 37 ++- internal/engine/claim.go | 51 ++++ internal/engine/context.go | 104 ++++++- internal/engine/dkt1054_test.go | 400 +++++++++++++++++++++++++ internal/engine/pregate.go | 12 +- 8 files changed, 760 insertions(+), 38 deletions(-) create mode 100644 internal/cli/step_context_live_test.go create mode 100644 internal/engine/dkt1054_test.go diff --git a/docs/tdd/engine-spine.md b/docs/tdd/engine-spine.md index 1d919703..06fb7613 100644 --- a/docs/tdd/engine-spine.md +++ b/docs/tdd/engine-spine.md @@ -1178,6 +1178,21 @@ same property at the CLI level. `claim --render` returns the assembled packet instead, atomically (§2). +**A read-back replays what the claim recorded** (DKT-1054). Source 4 is resolved +over run state, and run state keeps moving after a step is handed out: the +fixture's `fix@1` binds `reconcile@0` at claim, then `review@1`, `synthesize@1`, +and `reconcile@1` complete at fix@1's own ordinal, and a live re-resolution binds +`reconcile@1` — an artifact produced by reviewing fix@1's diff. The claim writes +the bindings it handed over to `step_inputs` (§6.1) in its own transaction, and +`step context`, `step render`, and `step show`'s target ref read a claimed step +(`attempt > 0`, not back at `pending`) over exactly that set, through the same +resolver, so the read-back is the claim-time bundle however far the run has moved. +A re-claim (a reaped lease, `resolve --as retry`) records its own bindings in place +of the last attempt's. A step not yet handed out — every `action`, `human`, and +`vote` step, and an executor step still `pending` — reads live, since the claim +that will hand it out is what a read of it previews; `step context --live` asks +that question of any step. + ## 6.7 Input resolution §2, verbatim: "Downstream `inputs` resolve over siblings that RECORDED their work diff --git a/internal/cli/step.go b/internal/cli/step.go index b93640ab..14132cb6 100644 --- a/internal/cli/step.go +++ b/internal/cli/step.go @@ -1125,42 +1125,61 @@ The bundle is assembled from the run's PINNED and SNAPSHOTTED state only: the issue as it read at activation, the recorded input artifacts, the pin list, and the run's recorded notes (` + "`notes`" + `, absent when the run has none — see ` + "`docket run note`" + `). It never reads the live issue, never reads the working -tree, and never opens a pinned file. Two calls at the same run state are -byte-identical, whatever has been edited in between. +tree, and never opens a pinned file. + +For a step that has been claimed, ` + "`inputs`" + ` — and the ` + "`target_sha`" + ` / +` + "`target_worktree`" + ` lifted from them — are the bindings its claim RECORDED: +the artifacts the worker was actually handed, replayed however far the run +has moved since. Later rounds completing at the step's own ordinal, or a +re-pin of a producer's diff, never re-bind a claimed step's bundle. A step +not yet claimed, or back at pending for a retry, resolves against the run's +current artifacts — what the claim that hands it out would assemble. + +--live resolves against the run's current artifacts for ANY step: not "what +did this step see" but "what would a claim made now hand over". --meta reports per-section byte counts alongside the bundle, and says whether the template a render would use is pinned.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - w := getWriter(cmd) - conn := getDB(cmd) + return runStepContext(cmd, args, getWriter(cmd)) + }, +} - id, err := stepArg(args[0]) - if err != nil { - return err - } - bundle, err := engine.ReadContext(conn, id, model.NowMS()) - if err != nil { - return stepErr(err, stepLabel(id)) - } +// runStepContext is `step context`'s body, taking its writer the way +// runStepShow does so a test can read the envelope back. +func runStepContext(cmd *cobra.Command, args []string, w *output.Writer) error { + conn := getDB(cmd) - result := stepContextResult{Context: bundle} - if withMeta, _ := cmd.Flags().GetBool("meta"); withMeta { - meta := bundle.Meta() - // The default template ships in the binary, so it is always pinned; - // a --template path is reported by `step render`. - meta.TemplatePinned = true - result.Meta = &meta - } + id, err := stepArg(args[0]) + if err != nil { + return err + } + read := engine.ReadContext + if live, _ := cmd.Flags().GetBool("live"); live { + read = engine.ReadLiveContext + } + bundle, err := read(conn, id, model.NowMS()) + if err != nil { + return stepErr(err, stepLabel(id)) + } - var message string - if !w.JSONMode { - message = fmt.Sprintf("Context for %s: %d input(s), %d pin(s)", - bundle.Step.Step, len(bundle.Inputs), len(bundle.Pins)) - } - w.Success(result, message) - return nil - }, + result := stepContextResult{Context: bundle} + if withMeta, _ := cmd.Flags().GetBool("meta"); withMeta { + meta := bundle.Meta() + // The default template ships in the binary, so it is always pinned; + // a --template path is reported by `step render`. + meta.TemplatePinned = true + result.Meta = &meta + } + + var message string + if !w.JSONMode { + message = fmt.Sprintf("Context for %s: %d input(s), %d pin(s)", + bundle.Step.Step, len(bundle.Inputs), len(bundle.Pins)) + } + w.Success(result, message) + return nil } var stepRenderCmd = &cobra.Command{ @@ -1570,6 +1589,8 @@ func init() { "there too") stepContextCmd.Flags().Bool("meta", false, "Report per-section byte counts alongside the bundle") + stepContextCmd.Flags().Bool("live", false, + "Resolve inputs against the run's current artifacts, not the bindings the step's claim recorded") stepRenderCmd.Flags().String("template", "", "Template file; defaults to the shipped one") stepRenderCmd.Flags().String("executor", "", "Resolved executor hint for {executor} substitution "+ diff --git a/internal/cli/step_context_live_test.go b/internal/cli/step_context_live_test.go new file mode 100644 index 00000000..1620a252 --- /dev/null +++ b/internal/cli/step_context_live_test.go @@ -0,0 +1,102 @@ +package cli + +import ( + "database/sql" + "encoding/json" + "fmt" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/engine" + "github.com/ALT-F4-LLC/docket/internal/model" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-1054 at the CLI boundary. internal/engine/dkt1054_test.go proves the +// RESOLUTION — a claimed step reads back the bindings its claim recorded, a +// pending one reads live; what this file asserts is the verb: `step context` +// replays, and `--live` is the switch back to the current resolution. + +// contextEnvelope runs `step context --json` (with or without --live) on one +// step and decodes the fields under test. +func contextEnvelope(t *testing.T, conn *sql.DB, stepID int, live bool) (targetSHA string, inputs []string) { + t.Helper() + cmd := cmdWithDB(conn) + cmd.Flags().Bool("meta", false, "") + cmd.Flags().Bool("live", live, "") + w, buf := bufWriter(true) + err := runStepContext(cmd, []string{model.FormatStepID(stepID)}, w) + testsupport.Must(t, err, "step context: %v\n%s", err, buf.String()) + + var envelope struct { + Data struct { + Context struct { + TargetSHA string `json:"target_sha"` + Inputs []struct { + Artifact string `json:"artifact"` + Kind string `json:"kind"` + ProducerStep string `json:"producer_step"` + } `json:"inputs"` + } `json:"context"` + } `json:"data"` + } + if err := json.Unmarshal(buf.Bytes(), &envelope); err != nil { + t.Fatalf("unmarshal: %v\n%s", err, buf.String()) + } + for _, in := range envelope.Data.Context.Inputs { + inputs = append(inputs, in.Kind+"/"+in.ProducerStep+"/"+in.Artifact) + } + return envelope.Data.Context.TargetSHA, inputs +} + +// TestStepContextReplaysTheClaimedBindingsUnlessLive: the judge is seated on +// the executor's commit; the executor's diff is then superseded at the same +// ordinal (a re-pin's row). The verb reports what the judge was handed; --live +// reports the superseding record. +func TestStepContextReplaysTheClaimedBindingsUnlessLive(t *testing.T) { + conn := newTestDB(t) + implementID, reviewID := showTargetRun(t, conn, "cafe1234cafe1234", "/worktrees/issue-under-test") + + claim, err := engine.ClaimStep(conn, reviewID, engine.ClaimOptions{Owner: "judge", NowMS: model.NowMS()}) + testsupport.Must(t, err, "claim review: %v", err) + handed := claim.Context.Inputs[0].Artifact + + var runID, prior int + err = conn.QueryRow(`SELECT run_id FROM steps WHERE id = ?`, implementID).Scan(&runID) + testsupport.Must(t, err, "reading the run: %v", err) + err = conn.QueryRow( + `SELECT MAX(id) FROM artifacts WHERE step_id = ? AND kind = ?`, + implementID, engine.ArtifactKindIssueDiff).Scan(&prior) + testsupport.Must(t, err, "reading the recorded diff: %v", err) + + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + superseding, err := db.InsertArtifactTx(tx, db.Artifact{ + RunID: runID, StepID: implementID, Kind: engine.ArtifactKindIssueDiff, + Body: "the re-pinned diff", Payload: `{"head":"feed5678feed5678","worktree":"/worktrees/repinned"}`, + SHA256: workflow.SHA256([]byte("the re-pinned diff")), Supersedes: &prior, + }, model.NowMS()) + if err != nil { + tx.Rollback() + t.Fatalf("InsertArtifactTx: %v", err) + } + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + + sha, inputs := contextEnvelope(t, conn, reviewID, false) + if sha != "cafe1234cafe1234" { + t.Errorf("step context target_sha = %q, want the handed-over cafe1234cafe1234", sha) + } + if len(inputs) != 1 || inputs[0] != "issue.diff/implement@0/"+handed { + t.Errorf("step context inputs = %v, want the handed-over [issue.diff/implement@0/%s]", inputs, handed) + } + + liveSHA, liveInputs := contextEnvelope(t, conn, reviewID, true) + if liveSHA != "feed5678feed5678" { + t.Errorf("step context --live target_sha = %q, want the superseding feed5678feed5678", liveSHA) + } + want := fmt.Sprintf("issue.diff/implement@0/ARTIFACT-%d", superseding) + if len(liveInputs) != 1 || liveInputs[0] != want { + t.Errorf("step context --live inputs = %v, want [%s]", liveInputs, want) + } +} diff --git a/internal/db/steps.go b/internal/db/steps.go index 78f9aae8..d483e3cb 100644 --- a/internal/db/steps.go +++ b/internal/db/steps.go @@ -961,7 +961,7 @@ func scanArtifacts(rows *sql.Rows, err error) ([]*Artifact, error) { // time, and run state moves: a later ordinal's artifact would re-resolve the // same input differently. Storing what was actually handed over is what makes // the ledger answer "what did this step see" rather than "what would it see -// now". +// now" — and ListStepInputArtifactsTx is how a read-back asks it (DKT-1054). func InsertStepInputTx(tx *sql.Tx, stepID, position, artifactID int) error { _, err := tx.Exec( `INSERT OR REPLACE INTO step_inputs (step_id, position, artifact_id) @@ -974,6 +974,41 @@ func InsertStepInputTx(tx *sql.Tx, stepID, position, artifactID int) error { return nil } +// ClearStepInputsTx drops a step's recorded input bindings, so a re-claim can +// record the bindings of the attempt that is about to run in their place +// (DKT-1054). +// +// The table's key is (step, position, artifact), so without this a retried +// step's second claim would ADD its bindings beside the first attempt's rather +// than replace them, and a read-back would see the union of two attempts — +// neither of which any executor saw. The step's snapshot is the snapshot of +// its CURRENT attempt; a lapsed or retried attempt's bindings are history the +// event log keeps, not a second answer to "what did this step see". +func ClearStepInputsTx(tx *sql.Tx, stepID int) error { + if _, err := tx.Exec(`DELETE FROM step_inputs WHERE step_id = ?`, stepID); err != nil { + return fmt.Errorf("clearing step inputs: %w", err) + } + return nil +} + +// ListStepInputArtifactsTx reads the artifacts a step's claim RECORDED as its +// inputs (InsertStepInputTx), ordered by id like every other artifact reader — +// the snapshot `step context` replays for a step that has been handed out +// (DKT-1054). +// +// It is the artifact rows themselves, not the (position, artifact) pairs, +// because the reader re-runs §6.7's resolution over exactly this set: the +// declared-position order, the engine-produced forms that bind no artifact +// (`issue.body`, an empty `issue.diff`, `gate-results`), and the within-input +// sort all come from the same code the claim ran, so a replayed bundle has +// the claim-time bundle's shape by construction rather than by a second +// rendering of it. Empty for a step whose claim bound no artifact at all. +func ListStepInputArtifactsTx(tx *sql.Tx, stepID int) ([]*Artifact, error) { + return scanArtifacts(tx.Query( + artifactSelect+` WHERE id IN (SELECT artifact_id FROM step_inputs WHERE step_id = ?) + ORDER BY id`, stepID)) +} + // nullableInt maps 0 to SQL NULL, for optional foreign keys. func nullableInt(n int) any { if n == 0 { diff --git a/internal/engine/claim.go b/internal/engine/claim.go index 57fe4cb3..4dd2ef8e 100644 --- a/internal/engine/claim.go +++ b/internal/engine/claim.go @@ -562,6 +562,22 @@ func claimStepWithGates( return nil, err } + // The claim's input snapshot is recorded HERE as well as in transaction B + // (DKT-1054). From this commit on the step reads as handed out + // (recordedClaim), and a read-back replays its recorded bindings — so a + // `step context` or `step show` that lands while the pre-gates run must + // find the bindings this claim resolved, not an empty set standing in for + // a snapshot not yet taken. B assembles the bundle the worker actually + // receives and re-records over these; the two differ only if another + // step of the run completed while the pre-gates ran. + provisional, err := AssembleContext(tx, sched, fresh, ttls) + if err != nil { + return nil, err + } + if err := recordStepInputs(tx, fresh.ID, provisional.Inputs); err != nil { + return nil, err + } + if err := tx.Commit(); err != nil { return nil, fmt.Errorf("committing the claim: %w", err) } @@ -656,7 +672,16 @@ func claimStepWithGates( // inputs (`issue.body`, and an `issue.diff` with no artifact) have no artifact // row to bind and are skipped — there is nothing to record but the fact they // were empty, which the bundle already says. +// +// The step's earlier bindings are cleared first (DKT-1054): the record is the +// snapshot of THIS claim, and a retried or re-claimed step's read-back +// (AssembleRecordedContext) must replay what this attempt was handed, not the +// union of every attempt's bindings. Same transaction as the claim, so a +// reader never sees the row between the two. func recordStepInputs(tx *sql.Tx, stepID int, inputs []ContextInput) error { + if err := db.ClearStepInputsTx(tx, stepID); err != nil { + return err + } for position, in := range inputs { artifactID, ok := artifactIDOf(in.Artifact) if !ok { @@ -708,7 +733,30 @@ func unclaimableReason(kind string) string { // `next`/`claim`, and this is neither: a read verb that reaped would make // "I only looked at it" untrue, and would let a `--meta` query change an // attempt counter. +// +// A step that has been handed out reads back the bundle its claim recorded +// (AssembleRecordedContext, DKT-1054): the inputs, and the target ref lifted +// from them, are the ones the worker was given, however far the run has moved +// since. A step not yet handed out — or back at `pending` for a retry — reads +// live, as the claim that will hand it out would assemble it (recordedClaim). +// ReadLiveContext asks the live question of any step. func ReadContext(conn *sql.DB, stepID int, nowMS int64) (*Context, error) { + return readContext(conn, stepID, nowMS, false) +} + +// ReadLiveContext is ReadContext resolved over the run's artifacts AS THEY +// STAND, whether or not the step has been handed out — `step context --live`. +// +// It answers a different question from ReadContext on a claimed step: not +// "what did this step see" but "what would a claim made now hand over" — the +// preview an operator wants before authorizing a retry, and the one reading +// every consumer had before DKT-1054, kept reachable by name rather than +// removed. +func ReadLiveContext(conn *sql.DB, stepID int, nowMS int64) (*Context, error) { + return readContext(conn, stepID, nowMS, true) +} + +func readContext(conn *sql.DB, stepID int, nowMS int64, live bool) (*Context, error) { step, err := db.GetStep(conn, stepID) if errors.Is(err, db.ErrStepNotFound) { return nil, notFoundErr(err, "step %s not found", model.FormatStepID(stepID)) @@ -744,6 +792,9 @@ func ReadContext(conn *sql.DB, stepID int, nowMS int64) (*Context, error) { "step %s not found", model.FormatStepID(stepID)) } + if !live && recordedClaim(fresh) { + return AssembleRecordedContext(tx, sched, fresh, ttls) + } return AssembleContext(tx, sched, fresh, ttls) } diff --git a/internal/engine/context.go b/internal/engine/context.go index 5f784243..b6288049 100644 --- a/internal/engine/context.go +++ b/internal/engine/context.go @@ -21,7 +21,10 @@ import ( // 1. the step row and its definition, from the PINNED workflow's `parsed` // 2. `run_issues.body_snapshot` — never the live issue body // 3. `run_issues.issue_snapshot` — title/kind/labels/scope as of activation -// 4. recorded `artifacts` bodies, resolved per §6.7's input rule +// 4. recorded `artifacts` bodies, resolved per §6.7's input rule — over the +// run's table at claim, and over the bindings THAT CLAIM RECORDED +// (`step_inputs`) on every read-back of a handed-out step, so the run +// moving on cannot re-bind what a step already saw (DKT-1054) // 5. `pins` rows (path + hash) — the LIST, not the file contents // 6. `run_notes` rows — the standing statements `run note add` recorded // against the run, every one, in insertion order @@ -248,13 +251,97 @@ const ( ArtifactKindIssueDiff = "issue.diff" ) -// AssembleContext builds one step's context bundle inside tx. +// AssembleContext builds one step's context bundle inside tx, resolving its +// inputs over the run's artifacts AS THEY STAND — the claim-time assembly. // // It takes a transaction because `step claim` mints the token and assembles the // bundle in ONE transaction — "one atomic mediation: an unclaimed executor has // nothing, a claimed one has everything" (engine-core §8). +// +// It is the right assembly for a step that has not been handed out: what it +// returns is what a claim made now would hand over. For a step that HAS been +// handed out, the answer to "what did this step see" is AssembleRecordedContext. func AssembleContext( tx *sql.Tx, sched *Scheduler, step *db.Step, ttls ttlConfig, +) (*Context, error) { + return assembleContext(tx, sched, step, ttls, liveArtifacts) +} + +// AssembleRecordedContext builds a step's context bundle over the artifacts its +// claim RECORDED as its inputs (`step_inputs`), instead of the run's current +// artifact table (DKT-1054). +// +// Resolution (§6.7) is a function of run state at assembly time, and run state +// keeps moving after a step is handed out: the fixture's `fix@1` declares +// `reconcile.findings`, binds `reconcile@0` at claim (the round that routed +// it), and then `review@1`, `synthesize@1`, and `reconcile@1` all complete AT +// ITS OWN ORDINAL. Re-resolving live after that binds `reconcile@1` — an +// artifact produced by reviewing fix@1's own diff — and reports it as the input +// fix@1 read. RUN-64's STEP-2992 read exactly that way (and STEP-2974, +// STEP-3002, and the RUN-53/56/61 passes before it), while `step_inputs` for +// the same row held ARTIFACT-2709 (`reconcile@0`) the whole time, written by +// the claim and read by nothing. The same drift moved `issue.diff` — and with +// it `target_sha` — onto a record superseded after the step ran (a re-pin, +// DKT-1034, revises a producer's diff in place at the same ordinal). +// +// THE RESOLVERS ARE NOT DUPLICATED. This runs the identical assembly over a +// different artifact set — the recorded rows — so the declared-position order, +// the engine-produced forms (`issue.body` from the snapshot, an empty +// `issue.diff`, `gate-results` from the ledger), the prior-round pass, the +// dedupe, and the target lift are the same code the claim ran. Every recorded +// artifact was selected by exactly the rule that re-selects it here, and no +// artifact recorded after the claim is a candidate, so the result is the +// claim-time bundle: byte-identical inputs, however far the run has moved. +// +// Two forms are ledger reads rather than artifact reads and stay so: +// `.gate-results` and `.vote-record` resolve the producer INSTANCE +// by the same ordinal rule and serve its recorded rows, which are append-only +// per instance. A self-declared `.gate-results` therefore carries the +// step's completion-side gate rows on a read-back that were not yet recorded +// at claim; that is the one engine-produced input whose read-back can grow. +func AssembleRecordedContext( + tx *sql.Tx, sched *Scheduler, step *db.Step, ttls ttlConfig, +) (*Context, error) { + return assembleContext(tx, sched, step, ttls, recordedArtifacts) +} + +// artifactSource loads the artifact set one assembly resolves over. +type artifactSource func(tx *sql.Tx, step *db.Step) ([]*db.Artifact, error) + +// liveArtifacts is the run's whole artifact table — the claim's view. +func liveArtifacts(tx *sql.Tx, step *db.Step) ([]*db.Artifact, error) { + return db.ListRunArtifactsTx(tx, step.RunID) +} + +// recordedArtifacts is the set the step's claim recorded — the read-back's view. +func recordedArtifacts(tx *sql.Tx, step *db.Step) ([]*db.Artifact, error) { + return db.ListStepInputArtifactsTx(tx, step.ID) +} + +// recordedClaim reports whether a step's context is the one a claim recorded — +// the rule by which a read verb picks AssembleRecordedContext over +// AssembleContext (DKT-1054). +// +// A step has a recorded snapshot from the claim that handed it out, and keeps +// it through everything that attempt becomes: `claimed`, `running`, `gated`, +// `done`, a gate park, a supersession by `resolve --as fix-round` — all of +// them are that attempt, and "what did this step see" has one answer for all +// of them. `attempt` counts claims for the step's whole life (it is never +// reset), so `attempt > 0` is "was ever handed out". +// +// A step back at `pending` is the exception, and it is deliberate: a lapsed +// lease reaped, or `resolve --as retry`, returns the row to the pool with its +// old bindings still in the table, and the next claim will record new ones. +// Until it does, the honest answer to a read is the live one — what THAT claim +// will hand over — not the bindings of an attempt that is over. A step never +// claimed (`attempt == 0`: every `action`, `human`, and `vote` step, and every +// executor step still waiting) has no snapshot, so it reads live too. +func recordedClaim(step *db.Step) bool { + return step.Attempt > 0 && step.Status != db.StepPending +} + +func assembleContext( + tx *sql.Tx, sched *Scheduler, step *db.Step, ttls ttlConfig, source artifactSource, ) (*Context, error) { def := sched.defs[step.WorkflowID] if def == nil { @@ -280,14 +367,15 @@ func AssembleContext( return nil, err } - // Source 4: the resolved input artifacts. The run's artifact rows are - // loaded ONCE and shared with the prior-round pass below — both read the - // same snapshot, and the table's bodies scale with the run's whole - // recorded output, so a second load doubled exactly the cost this path - // pays most of. + // Source 4: the resolved input artifacts. The artifact rows are loaded + // ONCE — from the run's table for a claim, from the claim's recorded + // bindings for a read-back (the source) — and shared with the prior-round + // pass below: both read the same snapshot, and the table's bodies scale + // with the run's whole recorded output, so a second load doubled exactly + // the cost this path pays most of. var artifacts []*db.Artifact if len(spec.Inputs) > 0 || loopReentryEmitter(sched, step, spec) { - artifacts, err = db.ListRunArtifactsTx(tx, step.RunID) + artifacts, err = source(tx, step) if err != nil { return nil, err } diff --git a/internal/engine/dkt1054_test.go b/internal/engine/dkt1054_test.go new file mode 100644 index 00000000..a4a1e155 --- /dev/null +++ b/internal/engine/dkt1054_test.go @@ -0,0 +1,400 @@ +package engine + +import ( + "database/sql" + "encoding/json" + "fmt" + "reflect" + "sort" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/db" + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-1054: `step context` on a step that has been handed out replays the +// bindings its claim RECORDED (`step_inputs`), instead of re-resolving §6.7 +// over the run's current artifact table. +// +// The claim always wrote the table; nothing read it. So a completed `fix@1` +// — which bound `reconcile@0` at claim, the round that routed it — read back +// as consuming `reconcile@1` once that step completed at fix@1's own ordinal, +// and a judge's `target_sha` moved onto whichever record superseded its +// producer's diff after the judge had already been seated. RUN-64's STEP-2992 +// is the first shape; RUN-53/56/61's target drift is the second. Both are the +// one defect: assembly is a function of run state, run state moves, and the +// read verb asked the live question about a step whose answer was frozen. + +// inputRefs renders inputs the way the issue's evidence table reads them — +// kind, producer, artifact — for a failure message a reader can check against +// the ledger. +func inputRefs(inputs []ContextInput) []string { + out := make([]string, 0, len(inputs)) + for _, in := range inputs { + out = append(out, fmt.Sprintf("%s/%s/%s", in.Kind, in.ProducerStep, in.Artifact)) + } + return out +} + +// producersOfKind lists the producer instances bound for one input kind, in +// bundle order. +func producersOfKind(inputs []ContextInput, kind string) []string { + var out []string + for _, in := range inputs { + if in.Kind == kind { + out = append(out, in.ProducerStep) + } + } + return out +} + +// artifactOfKind is the artifact ref bound for one input kind, or "" when the +// bundle carries none. +func artifactOfKind(inputs []ContextInput, kind string) string { + for _, in := range inputs { + if in.Kind == kind { + return in.Artifact + } + } + return "" +} + +// newestArtifactID is the highest artifact id one step recorded of a kind. +func newestArtifactID(t *testing.T, conn *sql.DB, stepID int, kind string) int { + t.Helper() + var id int + err := conn.QueryRow( + `SELECT MAX(id) FROM artifacts WHERE step_id = ? AND kind = ?`, stepID, kind, + ).Scan(&id) + testsupport.Must(t, err, "newest %s of step %d: %v", kind, stepID, err) + return id +} + +// recordedInputIDs reads a step's `step_inputs` bindings, by artifact id. +func recordedInputIDs(t *testing.T, conn *sql.DB, stepID int) []int { + t.Helper() + rows, err := conn.Query( + `SELECT artifact_id FROM step_inputs WHERE step_id = ? ORDER BY artifact_id`, stepID) + testsupport.Must(t, err, "reading step_inputs: %v", err) + defer rows.Close() + var out []int + for rows.Next() { + var id int + testsupport.Must(t, rows.Scan(&id), "scanning step_inputs") + out = append(out, id) + } + return out +} + +// supersedeIssueDiff records a NEW `issue.diff` for a step that supersedes +// its newest one — the row a re-pin (DKT-1034) writes — carrying the round +// record every consumer lifts its target from. +func supersedeIssueDiff( + t *testing.T, conn *sql.DB, runID, stepID int, head, worktree string, +) int { + t.Helper() + prior := newestArtifactID(t, conn, stepID, ArtifactKindIssueDiff) + payload, err := json.Marshal(map[string]string{"head": head, "worktree": worktree}) + testsupport.Must(t, err, "encoding the round record: %v", err) + body := "the re-pinned diff at " + head + + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + id, err := db.InsertArtifactTx(tx, db.Artifact{ + RunID: runID, StepID: stepID, Kind: ArtifactKindIssueDiff, + Body: body, Payload: string(payload), + SHA256: workflow.SHA256([]byte(body)), Supersedes: &prior, + }, nowMS+1) + if err != nil { + tx.Rollback() + t.Fatalf("InsertArtifactTx: %v", err) + } + testsupport.Must(t, tx.Commit(), "Commit: %v", err) + return id +} + +// reapStep returns a lapsed step to the pool, as `next`/`claim` do lazily. +func reapStep(t *testing.T, conn *sql.DB, stepID int) { + t.Helper() + tx, err := conn.Begin() + testsupport.Must(t, err, "Begin: %v", err) + if err := db.ReapStepTx(tx, stepID, nowMS); err != nil { + tx.Rollback() + t.Fatalf("ReapStepTx: %v", err) + } + testsupport.Must(t, tx.Commit(), "Commit: %v", err) +} + +// TestClaimedStepContextReplaysTheRecordedInputs is RUN-64's STEP-2992, in +// the fixture: `fix@1` reads `reconcile.findings`, is handed `reconcile@0` +// (the instance whose routing minted it), and then `review@1`, `synthesize@1` +// and `reconcile@1` complete AT ITS OWN ORDINAL. The read-back must be the +// bundle the worker was handed — and the live resolution, kept reachable as +// `--live`, must be the different answer it always was, or the test proves +// nothing. +func TestClaimedStepContextReplaysTheRecordedInputs(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + + // Round 0 through its reconcile, whose `fix-loop` routing mints fix@1. + driveRoundToReconcile(t, conn, e, 0) + + // fix@1 is claimed by hand, so the bundle the worker was HANDED is in + // hand for the comparison below. + driveFixtureRound(t, 1) + fixID := stepIDByInstance(t, conn, "fix@1") + claim, err := ClaimStep(conn, fixID, ClaimOptions{Owner: "worker", NowMS: nowMS}) + testsupport.Must(t, err, "claim fix@1: %v", err) + handed := claim.Context + err = e.CompleteStep(conn, fixID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the fix summary"), NowMS: nowMS, + }) + testsupport.Must(t, err, "complete fix@1: %v", err) + + if got := producersOfKind(handed.Inputs, "findings"); !reflect.DeepEqual(got, []string{"reconcile@0"}) { + t.Fatalf("premise: fix@1's claim bound `reconcile.findings` to %v, want [reconcile@0]", got) + } + + // The rest of round 1 completes at fix@1's OWN ordinal, ending in a + // `findings` artifact from `reconcile@1` — produced by reviewing fix@1's + // diff, and so an artifact fix@1 cannot have read. + completeReviewFanout(t, conn, e, 1) + claimAndComplete(t, conn, e, "synthesize@1", "the synthesis", roundPayload(1)) + driveAction(t, conn, e, "reconcile@1") + if got := stepStatus(t, conn, "reconcile@1"); got != db.StepDone { + t.Fatalf("premise: reconcile@1 = %q, want %q", got, db.StepDone) + } + + replayed, err := ReadContext(conn, fixID, nowMS) + testsupport.Must(t, err, "ReadContext(fix@1): %v", err) + + if !reflect.DeepEqual(replayed.Inputs, handed.Inputs) { + t.Errorf("fix@1's read-back is not the bundle its claim handed over.\n"+ + "handed: %v\nread-back: %v", inputRefs(handed.Inputs), inputRefs(replayed.Inputs)) + } + if got := producersOfKind(replayed.Inputs, "findings"); !reflect.DeepEqual(got, []string{"reconcile@0"}) { + t.Errorf("`reconcile.findings` reads back from %v, want [reconcile@0] — "+ + "the read-back re-bound to a later step of the same ordinal", got) + } + for _, in := range replayed.Inputs { + if in.Kind == "findings" && strings.Contains(in.Payload, "C-201") { + t.Errorf("fix@1's read-back carries round 2's cluster from %s: %s", + in.ProducerStep, in.Payload) + } + } + if got := producersOfKind(replayed.Inputs, "change-summary"); !reflect.DeepEqual(got, []string{"implement@0"}) { + t.Errorf("`implement.change-summary` reads back from %v, want [implement@0]", got) + } + + // The read-back is stable: a second call at a later run state, and after + // the run moved on, returns the same bytes. + first := bundleJSON(t, conn, fixID) + driveFixtureRound(t, 2) + claimAndComplete(t, conn, e, "fix@2", "the second fix summary", "") + if again := bundleJSON(t, conn, fixID); again != first { + t.Errorf("fix@1's bundle changed after fix@2 recorded:\n%s\n%s", first, again) + } + + // --live is the old reading, and it is DIFFERENT — which is what makes the + // assertions above non-vacuous. + live, err := ReadLiveContext(conn, fixID, nowMS) + testsupport.Must(t, err, "ReadLiveContext(fix@1): %v", err) + if got := producersOfKind(live.Inputs, "findings"); !reflect.DeepEqual(got, []string{"reconcile@1"}) { + t.Errorf("--live binds `reconcile.findings` to %v, want [reconcile@1] (the current resolution)", got) + } +} + +// TestClaimedStepTargetIsTheHandedOverTarget is the `target_sha` half: a +// record superseding the producer's diff at the SAME ordinal — what a re-pin +// writes — moves every not-yet-seated consumer onto the new tree (DKT-1034's +// contract) and moves NO consumer that was already seated. `step show` must +// say the same as `step context` on both, since the wave seats a panel from +// the row (DKT-1056). +func TestClaimedStepTargetIsTheHandedOverTarget(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + e.HeadFn = func(string) string { return "head-original" } + + implementID := stepIDByInstance(t, conn, "implement@0") + claim, err := ClaimStep(conn, implementID, ClaimOptions{Owner: "w", NowMS: nowMS}) + testsupport.Must(t, err, "claim implement: %v", err) + err = e.CompleteStep(conn, implementID, CompleteOptions{ + Token: claim.Token, Artifact: []byte("the change summary"), + WorkDir: "/worktrees/original", NowMS: nowMS, + }) + testsupport.Must(t, err, "complete implement: %v", err) + + seatedID := stepIDByInstance(t, conn, "review@0#0") + seated, err := ClaimStep(conn, seatedID, ClaimOptions{Owner: "judge", NowMS: nowMS}) + testsupport.Must(t, err, "claim review@0#0: %v", err) + if seated.Context.TargetSHA != "head-original" { + t.Fatalf("premise: the seated judge's target = %q, want head-original", seated.Context.TargetSHA) + } + handedDiff := artifactOfKind(seated.Context.Inputs, ArtifactKindIssueDiff) + + repinned := supersedeIssueDiff(t, conn, run.ID, implementID, "head-repinned", "/worktrees/repinned") + + // The seated judge: the bundle, the row, and the render all keep the + // target it was handed. + replayed, err := ReadContext(conn, seatedID, nowMS) + testsupport.Must(t, err, "ReadContext(review@0#0): %v", err) + if replayed.TargetSHA != "head-original" || replayed.TargetWorktree != "/worktrees/original" { + t.Errorf("the seated judge reads back target (%q, %q), want the handed-over (head-original, /worktrees/original)", + replayed.TargetSHA, replayed.TargetWorktree) + } + if got := artifactOfKind(replayed.Inputs, ArtifactKindIssueDiff); got != handedDiff { + t.Errorf("the seated judge's issue.diff reads back as %s, want the handed-over %s", got, handedDiff) + } + view, err := LoadStepView(conn, seatedID, nowMS) + testsupport.Must(t, err, "LoadStepView(review@0#0): %v", err) + if view.TargetSHA != replayed.TargetSHA || view.TargetWorktree != replayed.TargetWorktree { + t.Errorf("step show says (%q, %q) while the bundle says (%q, %q)", + view.TargetSHA, view.TargetWorktree, replayed.TargetSHA, replayed.TargetWorktree) + } + rendered, err := RenderStep(conn, seatedID, "", nowMS) + testsupport.Must(t, err, "RenderStep(review@0#0): %v", err) + if !strings.Contains(rendered.Packet, "target_sha: head-original") { + t.Errorf("the seated judge's packet no longer names its handed-over target:\n%s", rendered.Packet) + } + + // --live is the current resolution: the superseding record. + live, err := ReadLiveContext(conn, seatedID, nowMS) + testsupport.Must(t, err, "ReadLiveContext(review@0#0): %v", err) + if live.TargetSHA != "head-repinned" { + t.Errorf("--live target = %q, want head-repinned", live.TargetSHA) + } + if got := artifactOfKind(live.Inputs, ArtifactKindIssueDiff); got != fmt.Sprintf("ARTIFACT-%d", repinned) { + t.Errorf("--live issue.diff = %s, want the superseding ARTIFACT-%d", got, repinned) + } + + // An UNSEATED sibling has no snapshot: it binds the superseding record, + // and the row says so too. + unseatedID := stepIDByInstance(t, conn, "review@0#1") + pending, err := ReadContext(conn, unseatedID, nowMS) + testsupport.Must(t, err, "ReadContext(review@0#1): %v", err) + if pending.TargetSHA != "head-repinned" || pending.TargetWorktree != "/worktrees/repinned" { + t.Errorf("the unseated judge's target = (%q, %q), want the superseding (head-repinned, /worktrees/repinned)", + pending.TargetSHA, pending.TargetWorktree) + } + pendingView, err := LoadStepView(conn, unseatedID, nowMS) + testsupport.Must(t, err, "LoadStepView(review@0#1): %v", err) + if pendingView.TargetSHA != "head-repinned" { + t.Errorf("step show on the unseated judge says %q, want head-repinned", pendingView.TargetSHA) + } +} + +// TestReclaimRecordsTheNewAttemptsBindings pins the snapshot's lifecycle: it +// is the CURRENT attempt's. A lapsed lease still reads back its attempt; a +// step reaped to `pending` reads live, because no attempt is in flight and the +// next claim is what will hand it out; and that claim records its bindings IN +// PLACE of the old — not beside them, which the table's (step, position, +// artifact) key would otherwise allow. +func TestReclaimRecordsTheNewAttemptsBindings(t *testing.T) { + conn := mustDB(t) + run, _ := activatedRun(t, conn) + e := testEngine() + + implementID := stepIDByInstance(t, conn, "implement@0") + claimAndComplete(t, conn, e, "implement@0", "the change summary", "") + first := fmt.Sprintf("ARTIFACT-%d", newestArtifactID(t, conn, implementID, "change-summary")) + + reviewID := stepIDByInstance(t, conn, "review@0#0") + claim, err := ClaimStep(conn, reviewID, ClaimOptions{Owner: "judge-1", NowMS: nowMS}) + testsupport.Must(t, err, "claim review@0#0: %v", err) + if got := artifactOfKind(claim.Context.Inputs, "change-summary"); got != first { + t.Fatalf("premise: the first claim bound change-summary %s, want %s", got, first) + } + + // implement@0 records a newer change-summary after the judge was handed + // the first; latestPerProducer binds the newest live. + revised := fmt.Sprintf("ARTIFACT-%d", + recordArtifact(t, conn, run.ID, implementID, "change-summary", "the revised summary")) + + changeSummary := func(label string, read func(*sql.DB, int, int64) (*Context, error)) string { + t.Helper() + bundle, err := read(conn, reviewID, nowMS) + testsupport.Must(t, err, "%s: %v", label, err) + return artifactOfKind(bundle.Inputs, "change-summary") + } + if got := changeSummary("read-back under a live lease", ReadContext); got != first { + t.Errorf("under a live lease the read-back binds %s, want the handed-over %s", got, first) + } + if got := changeSummary("live under a live lease", ReadLiveContext); got != revised { + t.Errorf("--live binds %s, want the newest %s", got, revised) + } + + // The lease lapses: the attempt is over, but it is still the attempt this + // row's snapshot describes. + execSQL(t, conn, `UPDATE steps SET expires_ms = ? WHERE id = ?`, nowMS-1, reviewID) + if got := changeSummary("read-back under a lapsed lease", ReadContext); got != first { + t.Errorf("under a lapsed lease the read-back binds %s, want the attempt's %s", got, first) + } + + // Reaped to pending: nothing is in flight, so the read is what the next + // claim will hand over. + reapStep(t, conn, reviewID) + if got := stepStatus(t, conn, "review@0#0"); got != db.StepPending { + t.Fatalf("premise: reaped review@0#0 = %q, want %q", got, db.StepPending) + } + if got := changeSummary("read-back at pending", ReadContext); got != revised { + t.Errorf("back at pending the read-back binds %s, want the live %s", got, revised) + } + + // The re-claim records the new attempt's bindings in place of the old. + again, err := ClaimStep(conn, reviewID, ClaimOptions{Owner: "judge-2", NowMS: nowMS}) + testsupport.Must(t, err, "re-claim review@0#0: %v", err) + if got := artifactOfKind(again.Context.Inputs, "change-summary"); got != revised { + t.Errorf("the re-claim handed over %s, want %s", got, revised) + } + if got := changeSummary("read-back after the re-claim", ReadContext); got != revised { + t.Errorf("after the re-claim the read-back binds %s, want %s", got, revised) + } + ids := recordedInputIDs(t, conn, reviewID) + var want []int + for _, in := range again.Context.Inputs { + if id, ok := artifactIDOf(in.Artifact); ok { + want = append(want, id) + } + } + sort.Ints(want) + if !reflect.DeepEqual(ids, want) { + t.Errorf("step_inputs after the re-claim = %v, want exactly the re-claim's bindings %v — "+ + "the first attempt's %s must not linger beside them", ids, want, first) + } +} + +// TestNeverClaimedStepsReadLive pins the other side of the rule: a step the +// engine runs itself is never claimed, records no snapshot, and reads live — +// so the action's own stdin (actionContext) and a read of it agree. +func TestNeverClaimedStepsReadLive(t *testing.T) { + conn := mustDB(t) + activatedRun(t, conn) + e := testEngine() + driveRoundToReconcile(t, conn, e, 0) + + reconcileID := stepIDByInstance(t, conn, "reconcile@0") + step, err := db.GetStep(conn, reconcileID) + testsupport.Must(t, err, "GetStep: %v", err) + if step.Attempt != 0 || recordedClaim(step) { + t.Fatalf("premise: reconcile@0 (attempt %d, %s) reads as a recorded claim", step.Attempt, step.Status) + } + if ids := recordedInputIDs(t, conn, reconcileID); len(ids) != 0 { + t.Fatalf("premise: an action step recorded step_inputs %v", ids) + } + + recorded, err := ReadContext(conn, reconcileID, nowMS) + testsupport.Must(t, err, "ReadContext(reconcile@0): %v", err) + live, err := ReadLiveContext(conn, reconcileID, nowMS) + testsupport.Must(t, err, "ReadLiveContext(reconcile@0): %v", err) + if !reflect.DeepEqual(recorded.Inputs, live.Inputs) { + t.Errorf("a never-claimed step reads differently with and without --live:\n%v\n%v", + inputRefs(recorded.Inputs), inputRefs(live.Inputs)) + } + if got := producersOfKind(recorded.Inputs, "findings"); !reflect.DeepEqual(got, []string{"synthesize@0"}) { + t.Errorf("reconcile@0 reads `synthesize.findings` from %v, want [synthesize@0]", got) + } +} diff --git a/internal/engine/pregate.go b/internal/engine/pregate.go index 12d659cc..5e4e8543 100644 --- a/internal/engine/pregate.go +++ b/internal/engine/pregate.go @@ -129,6 +129,12 @@ func resolvedTargetFor( // resolvable round record. A vote panel seats itself on what this says; a // plausible-looking sha invented here would seat judges on the wrong tree, // which is worse than seating them on their own HEAD, because it is silent. +// +// A step that has been handed out resolves over the artifacts its claim +// recorded, exactly as its bundle does (recordedClaim, DKT-1054): the pair +// `step show` reports is the pair `step context` carries, on every step, or +// the two verbs would disagree about a completed step's target the moment a +// later round — or a re-pin of its producer's diff — recorded a newer one. func stepTargetRef( tx *sql.Tx, sched *Scheduler, step *db.Step, ) (sha, worktree string, err error) { @@ -136,7 +142,11 @@ func stepTargetRef( if spec == nil || !consumesIssueDiff(spec) { return "", "", nil } - artifacts, err := db.ListRunArtifactsTx(tx, step.RunID) + source := liveArtifacts + if recordedClaim(step) { + source = recordedArtifacts + } + artifacts, err := source(tx, step) if err != nil { return "", "", err } From b30108807e5183b264a3e2f79510b2281aefd3a9 Mon Sep 17 00:00:00 2001 From: Erik Reinert <4638629+erikreinert@users.noreply.github.com> Date: Thu, 3 Sep 2026 02:28:55 -0700 Subject: [PATCH 017/397] fix(engine): give a reconstructed pre-gate tree linter caches that die with it - golangci-lint and staticcheck cache issues by package content but store the absolute source path and re-open it to find a suppressing nolint comment - a pre-gate reconstruction is deleted within the minute, so a later run over the same content replayed a stale entry, could not find the nolint, and failed a clean tree - point GOLANGCI_LINT_CACHE and STATICCHECK_CACHE at a scratch sibling of the reconstruction that is removed with it; durable trees keep their shared caches --- docs/tdd/gates-trust.md | 41 +++ internal/engine/dkt1166_test.go | 399 +++++++++++++++++++++++++++++ internal/engine/gate.go | 11 + internal/engine/gate_exec.go | 2 +- internal/engine/pregate.go | 7 + internal/engine/pregate_scratch.go | 37 ++- internal/exec/env.go | 69 +++++ internal/exec/env_cache_test.go | 99 +++++++ 8 files changed, 663 insertions(+), 2 deletions(-) create mode 100644 internal/engine/dkt1166_test.go create mode 100644 internal/exec/env_cache_test.go diff --git a/docs/tdd/gates-trust.md b/docs/tdd/gates-trust.md index 378bb0a4..dd1ad73c 100644 --- a/docs/tdd/gates-trust.md +++ b/docs/tdd/gates-trust.md @@ -925,6 +925,7 @@ the next environment variable anyone invents. | `DOCKET_GATE` | the gate name | so a check can behave differently under docket if its author wants; opaque to core | | `DOCKET_REPO` | the repo root | the same value as `Dir`, for tools that need it in an env | | `DOCKET_GATE_BASE` | the step's base commit sha, **worktree-recorded completion gates only** | so a range-shaped check can scan exactly the step's committed change — `DOCKET_GATE_BASE..HEAD` of the tree it runs in — see below *(added 2026-09-01, DKT-992)* | +| `GOLANGCI_LINT_CACHE`, `STATICCHECK_CACHE` | a scratch directory deleted with the tree, **gates measuring a reconstruction only** | both tools cache issues by package content while storing the absolute path each was found at, and re-open that path to find the `//nolint` that suppresses it. A reconstruction outlives neither, so its entries must not either — see §7.6's DKT-1166 amendment *(added 2026-09-03, DKT-1166)* | **`DOCKET_GATE_BASE` — the step's committed range** *(DKT-992)*. Executors commit **before** `step record`, so at gate time a worktree-recorded step's @@ -1362,6 +1363,46 @@ PG4 is unchanged and still applies: a PRE-gate that could not bind its tree is data for the step's worker, not a park. Parking on it would be the engine judging a step by an input it handed the step itself. +### AMENDMENT (DKT-1166) — a throwaway tree gets throwaway linter caches + +DKT-254 gave a pre-gate the right tree. It did not give it a cache that dies +with that tree, and a class of tool needs exactly that. + +**The mechanism.** golangci-lint and staticcheck cache each reported issue +keyed by **package content**, storing the **absolute path** the issue was found +at, and re-open that path afterwards to look for the `//nolint` (or +`//lint:ignore`) comment that would suppress it. A reconstruction is deleted +within the minute; the content hash is not. So one reconstruction's entries are +replayed in the next, the suppression lookup re-opens a file that is gone, and +an already-suppressed issue is re-emitted as live. + +**Observed**: harness RUN-64/STEP-2939 recorded `ac-commands: fail, exit 2` over +a clean tree. Build and tests exited 0; `make lint` reported one forbidigo issue +at `../docket-pregate-4091742512/…/timelinecompare_test.go` — a directory an +earlier reconstruction had already removed — with golangci-lint warning it could +not read that file, while the source carried `//nolint:forbidigo` on the line +above and the same sha linted in place reported `0 issues.` + +**The cwd was never the problem.** DKT-254 already binds the reconstruction and +`gate_exec.go` already spawns in it, which is why the reported path was +*relative to* the current reconstruction. The carrier is the cache, and it +poisons in both directions: the operator's own persistent cache also receives +entries naming a `docket-pregate-*` path docket is about to delete. + +**The rule.** A gate that measures a tree docket will delete gets its +path-carrying result caches inside a scratch root docket deletes with that tree +(`GOLANGCI_LINT_CACHE`, `STATICCHECK_CACHE` — §5.3). A gate over a tree that +stays on disk keeps its shared caches: re-analysis is a real cost, and it is +only worth paying where the tree is genuinely throwaway. The Go **build** cache +is deliberately untouched — what makes an entry dangerous here is a stored +source path the tool re-opens to decide suppression, and relocating `GOCACHE` +would rebuild the standard library on every reconstruction for no such benefit. + +A cache root that cannot be created **fails the reconstruction**, which records +`skipped`, on the same reasoning as the rest of this section: a measurement that +can report a suppressed issue as live is measuring the wrong thing, and that is +the defect, while measuring nothing is a gap. + ### 7.6.1 Ordering and the claim restructure `ClaimStep` today is **one transaction** (internal/engine/claim.go:73–202): reap, diff --git a/internal/engine/dkt1166_test.go b/internal/engine/dkt1166_test.go new file mode 100644 index 00000000..d0657c4c --- /dev/null +++ b/internal/engine/dkt1166_test.go @@ -0,0 +1,399 @@ +package engine + +import ( + "context" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/ALT-F4-LLC/docket/internal/testsupport" + "github.com/ALT-F4-LLC/docket/internal/trust" + "github.com/ALT-F4-LLC/docket/internal/workflow" +) + +// DKT-1166: a pre-gate measuring a RECONSTRUCTION must not inherit — or leave +// behind — a linter result cache keyed to a directory that is about to be +// deleted. +// +// Harness RUN-64/STEP-2939 recorded `ac-commands: fail, exit 2` on a clean +// tree. Build and tests exited 0; `make lint` reported one forbidigo issue at +// `../docket-pregate-4091742512/internal/tui/screens/timelinecompare_test.go`, +// with golangci-lint warning that it could not read that file. The source +// carries `//nolint:forbidigo` on the line above the call, and the same sha +// linted in an ordinary checkout reports 0 issues. +// +// The cwd was never the problem: pregate.go already binds the reconstruction +// and gate_exec.go already spawns in it. The carrier is golangci-lint's own +// result cache, which keys issues by package CONTENT and stores the absolute +// path each was found at. The content hash outlives a reconstruction; the path +// does not — so an entry written from one reconstruction is replayed in the +// next, its `//nolint` lookup re-opens a file that is gone, and an already +// suppressed issue is re-emitted as live. + +// scratchGateScript writes a gate that records the environment its child +// received, and returns the trust argv plus the file it will write. +// +// `/bin/sh