From 1b4c494bf2680137acc4421b8b4ca8be344ad9d3 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Tue, 15 Sep 2026 14:02:19 -0700 Subject: [PATCH 1/2] perf(statistics): reduce repeated paired gate work --- src/analyst/benchmark-implementation.ts | 2 +- src/statistics/paired-binary.ts | 8 ++++++-- src/statistics/rank-tests.ts | 27 ++++++++++++++++++++++++- 3 files changed, 33 insertions(+), 4 deletions(-) diff --git a/src/analyst/benchmark-implementation.ts b/src/analyst/benchmark-implementation.ts index e33d4adaf..880c4a636 100644 --- a/src/analyst/benchmark-implementation.ts +++ b/src/analyst/benchmark-implementation.ts @@ -139,7 +139,7 @@ export const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([ ]) export const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = - 'e499ff9c5a3b24104d1a9dbc0c3742ffd08b67ff1b852922b423d38231954254' + '53b4c9f453ed05e46968ed39a8e365f8593163b264c0b1a770f7781005d3aae7' export function analystBenchmarkImplementationDigest() { return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 diff --git a/src/statistics/paired-binary.ts b/src/statistics/paired-binary.ts index 6a58e982c..c4c81a17f 100644 --- a/src/statistics/paired-binary.ts +++ b/src/statistics/paired-binary.ts @@ -439,7 +439,11 @@ export function pairedRiskDifferenceScore( // the right endpoint is always inside and the bisection is well posed. let lo = -1 let hi = riskDifference - for (let i = 0; i < 200; i++) { + // 64 halvings resolve the [-1, 1] search interval below 6e-20, past the + // precision available to the score calculation. More iterations only + // repeat identical floating-point values while multiplying every gate + // decision's cost. + for (let i = 0; i < 64; i++) { const mid = (lo + hi) / 2 if (tangoScore(b, c, n, mid) > z) lo = mid else hi = mid @@ -449,7 +453,7 @@ export function pairedRiskDifferenceScore( // Upper bound: root of score(delta) = -z on [riskDifference, 1]. let ulo = riskDifference let uhi = 1 - for (let i = 0; i < 200; i++) { + for (let i = 0; i < 64; i++) { const mid = (ulo + uhi) / 2 if (tangoScore(b, c, n, mid) > -z) ulo = mid else uhi = mid diff --git a/src/statistics/rank-tests.ts b/src/statistics/rank-tests.ts index 91e11ec9a..0ea84f7ef 100644 --- a/src/statistics/rank-tests.ts +++ b/src/statistics/rank-tests.ts @@ -1,7 +1,12 @@ import { ValidationError } from '../errors' import { normalCdf } from '../math/normal' import { lnGamma } from '../math/special-functions' -import { assertFiniteSample, makeRng, symmetricTwoSampleSeed } from './internal' +import { + assertFiniteSample, + binomialSignTwoSided, + makeRng, + symmetricTwoSampleSeed, +} from './internal' // ── Rank tests: exact by default ───────────────────────────────────── // @@ -213,6 +218,26 @@ export function wilcoxonSignedRank( const n = diffs.length if (n === 0) return { w: 0, p: 1, method: 'exact', pFloor: 1, nNonZero: 0 } + // When every non-zero difference has the same magnitude, the signed-rank + // null reduces exactly to a binomial sign distribution. This is the common + // pass/fail shape ({-1, 0, +1}); recognizing it avoids a 100k-draw + // permutation for every repeated gate evaluation while preserving the exact + // p-value and attainable floor. Explicit method requests still retain their + // documented behavior. + const abs = Math.abs(diffs[0]!) + const allEqualAbs = diffs.every((d) => Math.abs(d) === abs) + if ((opts.method === undefined || opts.method === 'auto') && allEqualAbs) { + const positive = diffs.filter((d) => d > 0).length + const negative = n - positive + return { + w: (positive * (n + 1)) / 2, + p: binomialSignTwoSided(positive, negative), + method: 'exact', + pFloor: Math.min(1, 2 ** (1 - n)), + nNonZero: n, + } + } + const order = diffs.map((d, i) => ({ abs: Math.abs(d), i })).sort((x, y) => x.abs - y.abs) const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs)) const ranks: number[] = new Array(n) From 7a9d8eba50c745083bf87e22618658473fe91aaa Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Tue, 15 Sep 2026 14:02:34 -0700 Subject: [PATCH 2/2] refactor(supervisor-run): remove legacy loops reader --- docs/public-api.md | 13 +- src/supervisor-run/analyze.test.ts | 2 +- src/supervisor-run/analyze.ts | 13 +- src/supervisor-run/claude-code-reader.test.ts | 4 +- src/supervisor-run/claude-code-reader.ts | 8 +- src/supervisor-run/fixtures.ts | 2 +- src/supervisor-run/index.ts | 6 +- src/supervisor-run/integrity.test.ts | 1 + src/supervisor-run/loops-reader.test.ts | 215 --------- src/supervisor-run/loops-reader.ts | 440 ------------------ src/supervisor-run/reader.ts | 180 +++++++ src/supervisor-run/render.ts | 4 +- src/supervisor-run/report-command.ts | 4 +- src/supervisor-run/rollout-nodes.ts | 4 +- src/supervisor-run/runtime-fixture.test.ts | 2 +- src/supervisor-run/runtime-r1-fixture.test.ts | 4 +- src/supervisor-run/runtime-reader.test.ts | 6 +- src/supervisor-run/runtime-reader.ts | 13 +- src/supervisor-run/source-facts.ts | 6 +- src/supervisor-run/terminal-record.test.ts | 30 +- src/supervisor-run/terminal-record.ts | 41 +- src/supervisor-run/types.ts | 39 +- 22 files changed, 255 insertions(+), 782 deletions(-) delete mode 100644 src/supervisor-run/loops-reader.test.ts delete mode 100644 src/supervisor-run/loops-reader.ts create mode 100644 src/supervisor-run/reader.ts diff --git a/docs/public-api.md b/docs/public-api.md index 682f8a890..7e69de7a8 100644 --- a/docs/public-api.md +++ b/docs/public-api.md @@ -7,11 +7,11 @@ Generated by `pnpm api:census` on demand — this is a dated reading, not a gate | measure | count | | --- | --- | | export subpaths | 28 | -| published value exports (subpath x symbol) | 1403 | -| distinct symbols | 1231 | +| published value exports (subpath x symbol) | 1400 | +| distinct symbols | 1228 | | production | 944 | -| planned | 218 | -| none | 241 | +| planned | 216 | +| none | 240 | Type-only exports are not listed: a type binds no runtime surface, and removing one cannot break a caller at run time. @@ -1289,7 +1289,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./supervisor-run` -33 value exports — 23 production, 4 planned, 6 none. +30 value exports — 23 production, 2 planned, 5 none. | symbol | consumer | evidence | | --- | --- | --- | @@ -1297,17 +1297,14 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `analyzeSupervisorRunIntegrity` | production | discovery-lab:pursuits/final-agent-authored-campaign-20260810/materialize-candidates.mjs | | `analyzeSupervisorRunSources` | production | discovery-lab:experiments/0005-run-visibility/probe.mjs | | `claudeCodeSupervisorRunReader` | none | only this package's tests: src/supervisor-run/claude-code-reader.test.ts:6 | -| `findSupervisorRunDirIn` | none | only this package's tests: src/supervisor-run/loops-reader.test.ts:11 | | `findSupervisorRunDirs` | production | traces:src/cli.ts | | `isRuntimeSupervisorRunDir` | production | discovery-lab:tools/disco.mjs | | `isUnavailable` | production | discovery-lab:tools/disco.mjs | -| `loopsSupervisorRunReader` | planned | named in traces (bind not in the import graph) | | `NO_SOURCE_LIMITS` | production | traces:src/supervisor-run-context.ts | | `NO_TERMINAL_RECORD` | none | only this package's tests: src/supervisor-run/terminal-record.test.ts:8 | | `parsePatch` | planned | named in browser-agent-driver (bind not in the import graph) | | `parseSupervisorTree` | production | discovery-lab:experiments/0005-run-visibility/probe.mjs | | `readClaudeCodeSupervisorRun` | none | only this package's tests: src/supervisor-run/claude-code-reader.test.ts:6 | -| `readLoopsSupervisorRun` | planned | named in discovery-lab (bind not in the import graph) | | `readRuntimeSupervisorRun` | production | discovery-lab:pursuits/final-agent-authored-campaign-20260810/materialize-candidates.mjs | | `readTerminalRecord` | production | this package: src/supervisor-run/analyze.ts:18 | | `renderSupervisorRollupMarkdown` | production | traces:src/cli.ts | diff --git a/src/supervisor-run/analyze.test.ts b/src/supervisor-run/analyze.test.ts index 6c764a693..eec8da20f 100644 --- a/src/supervisor-run/analyze.test.ts +++ b/src/supervisor-run/analyze.test.ts @@ -334,7 +334,7 @@ describe('analyzeSupervisorRun — decision quality and economics', () => { // UNAVAILABLE != ZERO: an older supervisor wrote no tap, so truncation cannot be ruled out. expect(r.economics.brainTruncations).toEqual({ unavailable: - 'brain.jsonl absent — loops predates the brain-call tap, so truncation cannot be ruled out', + 'brain.jsonl absent — this source has no brain-call tap, so truncation cannot be ruled out', }) }) diff --git a/src/supervisor-run/analyze.ts b/src/supervisor-run/analyze.ts index 6f075cb7e..3d7553756 100644 --- a/src/supervisor-run/analyze.ts +++ b/src/supervisor-run/analyze.ts @@ -2,7 +2,7 @@ * The pure analyzer. Takes already-read bytes (`SupervisorRunSources`) and * returns the report — every metric derivable from a synthetic journal string * with no filesystem, no process, and no network. All I/O lives in a reader - * (`loops-reader.ts` is one). + * (`reader.ts` is one). */ import { summarizeNumberSeries } from '../statistics' @@ -75,16 +75,13 @@ export function analyzeSupervisorRunSources( const journalMissing = src.journalMissingReason ?? - (src.supRunDir === null - ? 'no supervisor run dir under /.agent/supervisor (or legacy /.loops/supervisor)' - : 'journal.jsonl absent') + (src.supRunDir === null ? 'no Runtime supervisor run directory' : 'journal.jsonl absent') const haveJournal = src.journal !== null const tree = parseSupervisorTree(src) const state = tree.state const result = parseJson(src.result) const judge = parseJson(src.judge) - // Runtime's settle record outranks the legacy loops documents; the record - // that answered is named on the report so a status never arrives unlabeled. + // Runtime's terminal record is named on the report so a status never arrives unlabeled. const terminal = readTerminalRecord({ state, result, @@ -767,8 +764,8 @@ export function analyzeSupervisorRunSources( 'brain.brainTruncations', src.brainLogMissingReason ?? (src.supRunDir === null - ? 'no supervisor run dir under /.agent/supervisor (or legacy /.loops/supervisor)' - : 'brain.jsonl absent — loops predates the brain-call tap, so truncation cannot be ruled out'), + ? 'no Runtime supervisor run directory' + : 'brain.jsonl absent — this source has no brain-call tap, so truncation cannot be ruled out'), ) : brainCalls.filter((c) => c.finish_reason === 'length').length, workers: { diff --git a/src/supervisor-run/claude-code-reader.test.ts b/src/supervisor-run/claude-code-reader.test.ts index 153b80967..b1d39b8ed 100644 --- a/src/supervisor-run/claude-code-reader.test.ts +++ b/src/supervisor-run/claude-code-reader.test.ts @@ -397,7 +397,7 @@ describe('claudeCodeSupervisorRunReader', () => { // A worker with no retained transcript reports null tokens, not 0. const pruned = workers.find((w) => w.artifacts.transcript_ref === null) expect(pruned?.cost.tokens_in).toBeNull() - // The root row points at the real session transcript, not a loops path. + // The root row points at the real session transcript, not a Runtime path. expect(root?.artifacts.transcript_ref).toBe(s.transcriptPath) }) @@ -411,7 +411,7 @@ describe('claudeCodeSupervisorRunReader', () => { expect(isUnavailable(report.orchestration.steers)).toBe(true) }) - it('exposes the SupervisorRunReader contract loops implements', async () => { + it('exposes the SupervisorRunReader contract Runtime implements', async () => { const s = await writeSession() const reader = claudeCodeSupervisorRunReader({ transcriptPath: s.transcriptPath, diff --git a/src/supervisor-run/claude-code-reader.ts b/src/supervisor-run/claude-code-reader.ts index 0c11954d7..f8732ed29 100644 --- a/src/supervisor-run/claude-code-reader.ts +++ b/src/supervisor-run/claude-code-reader.ts @@ -1,7 +1,7 @@ /** * Supervision-tree reader over a THIRD-PARTY harness: Claude Code. * - * `loops-reader.ts` reads a supervisor we wrote, whose journal was designed + * The Runtime reader reads a supervisor journal designed for this analysis. * for this analysis. This reader reads a harness we do not control, whose * transcript was designed for replaying a chat — and recovers the same tree * from it. If both produce a `SupervisorRunSources`, the tree model is a @@ -31,12 +31,12 @@ * reports `unavailable — ` instead of the $0 / 0-accepted that summing * an empty field would produce. See `SourceLimits`. * - * ## Metric coverage vs the loops journal + * ## Metric coverage vs the Runtime journal * * Measured on a real 52-agent session (fixture: * `tests/fixtures/supervisor-run/claude-code-session-*`). * - * | Metric | loops | Claude Code | Why | + * | Metric | Runtime | Claude Code | Why | * |---|---|---|---| * | workersSpawned / Settled / Cancelled | full | full | spawn tool_use + task-notification + TaskStop | * | steers / steersDelivered / steersByWorker | full | full | `SendMessage`; delivery from its tool_result | @@ -642,7 +642,7 @@ export async function readClaudeCodeSupervisorRun( } } -/** A `SupervisorRunReader` over a Claude Code session — the same contract loops implements. */ +/** A `SupervisorRunReader` over a Claude Code session — the same contract Runtime implements. */ export function claudeCodeSupervisorRunReader(opts: ClaudeCodeReaderOptions): SupervisorRunReader { return { runRef: opts.runRef ?? opts.transcriptPath, diff --git a/src/supervisor-run/fixtures.ts b/src/supervisor-run/fixtures.ts index b61fcb2b2..c5df1dc6d 100644 --- a/src/supervisor-run/fixtures.ts +++ b/src/supervisor-run/fixtures.ts @@ -192,7 +192,7 @@ export function fixtureSources(over: Partial = {}): Superv runRef: '/tmp/cell/runs/inst-1/ARM', instanceId: 'inst-1', arm: 'ARM', - supRunDir: '/tmp/cell/ws/.loops/supervisor/sup-1-test', + supRunDir: '/tmp/cell/runtime-run', journal: null, brainLog: null, state: null, diff --git a/src/supervisor-run/index.ts b/src/supervisor-run/index.ts index 336b01688..b2ae1071f 100644 --- a/src/supervisor-run/index.ts +++ b/src/supervisor-run/index.ts @@ -45,16 +45,12 @@ export { } from './integrity' export { analyzeSupervisorRun, - findSupervisorRunDirIn, findSupervisorRunDirs, - type LoopsReaderOptions, - loopsSupervisorRunReader, - readLoopsSupervisorRun, reportSupervisorRound, type WriteSupervisorRunOptions, writeSupervisorRunReport, writeSupervisorRunReportSafe, -} from './loops-reader' +} from './reader' export { renderSupervisorRollupMarkdown, renderSupervisorRunHeadline, diff --git a/src/supervisor-run/integrity.test.ts b/src/supervisor-run/integrity.test.ts index fdba4f464..4d0e6d861 100644 --- a/src/supervisor-run/integrity.test.ts +++ b/src/supervisor-run/integrity.test.ts @@ -79,6 +79,7 @@ function completedControlSource(inbox: string | null, events: string | null): Su return fixtureSources({ journal: fixtureJournal({ workers: [['worker', 1, 4]] }), state: fixtureState({ startSec: 0, endSec: 5 }), + result: JSON.stringify({ kind: 'winner', tree: { root: 'sup-1-test', nodes: [] } }), workers: [ { workerId: 'sup-1-test:s0', diff --git a/src/supervisor-run/loops-reader.test.ts b/src/supervisor-run/loops-reader.test.ts deleted file mode 100644 index 08fd7b623..000000000 --- a/src/supervisor-run/loops-reader.test.ts +++ /dev/null @@ -1,215 +0,0 @@ -import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { describe, expect, it } from 'vitest' -import { analyzeSupervisorRunSources } from './analyze' -import { - fixtureJournal as journal, - fixtureState as state, - fixtureWorker as worker, -} from './fixtures' -import { - analyzeSupervisorRun, - findSupervisorRunDirIn, - findSupervisorRunDirs, - loopsSupervisorRunReader, - readLoopsSupervisorRun, - writeSupervisorRunReport, -} from './loops-reader' -import { renderSupervisorRunMarkdown } from './render' -import { isUnavailable, SUPERVISOR_RUN_SCHEMA } from './types' - -async function makeRun( - opts: { steers?: boolean; withJudge?: boolean; stateDir?: string } = {}, -): Promise { - const root = await mkdtemp(join(tmpdir(), 'supervisor-run-test-')) - const runDir = join(root, 'runs', 'inst-9', 'ARM') - const sup = join(runDir, 'ws', opts.stateDir ?? '.agent', 'supervisor', 'sup-1-abc') - await mkdir(join(sup, 'workers'), { recursive: true }) - await writeFile( - join(sup, 'journal.jsonl'), - journal({ - workers: [ - ['a', 10, 100], - ['b', 150, 220], - ], - }), - ) - await writeFile(join(sup, 'state.json'), state({ startSec: 0, endSec: 300 })) - const w0 = worker('w-0', { - startSec: 10, - finishSec: 100, - passed: true, - patchBytes: 40, - ...(opts.steers ? { steers: ['refocus'] } : {}), - }) - await writeFile(join(sup, 'workers', 'w-0.ndjson'), w0.events ?? '') - if (w0.inbox !== null) await writeFile(join(sup, 'workers', 'w-0.inbox.ndjson'), w0.inbox) - await writeFile(join(sup, 'workers', 'w-0.patch'), 'x'.repeat(40)) - const patchPath = join(root, 'delivered.patch') - await writeFile(patchPath, '+++ b/src/a.ts\n+line\n') - await writeFile( - join(runDir, 'result.json'), - JSON.stringify({ iid: 'inst-9', arm: 'ARM', verify_pass: true, verify_rc: 0, patchPath }), - ) - if (opts.withJudge) { - await writeFile(join(runDir, 'judge.json'), JSON.stringify({ resolved: true, score: 1 })) - } - await writeFile(join(runDir, 'driver.log'), '[driver] registered tools: supervisor_steer\n') - return runDir -} - -describe('loops reader + writeSupervisorRunReport over a real run directory', () => { - it('reads a legacy pre-rename run under ws/.loops via fallback', async () => { - const runDir = await makeRun({ stateDir: '.loops' }) - const src = await readLoopsSupervisorRun(runDir, { opencodeDb: null }) - expect(src.supRunDir).toBe(join(runDir, 'ws', '.loops', 'supervisor', 'sup-1-abc')) - const report = analyzeSupervisorRunSources(src) - expect(isUnavailable(report.orchestration.workersSpawned)).toBe(false) - }) - - it('prefers ws/.agent when both state dirs hold a run', async () => { - const runDir = await makeRun() - await mkdir(join(runDir, 'ws', '.loops', 'supervisor', 'sup-0-old'), { recursive: true }) - expect(await findSupervisorRunDirIn(join(runDir, 'ws'))).toBe( - join(runDir, 'ws', '.agent', 'supervisor', 'sup-1-abc'), - ) - }) - - it('reads a run end to end and writes both artifacts + the log headline', async () => { - const runDir = await makeRun({ steers: true, withJudge: true }) - const logPath = join(runDir, '..', '..', '..', 'run.log') - await writeFile(logPath, '') - const report = await writeSupervisorRunReport(runDir, { - opencodeDb: null, - appendHeadlineTo: logPath, - echo: false, - }) - - expect(report.orchestration.steers).toBe(1) - expect(report.orchestration.workersSpawned).toBe(2) - expect(report.outcome.judgeResolved).toBe(true) - expect(report.outcome.verifyPass).toBe(true) - expect(report.instanceId).toBe('inst-9') - - const json = JSON.parse(await readFile(join(runDir, 'run-report.json'), 'utf8')) as { - orchestration: { steers: number } - } - expect(json.orchestration.steers).toBe(1) - const md = await readFile(join(runDir, 'run-report.md'), 'utf8') - expect(md).toContain('# Run report — inst-9 [ARM]') - const log = await readFile(logPath, 'utf8') - expect(log).toContain('RUN-REPORT inst-9 [ARM]') - expect(log).toContain('steers=1 queued / 1 delivered') - }) - - it('reports a steer-free run as 0, and falls back to the ledger for the verdict', async () => { - const runDir = await makeRun() - const ledger = join(runDir, '..', '..', '..', 'ledger.jsonl') - await writeFile( - ledger, - `${JSON.stringify({ iid: 'inst-9', arm: 'ARM', resolved: false, score: 0.43, passed: 13, total: 30 })}\n`, - ) - const src = await readLoopsSupervisorRun(runDir, { opencodeDb: null, ledgerPath: ledger }) - const report = analyzeSupervisorRunSources(src) - expect(report.orchestration.steers).toBe(0) - expect(report.outcome.judgeResolved).toBe(false) - expect(report.outcome.judgeScore).toBe(0.43) - expect(report.outcome.judgeSource).toContain('ledger row') - }) - - it('marks worker token totals unavailable when worker artifacts cannot supply join keys', async () => { - const runDir = await makeRun() - const supervisorDir = join(runDir, 'ws', '.agent', 'supervisor', 'sup-1-abc') - await rm(join(supervisorDir, 'workers'), { recursive: true, force: true }) - - const src = await readLoopsSupervisorRun(runDir, { opencodeDb: null }) - expect(src.limits.workerTokens).toMatch(/2\/2 worker invocations have no clone cwd/) - const report = analyzeSupervisorRunSources(src) - expect(isUnavailable(report.economics.workers.tokensIn)).toBe(true) - }) - - it('binds a label file to an id only when every started attempt names the same id', async () => { - const runDir = await makeRun() - const workerPath = join( - runDir, - 'ws', - '.agent', - 'supervisor', - 'sup-1-abc', - 'workers', - 'w-0.ndjson', - ) - await writeFile( - workerPath, - `${[ - { kind: 'started', workerId: 'worker-a', cwd: '/tmp/worker-a' }, - { kind: 'finished', passed: true }, - { kind: 'started', workerId: 'worker-b', cwd: '/tmp/worker-b' }, - { kind: 'finished', passed: false }, - ] - .map((event) => JSON.stringify(event)) - .join('\n')}\n`, - ) - - const src = await readLoopsSupervisorRun(runDir, { opencodeDb: null }) - expect(src.workers?.[0]?.workerId).toBeUndefined() - }) - - it('honours reportDir so a READ-ONLY run directory is never written to', async () => { - const runDir = await makeRun() - const dest = await mkdtemp(join(tmpdir(), 'supervisor-run-out-')) - await writeSupervisorRunReport(runDir, { opencodeDb: null, reportDir: dest, echo: false }) - await expect(readFile(join(runDir, 'run-report.json'), 'utf8')).rejects.toThrow() - const written = await readFile(join(dest, 'inst-9.ARM.json'), 'utf8') - expect(JSON.parse(written)).toMatchObject({ schema: SUPERVISOR_RUN_SCHEMA }) - }) - - it('produces an all-unavailable report for a run directory with no artifacts at all', async () => { - const root = await mkdtemp(join(tmpdir(), 'supervisor-run-empty-')) - const runDir = join(root, 'runs', 'inst-0', 'ARM') - await mkdir(runDir, { recursive: true }) - const report = await writeSupervisorRunReport(runDir, { opencodeDb: null, echo: false }) - expect(isUnavailable(report.orchestration.steers)).toBe(true) - expect(isUnavailable(report.orchestration.workersSpawned)).toBe(true) - expect(report.gaps.length).toBeGreaterThan(3) - expect(renderSupervisorRunMarkdown(report)).toContain('unavailable —') - }) - - it('discovers every runs// run directory under an out dir', async () => { - const runDir = await makeRun() - const outDir = join(runDir, '..', '..', '..') - const found = await findSupervisorRunDirs(outDir) - expect(found).toHaveLength(1) - expect(found[0]).toContain('/runs/inst-9/ARM') - }) -}) - -describe('analyzeSupervisorRun input contract', () => { - it('accepts a run dir, a reader, and already-read sources interchangeably', async () => { - const runDir = await makeRun({ steers: true }) - const fromDir = await analyzeSupervisorRun(runDir, { opencodeDb: null }) - const fromReader = await analyzeSupervisorRun( - loopsSupervisorRunReader(runDir, { opencodeDb: null }), - ) - const fromSources = await analyzeSupervisorRun( - await readLoopsSupervisorRun(runDir, { opencodeDb: null }), - ) - for (const r of [fromReader, fromSources]) { - expect({ ...r, generatedAt: '' }).toEqual({ ...fromDir, generatedAt: '' }) - } - }) - - it('serves a custom reader that never touches the filesystem', async () => { - const sources = await readLoopsSupervisorRun(await makeRun({ steers: true }), { - opencodeDb: null, - }) - const inMemory = { - runRef: 'memory://run-1', - read: async () => ({ ...sources, runRef: 'memory://run-1' }), - } - const report = await analyzeSupervisorRun(inMemory) - expect(report.runRef).toBe('memory://run-1') - expect(report.orchestration.steers).toBe(1) - }) -}) diff --git a/src/supervisor-run/loops-reader.ts b/src/supervisor-run/loops-reader.ts deleted file mode 100644 index d5032ed54..000000000 --- a/src/supervisor-run/loops-reader.ts +++ /dev/null @@ -1,440 +0,0 @@ -/** - * ONE implementation of `SupervisorRunReader`: the on-disk layout the loops - * supervisor writes — `/ws/.agent/supervisor//{journal.jsonl, - * state.json, progress.ndjson, workers/*.ndjson}` alongside the run's - * `result.json` / `judge.json` / `driver.log` / delivered patch. Runs written - * before the `.agent` rename live under `/.loops/supervisor/` and are - * still found via fallback. - * - * Nothing in `analyze.ts` knows this layout exists. A different store (an - * archive, an object bucket, a database) implements the same interface and - * gets the same report. - * - * Worker token recovery reuses the rollout module's opencode reader rather - * than opening a second sqlite path — one store client, one corruption policy. - */ - -import { appendFile, mkdir, readdir, readFile, writeFile } from 'node:fs/promises' -import { basename, join } from 'node:path' -import { - DEFAULT_OPENCODE_DB, - findOpencodeSessionsByDirectory, - openOpencodeDb, -} from '../rollout/readers/opencode-sqlite' -import { analyzeSupervisorRunSources, parseJson, parseJsonl, rollupSupervisorRuns } from './analyze' -import { - renderSupervisorRollupMarkdown, - renderSupervisorRunHeadline, - renderSupervisorRunMarkdown, -} from './render' -import { isRuntimeSupervisorRunDir, readRuntimeSupervisorRun } from './runtime-reader' -import { - NO_SOURCE_LIMITS, - type SupervisorRunReader, - type SupervisorRunReport, - type SupervisorRunRollup, - type SupervisorRunSources, - type WorkerLogSource, -} from './types' - -async function readMaybe(path: string): Promise { - return readFile(path, 'utf8').catch(() => null) -} - -/** - * Locate the (single) supervisor run dir under `/.agent/supervisor`, falling back to the - * pre-rename `/.loops/supervisor` so runs written by older supervisors stay analyzable. - */ -export async function findSupervisorRunDirIn(ws: string): Promise { - for (const stateDir of ['.agent', '.loops']) { - const root = join(ws, stateDir, 'supervisor') - const entries = await readdir(root, { withFileTypes: true }).catch(() => []) - const dirs = entries.filter((e) => e.isDirectory()).map((e) => join(root, e.name)) - if (dirs[0] !== undefined) return dirs[0] - } - return null -} - -export interface LoopsReaderOptions { - /** Override the workspace dir (default `/ws`). */ - readonly ws?: string - /** Delivered patch path (default: `patchPath` from result.json). */ - readonly patchPath?: string - /** opencode sqlite store; set to `null` to skip the worker-token join entirely. */ - readonly opencodeDb?: string | null - /** Ledger to fall back to when the run has no `judge.json` (matched on iid + arm + runDir). */ - readonly ledgerPath?: string -} - -/** - * Read a loops supervisor run directory into source bytes. Never throws on a - * missing artifact — an absent file becomes a `null` field, which is what makes - * the dependent metric `unavailable` instead of 0. - */ -export async function readLoopsSupervisorRun( - runDir: string, - opts: LoopsReaderOptions = {}, -): Promise { - const ws = opts.ws ?? join(runDir, 'ws') - const supRunDir = await findSupervisorRunDirIn(ws) - const result = await readMaybe(join(runDir, 'result.json')) - const resultObj = parseJson(result) - const journal = supRunDir === null ? null : await readMaybe(join(supRunDir, 'journal.jsonl')) - const journalWorkerSpawns = parseJsonl(journal).filter( - (event) => - event.kind === 'spawned' && typeof event.parent === 'string' && event.role !== 'supervisor', - ).length - - let workers: WorkerLogSource[] | null = null - let workersMissingReason: string | null = null - const workerCwds: string[] = [] - let workerStarts = 0 - if (supRunDir === null) { - workersMissingReason = `no supervisor run dir under ${join(ws, '.agent', 'supervisor')} (or legacy ${join(ws, '.loops', 'supervisor')})` - } else { - const workersDir = join(supRunDir, 'workers') - const entries = await readdir(workersDir).catch(() => null) - if (entries === null) { - workersMissingReason = `workers/ directory absent under ${supRunDir}` - } else { - const labels = [ - ...new Set( - entries - .filter((f) => f.endsWith('.ndjson')) - .map((f) => f.replace(/\.inbox\.ndjson$/, '').replace(/\.ndjson$/, '')), - ), - ].sort() - workers = [] - for (const label of labels) { - const events = await readMaybe(join(workersDir, `${label}.ndjson`)) - const inbox = await readMaybe(join(workersDir, `${label}.inbox.ndjson`)) - const patch = await readMaybe(join(workersDir, `${label}.patch`)) - const eventRows = parseJsonl(events) - const startedRows = eventRows.filter((event) => event.kind === 'started') - workerStarts += startedRows.length - const startedIds = startedRows - .map((event) => - typeof event.workerId === 'string' - ? event.workerId - : typeof event.agentId === 'string' - ? event.agentId - : null, - ) - .filter((id): id is string => id !== null) - const distinctStartedIds = new Set(startedIds) - const workerId = - startedRows.length > 0 && - startedIds.length === startedRows.length && - distinctStartedIds.size === 1 - ? startedIds[0] - : undefined - workers.push({ - ...(workerId === undefined ? {} : { workerId }), - label, - events, - inbox, - patchBytes: patch === null ? null : Buffer.byteLength(patch), - }) - for (const ev of startedRows) { - if (ev.kind === 'started' && typeof ev.cwd === 'string') workerCwds.push(ev.cwd) - } - } - } - } - - let harnessWorkerTokens: SupervisorRunSources['harnessWorkerTokens'] = null - let harnessMissingReason: string | null = null - let workerCwdsWithoutSessions = 0 - if (opts.opencodeDb === null) { - harnessMissingReason = 'opencode join disabled' - } else if (workerCwds.length === 0) { - harnessMissingReason = 'no worker clone cwds in workers/*.ndjson (nothing to join)' - } else { - const db = await openOpencodeDb(opts.opencodeDb ?? DEFAULT_OPENCODE_DB) - if (db === null) { - harnessMissingReason = `opencode session store unreadable at ${opts.opencodeDb ?? DEFAULT_OPENCODE_DB}` - } else { - try { - const seen = new Set() - let sessions = 0 - let input = 0 - let output = 0 - const distinctWorkerCwds = new Set(workerCwds) - for (const cwd of distinctWorkerCwds) { - const rows = findOpencodeSessionsByDirectory(db, cwd) - if (rows.length === 0) workerCwdsWithoutSessions += 1 - for (const row of rows) { - if (seen.has(row.id)) continue - seen.add(row.id) - sessions += 1 - input += row.tokensInput - output += row.tokensOutput + row.tokensReasoning - } - } - harnessWorkerTokens = { store: 'opencode', sessions, input, output } - if (workerCwdsWithoutSessions > 0) { - harnessMissingReason = `${workerCwdsWithoutSessions}/${distinctWorkerCwds.size} worker clone cwds have no opencode session` - } - } finally { - db.close() - } - } - } - - const workerInvocations = Math.max(journalWorkerSpawns, workerStarts) - const workerTokenGaps: string[] = [] - if (workerCwds.length < workerInvocations) { - workerTokenGaps.push( - `${workerInvocations - workerCwds.length}/${workerInvocations} worker invocations have no clone cwd for the opencode token join`, - ) - } - if (workerInvocations > 0 && harnessWorkerTokens === null) { - workerTokenGaps.push(harnessMissingReason ?? 'worker harness token join unavailable') - } else if (workerCwdsWithoutSessions > 0 && harnessMissingReason !== null) { - workerTokenGaps.push(harnessMissingReason) - } - const workerTokenLimit = workerTokenGaps.length === 0 ? null : workerTokenGaps.join('; ') - - const patchPath = - opts.patchPath ?? (typeof resultObj?.patchPath === 'string' ? resultObj.patchPath : null) - - let judge = await readMaybe(join(runDir, 'judge.json')) - let judgeSource = judge === null ? null : join(runDir, 'judge.json') - if (judge === null && opts.ledgerPath !== undefined) { - const row = await findLedgerRow(opts.ledgerPath, runDir) - if (row !== null) { - judge = JSON.stringify(row) - judgeSource = `${opts.ledgerPath} (ledger row)` - } - } - - return { - runRef: runDir, - instanceId: typeof resultObj?.iid === 'string' ? resultObj.iid : instanceIdFromPath(runDir), - arm: typeof resultObj?.arm === 'string' ? resultObj.arm : basename(runDir), - supRunDir, - journal, - brainLog: supRunDir === null ? null : await readMaybe(join(supRunDir, 'brain.jsonl')), - state: supRunDir === null ? null : await readMaybe(join(supRunDir, 'state.json')), - progress: supRunDir === null ? null : await readMaybe(join(supRunDir, 'progress.ndjson')), - workers, - workersMissingReason, - result, - judge, - judgeSource, - patch: patchPath === null ? null : await readMaybe(patchPath), - driverLog: await readMaybe(join(runDir, 'driver.log')), - harnessWorkerTokens, - harnessMissingReason, - // loops prices its own inference, runs a verify per worker, and keeps each - // worker's patch. A missing external-harness join is declared explicitly. - limits: { - ...NO_SOURCE_LIMITS, - workerTokens: workerTokenLimit, - }, - traceCommand: null, - } -} - -/** The loops on-disk layout, as a `SupervisorRunReader`. */ -export function loopsSupervisorRunReader( - runDir: string, - opts: LoopsReaderOptions = {}, -): SupervisorRunReader { - return { runRef: runDir, read: () => readLoopsSupervisorRun(runDir, opts) } -} - -/** The ledger row whose `runDir` is this run (falling back to iid + arm match). */ -async function findLedgerRow( - ledgerPath: string, - runDir: string, -): Promise | null> { - const rows = parseJsonl(await readMaybe(ledgerPath)) - const exact = rows.find((r) => r.runDir === runDir) - if (exact !== undefined) return exact - const iid = instanceIdFromPath(runDir) - const arm = basename(runDir) - return rows.find((r) => r.iid === iid && r.arm === arm) ?? null -} - -/** `/runs//` → ``. */ -function instanceIdFromPath(runDir: string): string | null { - const parts = runDir.split('/').filter(Boolean) - const armIdx = parts.length - 1 - const iid = parts[armIdx - 1] - return parts[armIdx - 2] === 'runs' && iid !== undefined ? iid : null -} - -// --------------------------------------------------------------------------- -// Entry point + write helpers. -// --------------------------------------------------------------------------- - -/** - * Analyze a supervisor run. Accepts a run directory (read through the loops - * reader), any `SupervisorRunReader`, or already-read source bytes — so a - * caller with its own store never has to touch the filesystem layout. - */ -export async function analyzeSupervisorRun( - input: string | SupervisorRunReader | SupervisorRunSources, - opts: LoopsReaderOptions = {}, -): Promise { - if (typeof input === 'string') { - return analyzeSupervisorRunSources( - (await isRuntimeSupervisorRunDir(input)) - ? await readRuntimeSupervisorRun(input) - : await readLoopsSupervisorRun(input, opts), - ) - } - if (isReader(input)) return analyzeSupervisorRunSources(await input.read()) - return analyzeSupervisorRunSources(input) -} - -function isReader(input: SupervisorRunReader | SupervisorRunSources): input is SupervisorRunReader { - return typeof (input as SupervisorRunReader).read === 'function' -} - -export interface WriteSupervisorRunOptions extends LoopsReaderOptions { - /** Append the headline block here (the experiment's run log). */ - readonly appendHeadlineTo?: string - /** Also console.log the headline (default true). */ - readonly echo?: boolean - /** - * Write `run-report.{json,md}` here instead of into the run dir. Set when - * reporting over a run directory that must stay READ-ONLY (a live run, an - * archived generation). - */ - readonly reportDir?: string -} - -/** - * Read a completed run, write `run-report.json` + `run-report.md` beside its - * artifacts, and append the headline block to the run log. Never throws on a - * missing artifact — a run that produced nothing still yields a report whose - * every metric says why. - */ -export async function writeSupervisorRunReport( - runDir: string, - opts: WriteSupervisorRunOptions = {}, -): Promise { - const sources = (await isRuntimeSupervisorRunDir(runDir)) - ? await readRuntimeSupervisorRun(runDir) - : await readLoopsSupervisorRun(runDir, opts) - const report = analyzeSupervisorRunSources(sources) - const md = renderSupervisorRunMarkdown(report) - const dest = opts.reportDir ?? runDir - const stem = opts.reportDir === undefined ? 'run-report' : supervisorReportStem(runDir) - if (opts.reportDir !== undefined) await mkdir(opts.reportDir, { recursive: true }).catch(() => {}) - await writeFile(join(dest, `${stem}.json`), JSON.stringify(report, null, 1)).catch(() => {}) - await writeFile(join(dest, `${stem}.md`), md).catch(() => {}) - const headline = renderSupervisorRunHeadline(report) - if (opts.appendHeadlineTo !== undefined) { - await appendFile(opts.appendHeadlineTo, `${headline}\n`).catch(() => {}) - } - if (opts.echo !== false) console.log(headline) - return report -} - -/** - * File stem for out-of-tree reports. Built from the run path's identifying - * segments — candidate tag (the segment under `arm-runs/`), rep, instance, arm - * — so two runs of the same instance from different candidates/reps never - * overwrite each other. - */ -function supervisorReportStem(runDir: string): string { - const parts = runDir.split('/').filter(Boolean) - const arm = parts[parts.length - 1] ?? 'cell' - const iid = parts[parts.length - 2] ?? 'instance' - const rep = parts.find((p) => /^rep-\d+$/.test(p)) - const armRunsIdx = parts.indexOf('arm-runs') - const tag = armRunsIdx >= 0 ? parts[armRunsIdx + 1] : undefined - return [tag, rep, iid, arm] - .filter((s): s is string => s !== undefined && s !== 'runs') - .join('.') - .replace(/[^A-Za-z0-9._-]/g, '_') -} - -/** - * Best-effort wrapper for a hot path: a reporting failure must never kill a run - * that already produced real work. Returns null and logs the reason instead. - */ -export async function writeSupervisorRunReportSafe( - runDir: string, - opts: WriteSupervisorRunOptions = {}, -): Promise { - try { - return await writeSupervisorRunReport(runDir, opts) - } catch (err) { - console.log( - `RUN-REPORT failed for ${runDir}: ${err instanceof Error ? err.message : String(err)}`, - ) - return null - } -} - -/** - * Report every run under an experiment `outDir` (any depth of - * `runs//`), write each run's report, and write the rollup at - * `/run-report-round.{json,md}`. - */ -export async function reportSupervisorRound( - outDir: string, - opts: WriteSupervisorRunOptions & { title?: string } = {}, -): Promise { - const runDirs = await findSupervisorRunDirs(outDir) - const reports: SupervisorRunReport[] = [] - for (const runDir of runDirs) { - const r = await writeSupervisorRunReportSafe(runDir, { ...opts, echo: opts.echo ?? false }) - if (r !== null) reports.push(r) - } - const rollup = rollupSupervisorRuns(reports) - const md = renderSupervisorRollupMarkdown( - rollup, - opts.title ?? `Round rollup — ${basename(outDir)}`, - ) - const dest = opts.reportDir ?? outDir - if (opts.reportDir !== undefined) await mkdir(opts.reportDir, { recursive: true }).catch(() => {}) - await writeFile(join(dest, 'run-report-round.json'), JSON.stringify(rollup, null, 1)).catch( - () => {}, - ) - await writeFile(join(dest, 'run-report-round.md'), md).catch(() => {}) - if (opts.appendHeadlineTo !== undefined) { - await appendFile(opts.appendHeadlineTo, `${md}\n`).catch(() => {}) - } - if (opts.echo !== false) console.log(md) - return rollup -} - -/** - * Every loops or Runtime supervisor run below `root`. - * - * When `root` itself is one run, return no children so callers can distinguish - * a single report from a parent-directory rollup. - */ -export async function findSupervisorRunDirs(root: string): Promise { - if ( - (await isRuntimeSupervisorRunDir(root)) || - (await findSupervisorRunDirIn(join(root, 'ws'))) !== null - ) { - return [] - } - const found: string[] = [] - const walk = async (dir: string, depth: number): Promise => { - if (depth > 8) return - const entries = await readdir(dir, { withFileTypes: true }).catch(() => []) - for (const e of entries) { - if (!e.isDirectory()) continue - if (e.name === 'node_modules' || e.name === '.git') continue - const full = join(dir, e.name) - if ( - (await isRuntimeSupervisorRunDir(full)) || - (await findSupervisorRunDirIn(join(full, 'ws'))) !== null - ) { - found.push(full) - continue - } - await walk(full, depth + 1) - } - } - await walk(root, 0) - return found.sort() -} diff --git a/src/supervisor-run/reader.ts b/src/supervisor-run/reader.ts new file mode 100644 index 000000000..2b574d3d6 --- /dev/null +++ b/src/supervisor-run/reader.ts @@ -0,0 +1,180 @@ +/** + * Runtime supervisor-run reader and report writers. + * + * Runtime's file-backed supervision context is the only live on-disk contract. + * Other stores implement `SupervisorRunReader` and feed the same pure analyzer; + * this module owns path discovery and report output for Runtime runs. + */ + +import { appendFile, mkdir, readdir, writeFile } from 'node:fs/promises' +import { basename, join } from 'node:path' +import { analyzeSupervisorRunSources, rollupSupervisorRuns } from './analyze' +import { + renderSupervisorRollupMarkdown, + renderSupervisorRunHeadline, + renderSupervisorRunMarkdown, +} from './render' +import { isRuntimeSupervisorRunDir, readRuntimeSupervisorRun } from './runtime-reader' +import type { + SupervisorRunReader, + SupervisorRunReport, + SupervisorRunRollup, + SupervisorRunSources, +} from './types' + +/** + * Analyze a supervisor run. Accepts a Runtime run directory, any + * `SupervisorRunReader`, or already-read source bytes — so a + * caller with its own store never has to touch the filesystem layout. + */ +export async function analyzeSupervisorRun( + input: string | SupervisorRunReader | SupervisorRunSources, +): Promise { + if (typeof input === 'string') { + return analyzeSupervisorRunSources(await readRuntimeSupervisorRun(input)) + } + if (isReader(input)) return analyzeSupervisorRunSources(await input.read()) + return analyzeSupervisorRunSources(input) +} + +function isReader(input: SupervisorRunReader | SupervisorRunSources): input is SupervisorRunReader { + return typeof (input as SupervisorRunReader).read === 'function' +} + +export interface WriteSupervisorRunOptions { + /** Append the headline block here (the experiment's run log). */ + readonly appendHeadlineTo?: string + /** Also console.log the headline (default true). */ + readonly echo?: boolean + /** + * Write `run-report.{json,md}` here instead of into the run dir. Set when + * reporting over a run directory that must stay READ-ONLY (a live run, an + * archived generation). + */ + readonly reportDir?: string +} + +/** + * Read a completed run, write `run-report.json` + `run-report.md` beside its + * artifacts, and append the headline block to the run log. Never throws on a + * missing artifact — a run that produced nothing still yields a report whose + * every metric says why. + */ +export async function writeSupervisorRunReport( + runDir: string, + opts: WriteSupervisorRunOptions = {}, +): Promise { + const sources = await readRuntimeSupervisorRun(runDir) + const report = analyzeSupervisorRunSources(sources) + const md = renderSupervisorRunMarkdown(report) + const dest = opts.reportDir ?? runDir + const stem = opts.reportDir === undefined ? 'run-report' : supervisorReportStem(runDir) + if (opts.reportDir !== undefined) await mkdir(opts.reportDir, { recursive: true }).catch(() => {}) + await writeFile(join(dest, `${stem}.json`), JSON.stringify(report, null, 1)).catch(() => {}) + await writeFile(join(dest, `${stem}.md`), md).catch(() => {}) + const headline = renderSupervisorRunHeadline(report) + if (opts.appendHeadlineTo !== undefined) { + await appendFile(opts.appendHeadlineTo, `${headline}\n`).catch(() => {}) + } + if (opts.echo !== false) console.log(headline) + return report +} + +/** + * File stem for out-of-tree reports. Built from the run path's identifying + * segments — candidate tag (the segment under `arm-runs/`), rep, instance, arm + * — so two runs of the same instance from different candidates/reps never + * overwrite each other. + */ +function supervisorReportStem(runDir: string): string { + const parts = runDir.split('/').filter(Boolean) + const arm = parts[parts.length - 1] ?? 'cell' + const iid = parts[parts.length - 2] ?? 'instance' + const rep = parts.find((p) => /^rep-\d+$/.test(p)) + const armRunsIdx = parts.indexOf('arm-runs') + const tag = armRunsIdx >= 0 ? parts[armRunsIdx + 1] : undefined + return [tag, rep, iid, arm] + .filter((s): s is string => s !== undefined && s !== 'runs') + .join('.') + .replace(/[^A-Za-z0-9._-]/g, '_') +} + +/** + * Best-effort wrapper for a hot path: a reporting failure must never kill a run + * that already produced real work. Returns null and logs the reason instead. + */ +export async function writeSupervisorRunReportSafe( + runDir: string, + opts: WriteSupervisorRunOptions = {}, +): Promise { + try { + return await writeSupervisorRunReport(runDir, opts) + } catch (err) { + console.log( + `RUN-REPORT failed for ${runDir}: ${err instanceof Error ? err.message : String(err)}`, + ) + return null + } +} + +/** + * Report every run under an experiment `outDir` (any depth of + * `runs//`), write each run's report, and write the rollup at + * `/run-report-round.{json,md}`. + */ +export async function reportSupervisorRound( + outDir: string, + opts: WriteSupervisorRunOptions & { title?: string } = {}, +): Promise { + const runDirs = await findSupervisorRunDirs(outDir) + const reports: SupervisorRunReport[] = [] + for (const runDir of runDirs) { + const r = await writeSupervisorRunReportSafe(runDir, { ...opts, echo: opts.echo ?? false }) + if (r !== null) reports.push(r) + } + const rollup = rollupSupervisorRuns(reports) + const md = renderSupervisorRollupMarkdown( + rollup, + opts.title ?? `Round rollup — ${basename(outDir)}`, + ) + const dest = opts.reportDir ?? outDir + if (opts.reportDir !== undefined) await mkdir(opts.reportDir, { recursive: true }).catch(() => {}) + await writeFile(join(dest, 'run-report-round.json'), JSON.stringify(rollup, null, 1)).catch( + () => {}, + ) + await writeFile(join(dest, 'run-report-round.md'), md).catch(() => {}) + if (opts.appendHeadlineTo !== undefined) { + await appendFile(opts.appendHeadlineTo, `${md}\n`).catch(() => {}) + } + if (opts.echo !== false) console.log(md) + return rollup +} + +/** + * Every Runtime supervisor run below `root`. + * + * When `root` itself is one run, return no children so callers can distinguish + * a single report from a parent-directory rollup. + */ +export async function findSupervisorRunDirs(root: string): Promise { + if (await isRuntimeSupervisorRunDir(root)) { + return [] + } + const found: string[] = [] + const walk = async (dir: string, depth: number): Promise => { + if (depth > 8) return + const entries = await readdir(dir, { withFileTypes: true }).catch(() => []) + for (const e of entries) { + if (!e.isDirectory()) continue + if (e.name === 'node_modules' || e.name === '.git') continue + const full = join(dir, e.name) + if (await isRuntimeSupervisorRunDir(full)) { + found.push(full) + continue + } + await walk(full, depth + 1) + } + } + await walk(root, 0) + return found.sort() +} diff --git a/src/supervisor-run/render.ts b/src/supervisor-run/render.ts index 233f85c17..5d77abc09 100644 --- a/src/supervisor-run/render.ts +++ b/src/supervisor-run/render.ts @@ -151,7 +151,7 @@ export function renderSupervisorRunMarkdown(r: SupervisorRunReport): string { out.push('|---|---|---:|---:|') for (const s of o.steersByWorker) { out.push( - `| ${s.workerId === null ? 'unavailable — legacy label join' : `\`${s.workerId}\``} | \`${s.worker}\` | ${s.queued} | ${s.delivered} |`, + `| ${s.workerId === null ? 'unavailable — worker id absent; label join used' : `\`${s.workerId}\``} | \`${s.worker}\` | ${s.queued} | ${s.delivered} |`, ) } out.push('') @@ -221,7 +221,7 @@ export function renderSupervisorRunMarkdown(r: SupervisorRunReport): string { out.push('|---|---|---|---|---|---|---|---|---:|---:|---:|---:|---|---:|') for (const w of e.perWorker) { out.push( - `| ${w.workerId === null ? 'unavailable — legacy label join' : `\`${w.workerId}\``} | \`${w.worker}\` | ${w.role ?? 'unavailable — source recorded no role'} | ${w.runtime ?? 'unavailable — source recorded no runtime'} | ${w.profileDigest === null ? 'unavailable — source recorded no profile digest' : `\`${w.profileDigest}\``} | ${w.status ?? 'unavailable — no terminal event'} | ${w.failure ?? 'none recorded'} | ${w.infra ?? 'unavailable'} | ${w.wallMs === null ? 'unavailable — no spawn/finish pair' : fmtMs(w.wallMs)} | ${w.tokensIn ?? 'unavailable — store does not attribute tokens per worker'} | ${w.tokensOut ?? 'unavailable — store does not attribute tokens per worker'} | ${w.patchBytes ?? 'unavailable — no worker patch file'} | ${w.passed === null ? 'unavailable — no verdict' : String(w.passed)} | ${w.score ?? 'unavailable — no numeric score'} |`, + `| ${w.workerId === null ? 'unavailable — worker id absent; label join used' : `\`${w.workerId}\``} | \`${w.worker}\` | ${w.role ?? 'unavailable — source recorded no role'} | ${w.runtime ?? 'unavailable — source recorded no runtime'} | ${w.profileDigest === null ? 'unavailable — source recorded no profile digest' : `\`${w.profileDigest}\``} | ${w.status ?? 'unavailable — no terminal event'} | ${w.failure ?? 'none recorded'} | ${w.infra ?? 'unavailable'} | ${w.wallMs === null ? 'unavailable — no spawn/finish pair' : fmtMs(w.wallMs)} | ${w.tokensIn ?? 'unavailable — store does not attribute tokens per worker'} | ${w.tokensOut ?? 'unavailable — store does not attribute tokens per worker'} | ${w.patchBytes ?? 'unavailable — no worker patch file'} | ${w.passed === null ? 'unavailable — no verdict' : String(w.passed)} | ${w.score ?? 'unavailable — no numeric score'} |`, ) } out.push('') diff --git a/src/supervisor-run/report-command.ts b/src/supervisor-run/report-command.ts index 0017f0153..b27121a33 100644 --- a/src/supervisor-run/report-command.ts +++ b/src/supervisor-run/report-command.ts @@ -1,7 +1,7 @@ /** * `agent-eval supervisor-run report ` takes one run directory and * prints the report the module already renders. The command reads through - * `analyzeSupervisorRun`, so a Runtime run dir and a loops run dir take the + * `analyzeSupervisorRun`, so a Runtime run directory takes the * same path and the status comes from the record `terminal-record.ts` names. * * Exit codes follow the other self-parsing subcommands: 2 for a usage error, @@ -11,7 +11,7 @@ import { stat } from 'node:fs/promises' import { resolve } from 'node:path' -import { analyzeSupervisorRun } from './loops-reader' +import { analyzeSupervisorRun } from './reader' import { renderSupervisorRunHeadline, renderSupervisorRunMarkdown } from './render' export type SupervisorRunReportFormat = 'headline' | 'markdown' | 'json' diff --git a/src/supervisor-run/rollout-nodes.ts b/src/supervisor-run/rollout-nodes.ts index 866204d35..1a2da1697 100644 --- a/src/supervisor-run/rollout-nodes.ts +++ b/src/supervisor-run/rollout-nodes.ts @@ -373,7 +373,7 @@ export function supervisorRunRolloutLinesFromFacts( usd: src.limits.spendUsd !== null || !close?.hasSpend ? null : close.spend.usd, tokens_in: workerSource?.tokensIn ?? (close?.hasSpend ? close.spend.tokens.input : null), tokens_out: workerSource?.tokensOut ?? (close?.hasSpend ? close.spend.tokens.output : null), - // Mirror the tokens_in/out journal fallback so a loops-shaped store whose + // Mirror the tokens_in/out journal fallback so a source whose // `settled` spend carries cache counters is not reported as null cache // beside real tokens. Gate on `hasCache` — a spend object without cache // counters must stay null, not a fabricated 0. @@ -386,7 +386,7 @@ export function supervisorRunRolloutLinesFromFacts( wall_s: wallMs === null ? null : wallMs / 1000, }, // A reader that knows where its worker artifacts live says so; only the - // loops layout is derivable from `supRunDir`, so guessing it for another + // A local layout is derivable from `supRunDir`, so guessing it for another // store would mint rows pointing at paths that never existed. artifacts: { patch_path: diff --git a/src/supervisor-run/runtime-fixture.test.ts b/src/supervisor-run/runtime-fixture.test.ts index 977f8a15f..c33f9173e 100644 --- a/src/supervisor-run/runtime-fixture.test.ts +++ b/src/supervisor-run/runtime-fixture.test.ts @@ -40,7 +40,7 @@ describe('real agent-runtime run — a clean run reports no gaps it does not hav expect(report.outcome.failure).toBeNull() expect(report.gaps.some((gap) => gap.startsWith('supStatus:'))).toBe(false) // The begin stamp carries identity and start only; the status is read from - // result.json by name, never copied onto a synthetic legacy state document. + // result.json by name, never copied onto a synthetic state document. expect(JSON.parse(source.state as string)).toEqual({ id: ROOT, startedAt: '2026-09-01T15:12:25.228Z', diff --git a/src/supervisor-run/runtime-r1-fixture.test.ts b/src/supervisor-run/runtime-r1-fixture.test.ts index fed6523c7..0547c4fda 100644 --- a/src/supervisor-run/runtime-r1-fixture.test.ts +++ b/src/supervisor-run/runtime-r1-fixture.test.ts @@ -19,7 +19,7 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { analyzeSupervisorRunSources } from './analyze' -import { analyzeSupervisorRun, findSupervisorRunDirs } from './loops-reader' +import { analyzeSupervisorRun, findSupervisorRunDirs } from './reader' import { renderSupervisorRunHeadline } from './render' import { supervisorRunRolloutLines } from './rollout-nodes' import { isRuntimeSupervisorRunDir, readRuntimeSupervisorRun } from './runtime-reader' @@ -55,7 +55,7 @@ describe('r1 settled no-winner: status comes from Runtime result.json kind', () expect(report.outcome.supReason).toBe('all-children-down') expect(report.gaps.some((gap) => gap.startsWith('supStatus:'))).toBe(false) // The reader passes Runtime's records through as bytes and fabricates no - // legacy status on the begin stamp. + // synthetic status on the begin stamp. expect(JSON.parse(source.state as string)).toEqual({ id: ROOT, startedAt: '2026-09-06T06:14:44.243Z', diff --git a/src/supervisor-run/runtime-reader.test.ts b/src/supervisor-run/runtime-reader.test.ts index da7882943..ec6f4b563 100644 --- a/src/supervisor-run/runtime-reader.test.ts +++ b/src/supervisor-run/runtime-reader.test.ts @@ -4,7 +4,7 @@ import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { analyzeSupervisorRunSources } from './analyze' import { analyzeSupervisorRunIntegrity } from './integrity' -import { analyzeSupervisorRun, findSupervisorRunDirs } from './loops-reader' +import { analyzeSupervisorRun, findSupervisorRunDirs } from './reader' import { supervisorRunRolloutLines } from './rollout-nodes' import { readRuntimeSupervisorRun } from './runtime-reader' import { parseSupervisorTree } from './source-facts' @@ -570,7 +570,7 @@ describe('Runtime FileRunContext supervisor reader', () => { ]) }) - it('refuses a legacy child-id tree when the spawn names a different owned tree', async () => { + it('refuses an ambiguous child-id tree when the spawn names a different owned tree', async () => { const parent = await mkdtemp(join(tmpdir(), 'runtime-supervisor-run-')) const runDir = join(parent, 'ambiguous-owned-tree-root') const childId = 'root:s0' @@ -1202,7 +1202,7 @@ describe('Runtime FileRunContext supervisor reader', () => { }) describe('journal-less Runtime run dirs — absence is a modeled result', () => { - it('returns the loops-shaped absent sources instead of throwing', async () => { + it('returns the absent source shape instead of throwing', async () => { const parent = await mkdtemp(join(tmpdir(), 'runtime-supervisor-run-')) const runDir = join(parent, 'journal-less') await mkdir(runDir, { recursive: true }) diff --git a/src/supervisor-run/runtime-reader.ts b/src/supervisor-run/runtime-reader.ts index 054231a3e..69afa36ec 100644 --- a/src/supervisor-run/runtime-reader.ts +++ b/src/supervisor-run/runtime-reader.ts @@ -18,9 +18,8 @@ * record Runtime writes when `supervise()` threw before a result landed. Both * pass through as bytes; `terminal-record.ts` reads them. The reader does not * decide which kinds count: a kind it refuses is a run that recorded its - * outcome and got reported as having none. Nothing here manufactures a loops - * `state.json` status from Runtime's documents: the analyzer names the record - * a status came from, and a synthetic legacy document would mislabel it. + * outcome and got reported as having none. The reader does not synthesize status fields. The analyzer names the Runtime + * record that supplied a status, so status provenance cannot be mislabelled. * * `usdKnown: false` / `tokensKnown: false` on ONE record is not a limit of this * store. The store recorded every other record completely, so the flags travel @@ -53,7 +52,7 @@ const FAILURE_FILE = 'failure.json' * attempt threw on a caller input error 16 minutes before the corrected attempt * opened the spawn journal. At that instant the directory held `observer.jsonl` * (2 records) and the failure record and no journal, so a journal-only test - * routed it to the loops reader, which reads neither file, and the recorded + * routed it to an unrelated reader, which reads neither file, and the recorded * throw was reported as no run at all. */ const RUNTIME_RUN_DIR_MARKERS: readonly string[] = [JOURNAL_FILE, OBSERVER_FILE, FAILURE_FILE] @@ -481,7 +480,7 @@ function assertFailureRecord(failure: Record | null, failurePat * The journal's begin stamp in the analyzer's state-document shape. It carries * the run identity and start instant only; the terminal status lives in * Runtime's own `result.json` / `failure.json`, which the analyzer reads by - * name, so no legacy `status` field is fabricated here. + * name, so no synthetic status field is fabricated here. */ function runtimeBeginState(root: string, startedAt: string): string { return JSON.stringify({ id: root, startedAt }) @@ -498,7 +497,7 @@ export interface RuntimeReaderOptions { /** * Sources for a run dir whose spawn journal does not exist. Mirrors the - * absent shape `readLoopsSupervisorRun` returns for a missing store: every + * absent shape the reader returns for a missing store: every * journal-dependent metric downstream reads `unavailable`, never 0. */ /** @@ -553,7 +552,7 @@ function absentRuntimeSupervisorRun( * Read one agent-runtime `createFileRunContext(dir)` directory. * * A run dir without `spawn-journal.jsonl` returns the same absent-shaped - * sources `readLoopsSupervisorRun` returns for a missing store: `journal` and + * sources the reader returns for a missing store: `journal` and * `workers` null, each with its reason, so every dependent metric reads * `unavailable` — never 0 and never a throw. Its `result.json` and * `failure.json` are still read, so a run that died before its first spawn diff --git a/src/supervisor-run/source-facts.ts b/src/supervisor-run/source-facts.ts index 6dad5f8a7..f1969a1f5 100644 --- a/src/supervisor-run/source-facts.ts +++ b/src/supervisor-run/source-facts.ts @@ -190,7 +190,7 @@ export interface CloseRow { id: string kind: 'settled' | 'cancelled' status: string | null - /** String verdict from legacy journals, or valid/invalid for a structured verdict. */ + /** String verdict from flat journals, or valid/invalid for a structured verdict. */ verdict: string | null /** Structured verdict validity, when recorded. */ valid: boolean | null @@ -369,8 +369,8 @@ export function workerSourceKey( /** * How a journal's rows were shaped on disk. * - * `flat` — one event object per line (`{kind:'spawned', id, ...}`). The loops - * supervisor writes this, and `readClaudeCodeSupervisorRun` synthesizes it. + * `flat` — one event object per line (`{kind:'spawned', id, ...}`). Claude Code + * and other adapters may synthesize this shape for the pure analyzer. * * `runtime-envelope` — the records `agent-runtime`'s `FileSpawnJournal` writes: * a `{kind:'begin', root, at}` header followed by `{kind:'event', root, event}` diff --git a/src/supervisor-run/terminal-record.test.ts b/src/supervisor-run/terminal-record.test.ts index cc46129aa..f63479eac 100644 --- a/src/supervisor-run/terminal-record.test.ts +++ b/src/supervisor-run/terminal-record.test.ts @@ -1,7 +1,7 @@ /** * Precedence of the terminal record: Runtime's settle record, then Runtime's - * failure record, then the legacy loops documents by name, then a named - * absence. Every branch states which record answered. + * failure record, then a named absence. Every branch states which record + * answered. */ import { describe, expect, it } from 'vitest' @@ -45,7 +45,7 @@ describe('readTerminalRecord', () => { expect(record.completed).toBe(true) }) - it('outranks legacy fields present on the same documents', () => { + it('ignores unrelated status fields on the same documents', () => { const record = readTerminalRecord({ state: { status: 'completed' }, result: { ...RUNTIME_RESULT, sup_status: 'completed' }, @@ -114,29 +114,19 @@ describe('readTerminalRecord', () => { expect(record.failure).toEqual({ unavailable: 'Runtime failure.json carries no error record' }) }) - it('falls back to legacy state.json status by name', () => { + it('does not treat legacy status fields as a terminal record', () => { const record = readTerminalRecord({ state: { status: 'completed' }, result: { sup_status: 'interrupted' }, failure: undefined, }) - expect(record.supStatus).toBe('completed') - expect(record.supStatusSource).toBe('legacy-state') - expect(record.completed).toBe(true) - expect(record.failure).toEqual({ - unavailable: 'legacy state.json records no failure document', - }) - }) - - it('falls back to legacy result.json sup_status by name', () => { - const record = readTerminalRecord({ - state: { id: 'x' }, - result: { sup_status: 'interrupted' }, - failure: undefined, + expect(record).toEqual({ + supStatus: { unavailable: NO_TERMINAL_RECORD }, + supStatusSource: { unavailable: NO_TERMINAL_RECORD }, + supReason: { unavailable: NO_TERMINAL_RECORD }, + failure: { unavailable: NO_TERMINAL_RECORD }, + completed: null, }) - expect(record.supStatus).toBe('interrupted') - expect(record.supStatusSource).toBe('legacy-result') - expect(record.completed).toBe(false) }) it('names the absence when no document carries a status', () => { diff --git a/src/supervisor-run/terminal-record.ts b/src/supervisor-run/terminal-record.ts index 6d32731ba..4be334afa 100644 --- a/src/supervisor-run/terminal-record.ts +++ b/src/supervisor-run/terminal-record.ts @@ -3,9 +3,8 @@ * run ended. Runtime's own record outranks every other source: `result.json` is * the `SupervisedResult` that `supervise()` returned, verbatim, and its `kind` * is the run's status. `failure.json` is the record Runtime writes when - * `supervise()` threw before a result landed. The control-plane-era loops - * documents (`state.json` `status`, `result.json` `sup_status`) stay readable - * as named legacy sources, so a report always says which record it read. + * `supervise()` threw before a result landed. Runtime's result and failure + * records are the only terminal status contract; state is timing metadata. * * Pure: takes already-parsed documents and returns `Measured` values. The * analyzer and the rollout minter share it so a run never has two statuses. @@ -21,15 +20,15 @@ import { /** The reason every terminal field reads `unavailable` when no store wrote one. */ export const NO_TERMINAL_RECORD = - 'no terminal record: Runtime result.json kind, Runtime failure.json, or legacy state.json / result.json status' + 'no terminal record: Runtime result.json kind or Runtime failure.json' /** The status reported for a run whose directory holds Runtime's failure record and no result. */ export const RUNTIME_FAILED_STATUS = 'failed' export interface TerminalRecordInput { - /** Legacy loops `state.json`, parsed; null when absent. */ + /** Runtime state document, parsed; null when absent. */ readonly state: Record | null - /** `result.json`, parsed; Runtime's `SupervisedResult` or the legacy loops result. */ + /** Runtime `result.json`, parsed; null when absent. */ readonly result: Record | null /** * Runtime's `failure.json`, parsed. `null` when the store was read and holds @@ -43,15 +42,14 @@ export interface TerminalRecord { readonly supStatusSource: Measured /** * Runtime's `reason` on a `no-winner` result. `null` when the record carries - * none (a winner, a failure record, or a legacy document). + * none (a winner or a failure record). */ readonly supReason: Measured /** The recorded error. `null` when the run recorded a result and no error. */ readonly failure: Measured /** - * True when the run recorded a delivered result (Runtime `winner`, legacy - * `completed`); false on every other recorded terminal state; null when no - * terminal record exists. + * True when the run recorded a delivered Runtime `winner`; false on every + * other recorded terminal state; null when no terminal record exists. */ readonly completed: boolean | null } @@ -77,7 +75,7 @@ function errorRecord( } export function readTerminalRecord(input: TerminalRecordInput): TerminalRecord { - const { state, result } = input + const { result } = input const resultKind = str(result?.kind) const settled = resultKind !== null && result !== null const failureRecord = @@ -120,27 +118,6 @@ export function readTerminalRecord(input: TerminalRecordInput): TerminalRecord { } } - const legacyStateStatus = str(state?.status) - if (legacyStateStatus !== null) { - return { - supStatus: legacyStateStatus, - supStatusSource: 'legacy-state', - supReason: null, - failure: unavailable('legacy state.json records no failure document'), - completed: legacyStateStatus === 'completed', - } - } - const legacyResultStatus = str(result?.sup_status) - if (legacyResultStatus !== null) { - return { - supStatus: legacyResultStatus, - supStatusSource: 'legacy-result', - supReason: null, - failure: unavailable('legacy result.json records no failure document'), - completed: legacyResultStatus === 'completed', - } - } - return { supStatus: unavailable(NO_TERMINAL_RECORD), supStatusSource: unavailable(NO_TERMINAL_RECORD), diff --git a/src/supervisor-run/types.ts b/src/supervisor-run/types.ts index 4d7a840ce..802a2e991 100644 --- a/src/supervisor-run/types.ts +++ b/src/supervisor-run/types.ts @@ -96,7 +96,7 @@ export interface WorkerLogSource { * The difference between "the artifact is missing" and "this store never * records that fact" is the difference between a run that spent $0 and a * harness that does not price inference — and the second harness is where a - * loops-shaped assumption becomes a fabricated zero. A reader declares its + * store-specific assumption becomes a fabricated zero. A reader declares its * limits once; the analyzer reports `unavailable` for everything downstream. * * `null` on a field means the source DOES carry that fact. @@ -129,8 +129,8 @@ export const NO_SOURCE_LIMITS: SourceLimits = { * dependent metrics into `unavailable` rather than 0. * * This is the whole input contract. Any store that can produce these strings - * (an on-disk loops run, an object-store archive, a database, a test fixture) - * is a valid source; `loopsSupervisorRunReader` is ONE implementation. + * (a Runtime run, an object-store archive, a database, or a test fixture) is a + * valid source; the Runtime reader is one implementation. */ export interface SupervisorRunSources { /** Stable identity of the run being analyzed (a directory, a run id, a URL). */ @@ -143,13 +143,13 @@ export interface SupervisorRunSources { /** * Supervision journal — spawned / settled / cancelled / metered events (JSONL). * Recursive readers put `role: 'supervisor' | 'worker'` on spawned rows; - * settled `verdict` may be a legacy string or `{ valid, score, ... }`. + * settled `verdict` may be a string or `{ valid, score, ... }`. */ readonly journal: string | null /** * Source-specific reason `journal` is null. The analyzer uses it verbatim as - * the `unavailable` reason on every journal-dependent metric, so a non-loops - * layout names its own journal file instead of inheriting the loops paths. + * the `unavailable` reason on every journal-dependent metric, so another + * layout names its own journal file instead of inheriting a path assumption. */ readonly journalMissingReason?: string /** Per-brain-call tap (JSONL): finish_reason, completion tokens, requested max tokens. */ @@ -165,16 +165,13 @@ export interface SupervisorRunSources { /** Why `workers` is null (only set when it is). */ readonly workersMissingReason: string | null /** - * Run result document (JSON). For a Runtime run dir this is `result.json`, - * the `SupervisedResult` that `supervise()` returned, verbatim; its `kind` - * is the run's status. For a loops run dir it is the legacy result document. + * Runtime `result.json`, the `SupervisedResult` returned by `supervise()`. */ readonly result: string | null /** * Runtime's terminal failure record (`failure.json`), written when * `supervise()` threw before a result landed. `null` = the store was read and - * holds no such record; `undefined` = the store has no failure document at - * all (the loops layout), which the analyzer reports as its own absence. + * holds no such record; `undefined` = the store has no failure document. */ readonly failure?: string | null /** @@ -207,9 +204,8 @@ export interface SupervisorRunSources { /** What this store structurally cannot record. See `SourceLimits`. */ readonly limits: SourceLimits /** - * Where the ROOT invocation's transcript lives. Undefined lets the rollout - * minter fall back to the loops layout (`/journal.jsonl`); any - * other store must say, or the row points at a path that never existed. + * Where the ROOT invocation's transcript lives. Any store that does not + * retain one must set this to `null`, or omit the path entirely. */ readonly rootTranscriptRef?: string | null /** @@ -381,8 +377,7 @@ export interface SpendMeasurement { * The run's total inference spend, measured two ways. * * `closeRecord` is the spend the store recorded as settled when the run - * closed (loops `state.json` `result.spentUsd`; Runtime `result.json` - * `spentTotal.usd`) — the billing-shaped answer. `journalDerived` is the + * closed (`result.json` `spentTotal.usd`) — the billing-shaped answer. `journalDerived` is the * spend execution observably consumed (journal `metered` + `settled` rows) — * the execution-accounting answer. Neither is canonical for the other's * question. The two cover different records at different moments, so @@ -459,10 +454,6 @@ export type SupervisorStatusSource = | 'runtime-result' /** Runtime's `failure.json`: `supervise()` threw before a result landed. */ | 'runtime-failure' - /** Control-plane-era loops `state.json` `status`. */ - | 'legacy-state' - /** Control-plane-era loops `result.json` `sup_status`. */ - | 'legacy-result' /** The error a run directory recorded. */ export interface TerminalFailure { @@ -489,16 +480,16 @@ export interface TerminalFailure { export interface OutcomeMetrics { /** - * The run's terminal status, exactly as its record spells it: Runtime's - * `winner` / `no-winner`, `failed` for a Runtime failure record, or the - * legacy loops status. `supStatusSource` names the record it came from. + * The run's terminal status, exactly as its Runtime record spells it: + * `winner` / `no-winner`, or `failed` for a Runtime failure record. + * `supStatusSource` names the record it came from. */ readonly supStatus: Measured readonly supStatusSource: Measured /** * Runtime's `reason` on a `no-winner` result (`all-children-down`, * `budget-exhausted`, `aborted`, `driver-failed`). `null` = the terminal - * record carries no reason (a winner, a failure record, a legacy document). + * record carries no reason (a winner or a failure record). */ readonly supReason: Measured /**