Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions docs/public-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,11 @@ Generated by `pnpm api:census` on demand — this is a dated reading, not a gate
| measure | count |
| --- | --- |
| export subpaths | 28 |
| published value exports (subpath x symbol) | 1403 |
| distinct symbols | 1231 |
| published value exports (subpath x symbol) | 1400 |
| distinct symbols | 1228 |
| production | 944 |
| planned | 218 |
| none | 241 |
| planned | 216 |
| none | 240 |

Type-only exports are not listed: a type binds no runtime surface, and removing one cannot break a caller at run time.

Expand Down Expand Up @@ -1289,25 +1289,22 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m

### `./supervisor-run`

33 value exports — 23 production, 4 planned, 6 none.
30 value exports — 23 production, 2 planned, 5 none.

| symbol | consumer | evidence |
| --- | --- | --- |
| `analyzeSupervisorRun` | production | discovery-lab:pursuits/agent-authored-runtime-control-deepseek-20260812h/workspaces/runtime-control-deps/runtime-control.mjs |
| `analyzeSupervisorRunIntegrity` | production | discovery-lab:pursuits/final-agent-authored-campaign-20260810/materialize-candidates.mjs |
| `analyzeSupervisorRunSources` | production | discovery-lab:experiments/0005-run-visibility/probe.mjs |
| `claudeCodeSupervisorRunReader` | none | only this package's tests: src/supervisor-run/claude-code-reader.test.ts:6 |
| `findSupervisorRunDirIn` | none | only this package's tests: src/supervisor-run/loops-reader.test.ts:11 |
| `findSupervisorRunDirs` | production | traces:src/cli.ts |
| `isRuntimeSupervisorRunDir` | production | discovery-lab:tools/disco.mjs |
| `isUnavailable` | production | discovery-lab:tools/disco.mjs |
| `loopsSupervisorRunReader` | planned | named in traces (bind not in the import graph) |
| `NO_SOURCE_LIMITS` | production | traces:src/supervisor-run-context.ts |
| `NO_TERMINAL_RECORD` | none | only this package's tests: src/supervisor-run/terminal-record.test.ts:8 |
| `parsePatch` | planned | named in browser-agent-driver (bind not in the import graph) |
| `parseSupervisorTree` | production | discovery-lab:experiments/0005-run-visibility/probe.mjs |
| `readClaudeCodeSupervisorRun` | none | only this package's tests: src/supervisor-run/claude-code-reader.test.ts:6 |
| `readLoopsSupervisorRun` | planned | named in discovery-lab (bind not in the import graph) |
| `readRuntimeSupervisorRun` | production | discovery-lab:pursuits/final-agent-authored-campaign-20260810/materialize-candidates.mjs |
| `readTerminalRecord` | production | this package: src/supervisor-run/analyze.ts:18 |
| `renderSupervisorRollupMarkdown` | production | traces:src/cli.ts |
Expand Down
2 changes: 1 addition & 1 deletion src/analyst/benchmark-implementation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -139,7 +139,7 @@ export const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
])

export const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 =
'e499ff9c5a3b24104d1a9dbc0c3742ffd08b67ff1b852922b423d38231954254'
'53b4c9f453ed05e46968ed39a8e365f8593163b264c0b1a770f7781005d3aae7'

export function analystBenchmarkImplementationDigest() {
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256
Expand Down
8 changes: 6 additions & 2 deletions src/statistics/paired-binary.ts
Original file line number Diff line number Diff line change
Expand Up @@ -439,7 +439,11 @@ export function pairedRiskDifferenceScore(
// the right endpoint is always inside and the bisection is well posed.
let lo = -1
let hi = riskDifference
for (let i = 0; i < 200; i++) {
// 64 halvings resolve the [-1, 1] search interval below 6e-20, past the
// precision available to the score calculation. More iterations only
// repeat identical floating-point values while multiplying every gate
// decision's cost.
for (let i = 0; i < 64; i++) {
const mid = (lo + hi) / 2
if (tangoScore(b, c, n, mid) > z) lo = mid
else hi = mid
Expand All @@ -449,7 +453,7 @@ export function pairedRiskDifferenceScore(
// Upper bound: root of score(delta) = -z on [riskDifference, 1].
let ulo = riskDifference
let uhi = 1
for (let i = 0; i < 200; i++) {
for (let i = 0; i < 64; i++) {
const mid = (ulo + uhi) / 2
if (tangoScore(b, c, n, mid) > -z) ulo = mid
else uhi = mid
Expand Down
27 changes: 26 additions & 1 deletion src/statistics/rank-tests.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,12 @@
import { ValidationError } from '../errors'
import { normalCdf } from '../math/normal'
import { lnGamma } from '../math/special-functions'
import { assertFiniteSample, makeRng, symmetricTwoSampleSeed } from './internal'
import {
assertFiniteSample,
binomialSignTwoSided,
makeRng,
symmetricTwoSampleSeed,
} from './internal'

// ── Rank tests: exact by default ─────────────────────────────────────
//
Expand Down Expand Up @@ -213,6 +218,26 @@ export function wilcoxonSignedRank(
const n = diffs.length
if (n === 0) return { w: 0, p: 1, method: 'exact', pFloor: 1, nNonZero: 0 }

// When every non-zero difference has the same magnitude, the signed-rank
// null reduces exactly to a binomial sign distribution. This is the common
// pass/fail shape ({-1, 0, +1}); recognizing it avoids a 100k-draw
// permutation for every repeated gate evaluation while preserving the exact
// p-value and attainable floor. Explicit method requests still retain their
// documented behavior.
const abs = Math.abs(diffs[0]!)
const allEqualAbs = diffs.every((d) => Math.abs(d) === abs)
if ((opts.method === undefined || opts.method === 'auto') && allEqualAbs) {
const positive = diffs.filter((d) => d > 0).length
const negative = n - positive
return {
w: (positive * (n + 1)) / 2,
p: binomialSignTwoSided(positive, negative),
method: 'exact',
pFloor: Math.min(1, 2 ** (1 - n)),
nNonZero: n,
}
}

const order = diffs.map((d, i) => ({ abs: Math.abs(d), i })).sort((x, y) => x.abs - y.abs)
const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs))
const ranks: number[] = new Array(n)
Expand Down
2 changes: 1 addition & 1 deletion src/supervisor-run/analyze.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -334,7 +334,7 @@ describe('analyzeSupervisorRun — decision quality and economics', () => {
// UNAVAILABLE != ZERO: an older supervisor wrote no tap, so truncation cannot be ruled out.
expect(r.economics.brainTruncations).toEqual({
unavailable:
'brain.jsonl absent — loops predates the brain-call tap, so truncation cannot be ruled out',
'brain.jsonl absent — this source has no brain-call tap, so truncation cannot be ruled out',
})
})

Expand Down
13 changes: 5 additions & 8 deletions src/supervisor-run/analyze.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
* The pure analyzer. Takes already-read bytes (`SupervisorRunSources`) and
* returns the report — every metric derivable from a synthetic journal string
* with no filesystem, no process, and no network. All I/O lives in a reader
* (`loops-reader.ts` is one).
* (`reader.ts` is one).
*/

import { summarizeNumberSeries } from '../statistics'
Expand Down Expand Up @@ -75,16 +75,13 @@ export function analyzeSupervisorRunSources(

const journalMissing =
src.journalMissingReason ??
(src.supRunDir === null
? 'no supervisor run dir under <ws>/.agent/supervisor (or legacy <ws>/.loops/supervisor)'
: 'journal.jsonl absent')
(src.supRunDir === null ? 'no Runtime supervisor run directory' : 'journal.jsonl absent')
const haveJournal = src.journal !== null
const tree = parseSupervisorTree(src)
const state = tree.state
const result = parseJson(src.result)
const judge = parseJson(src.judge)
// Runtime's settle record outranks the legacy loops documents; the record
// that answered is named on the report so a status never arrives unlabeled.
// Runtime's terminal record is named on the report so a status never arrives unlabeled.
const terminal = readTerminalRecord({
state,
result,
Expand Down Expand Up @@ -767,8 +764,8 @@ export function analyzeSupervisorRunSources(
'brain.brainTruncations',
src.brainLogMissingReason ??
(src.supRunDir === null
? 'no supervisor run dir under <ws>/.agent/supervisor (or legacy <ws>/.loops/supervisor)'
: 'brain.jsonl absent — loops predates the brain-call tap, so truncation cannot be ruled out'),
? 'no Runtime supervisor run directory'
: 'brain.jsonl absent — this source has no brain-call tap, so truncation cannot be ruled out'),
)
: brainCalls.filter((c) => c.finish_reason === 'length').length,
workers: {
Expand Down
4 changes: 2 additions & 2 deletions src/supervisor-run/claude-code-reader.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -397,7 +397,7 @@ describe('claudeCodeSupervisorRunReader', () => {
// A worker with no retained transcript reports null tokens, not 0.
const pruned = workers.find((w) => w.artifacts.transcript_ref === null)
expect(pruned?.cost.tokens_in).toBeNull()
// The root row points at the real session transcript, not a loops path.
// The root row points at the real session transcript, not a Runtime path.
expect(root?.artifacts.transcript_ref).toBe(s.transcriptPath)
})

Expand All @@ -411,7 +411,7 @@ describe('claudeCodeSupervisorRunReader', () => {
expect(isUnavailable(report.orchestration.steers)).toBe(true)
})

it('exposes the SupervisorRunReader contract loops implements', async () => {
it('exposes the SupervisorRunReader contract Runtime implements', async () => {
const s = await writeSession()
const reader = claudeCodeSupervisorRunReader({
transcriptPath: s.transcriptPath,
Expand Down
8 changes: 4 additions & 4 deletions src/supervisor-run/claude-code-reader.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
/**
* Supervision-tree reader over a THIRD-PARTY harness: Claude Code.
*
* `loops-reader.ts` reads a supervisor we wrote, whose journal was designed
* The Runtime reader reads a supervisor journal designed for this analysis.
* for this analysis. This reader reads a harness we do not control, whose
* transcript was designed for replaying a chat — and recovers the same tree
* from it. If both produce a `SupervisorRunSources`, the tree model is a
Expand Down Expand Up @@ -31,12 +31,12 @@
* reports `unavailable — <reason>` instead of the $0 / 0-accepted that summing
* an empty field would produce. See `SourceLimits`.
*
* ## Metric coverage vs the loops journal
* ## Metric coverage vs the Runtime journal
*
* Measured on a real 52-agent session (fixture:
* `tests/fixtures/supervisor-run/claude-code-session-*`).
*
* | Metric | loops | Claude Code | Why |
* | Metric | Runtime | Claude Code | Why |
* |---|---|---|---|
* | workersSpawned / Settled / Cancelled | full | full | spawn tool_use + task-notification + TaskStop |
* | steers / steersDelivered / steersByWorker | full | full | `SendMessage`; delivery from its tool_result |
Expand Down Expand Up @@ -642,7 +642,7 @@ export async function readClaudeCodeSupervisorRun(
}
}

/** A `SupervisorRunReader` over a Claude Code session — the same contract loops implements. */
/** A `SupervisorRunReader` over a Claude Code session — the same contract Runtime implements. */
export function claudeCodeSupervisorRunReader(opts: ClaudeCodeReaderOptions): SupervisorRunReader {
return {
runRef: opts.runRef ?? opts.transcriptPath,
Expand Down
2 changes: 1 addition & 1 deletion src/supervisor-run/fixtures.ts
Original file line number Diff line number Diff line change
Expand Up @@ -192,7 +192,7 @@ export function fixtureSources(over: Partial<SupervisorRunSources> = {}): Superv
runRef: '/tmp/cell/runs/inst-1/ARM',
instanceId: 'inst-1',
arm: 'ARM',
supRunDir: '/tmp/cell/ws/.loops/supervisor/sup-1-test',
supRunDir: '/tmp/cell/runtime-run',
journal: null,
brainLog: null,
state: null,
Expand Down
6 changes: 1 addition & 5 deletions src/supervisor-run/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -45,16 +45,12 @@ export {
} from './integrity'
export {
analyzeSupervisorRun,
findSupervisorRunDirIn,
findSupervisorRunDirs,
type LoopsReaderOptions,
loopsSupervisorRunReader,
readLoopsSupervisorRun,
reportSupervisorRound,
type WriteSupervisorRunOptions,
writeSupervisorRunReport,
writeSupervisorRunReportSafe,
} from './loops-reader'
} from './reader'
export {
renderSupervisorRollupMarkdown,
renderSupervisorRunHeadline,
Expand Down
1 change: 1 addition & 0 deletions src/supervisor-run/integrity.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,7 @@ function completedControlSource(inbox: string | null, events: string | null): Su
return fixtureSources({
journal: fixtureJournal({ workers: [['worker', 1, 4]] }),
state: fixtureState({ startSec: 0, endSec: 5 }),
result: JSON.stringify({ kind: 'winner', tree: { root: 'sup-1-test', nodes: [] } }),
workers: [
{
workerId: 'sup-1-test:s0',
Expand Down
Loading