diff --git a/docs/public-api.md b/docs/public-api.md index 7e69de7a..ad5b98ab 100644 --- a/docs/public-api.md +++ b/docs/public-api.md @@ -7,11 +7,11 @@ Generated by `pnpm api:census` on demand — this is a dated reading, not a gate | measure | count | | --- | --- | | export subpaths | 28 | -| published value exports (subpath x symbol) | 1400 | -| distinct symbols | 1228 | -| production | 944 | -| planned | 216 | -| none | 240 | +| published value exports (subpath x symbol) | 1416 | +| distinct symbols | 1240 | +| production | 941 | +| planned | 239 | +| none | 236 | Type-only exports are not listed: a type binds no runtime surface, and removing one cannot break a caller at run time. @@ -111,7 +111,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `argHash` | production | agent-runtime:src/runtime/supervise/detector-monitor.ts | | `assertCapabilityHeadroom` | production | blueprint-agent:scripts/experiments/lib/validity-gates.ts | | `assertCrossFamily` | production | agent-builder:frontier/judges/artifact-head-to-head.ts | -| `assertCrossFamilyServed` | planned | doc: README.md | +| `assertCrossFamilyServed` | planned | doc: docs/building-doctrine.md | | `assertNoHiddenLeak` | planned | doc: docs/design/statistics-decisions.md | | `assertNoJudgeVerdict` | production | this package: src/analyst/policy-edit.ts:6 | | `assertProductBenchmarkRun` | production | creative-agent:eval/research-package.ts | @@ -191,7 +191,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `createTraceAnalyst` | production | blueprint-agent:scripts/experiments/lib/sandbox-driver/analyst-steer.ts | | `CrossFamilyError` | production | creative-agent:eval/lib/judge-ensemble.ts | | `decidePairedPromotion` | production | discovery-lab:tools/decide.mjs | -| `DECISION_PAIRED_DELTA_STATISTIC` | production | this package: src/campaign/gates/promotion-policy.ts:33 | +| `DECISION_PAIRED_DELTA_STATISTIC` | production | this package: src/campaign/gates/promotion-policy.ts:19 | | `DEFAULT_BOUNDED_PROCESS_TIMEOUT_MS` | production | this package: src/command-runner.ts:28 | | `DEFAULT_FAILURE_REASON_RULES` | none | only this package's tests: tests/failure-taxonomy.test.ts:2 | | `DEFAULT_PERMUTATIONS` | none | — | @@ -211,7 +211,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `domainEvidencePattern` | production | blueprint-agent:scripts/experiments/lib/analyze-vb-run/session-loader.ts | | `dominates` | production | blueprint-agent:scripts/experiments/lib/competition-analytics.ts | | `ensembleJudge` | production | blueprint-agent:scripts/experiments/lib/qa-grader.ts | -| `eProcess` | production | this package: src/campaign/gates/sequential.ts:35 | +| `eProcess` | production | this package: src/campaign/gates/sequential.ts:37 | | `EquivalenceProtocolError` | planned | doc: docs/verification-strategies.md | | `equivalenceVerdict` | planned | doc: docs/verdicts.md | | `ERROR_COUNT_PATTERNS` | production | blueprint-agent:scripts/experiments/lib/error-count-extractor.ts | @@ -292,13 +292,13 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `makeFinding` | production | agent-dev-container:products/intelligence/api/src/lib/consultant/candidates.ts | | `makePolicyEdit` | none | only this package's tests: src/analyst/policy-edit.test.ts:2 | | `makePolicyEditCandidateRecord` | planned | doc: docs/campaign-proposers.md | -| `makeProposalFinding` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | -| `manifestContentDigest` | production | this package: src/campaign/gates/sequential.ts:34 | +| `makeProposalFinding` | production | agent-runtime:examples/improve/improve.ts | +| `manifestContentDigest` | production | this package: src/campaign/gates/sequential.ts:36 | | `MANN_WHITNEY_EXACT_MAX_STATES` | none | — | | `MANN_WHITNEY_EXACT_MAX_WORK` | none | — | | `mannWhitneyU` | production | blueprint-agent:scripts/experiments/lib/competition-analytics.ts | | `maximumChargeForLlmRequest` | production | agent-dev-container:products/intelligence/api/src/lib/optimization-engine.ts | -| `mcnemar` | production | agent-runtime:bench/src/swe-arena/analyze.ts | +| `mcnemar` | production | blueprint-agent:scripts/experiments/analyze-convergence.ts | | `mcnemarPower` | production | supervisor-lab:bench/deepswe/headroom.ts | | `mcnemarRequiredN` | production | blueprint-agent:scripts/experiments/analyze-convergence.ts | | `minimumPairsForPairedDeltaTest` | production | agent-dev-container:products/intelligence/api/src/routes/optimizations.ts | @@ -318,18 +318,18 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `otlpTextToTraceAnalysisStore` | production | braid:src/adapters/analysis/trace-store.ts | | `OUTPUT_VALUE` | production | agent-runtime:src/runtime/supervise-surface.ts | | `pairArms` | production | agent-knowledge:src/memory/experiment/learning-pairs.ts | -| `pairedBinaryScale` | production | this package: src/paired-promotion-decision.ts:51 | +| `pairedBinaryScale` | production | this package: src/paired-promotion-decision.ts:50 | | `pairedBootstrap` | production | agent-builder:frontier/adapters/fleet-eval-runner.ts | -| `pairedCohensDz` | production | this package: src/contract/analyze-runs.ts:37 | +| `pairedCohensDz` | production | this package: src/contract/analyze-runs.ts:38 | | `pairedDeltaTest` | production | gtm-agent:eval/matrix/report.ts | -| `pairedDeltaTieFraction` | production | this package: src/paired-promotion-decision.ts:51 | +| `pairedDeltaTieFraction` | production | this package: src/paired-promotion-decision.ts:50 | | `pairedEvalueSequence` | production | creative-agent:src/lib/experiments/ab-design.ts | | `pairedMde` | production | discovery-lab:tools/design-gate.mjs | | `pairedRiskDifference` | production | blueprint-agent:scripts/experiments/lib/validity-gates.ts | -| `pairedRiskDifferenceExact` | production | this package: src/paired-promotion-decision.ts:51 | -| `pairedRiskDifferenceScore` | production | this package: src/paired-promotion-decision.ts:51 | +| `pairedRiskDifferenceExact` | production | this package: src/paired-promotion-decision.ts:50 | +| `pairedRiskDifferenceScore` | production | this package: src/paired-promotion-decision.ts:50 | | `pairedSignTest` | production | agent-dev-container:products/intelligence/api/src/lib/paired-stats.ts | -| `pairedTTest` | production | this package: src/contract/analyze-runs.ts:37 | +| `pairedTTest` | production | this package: src/contract/analyze-runs.ts:38 | | `pairRunRecords` | production | agent-dev-container:products/intelligence/api/src/lib/eval-engine.ts | | `PairwiseSteeringOptimizer` | production | browser-agent-driver:bench/research/webvoyager-agent-eval-loop.mjs | | `paretoChart` | production | workcomp-agent:eval/benchmark/select.ts | @@ -507,7 +507,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS` | production | this package: src/analyst/chat-trace-engine.ts:46 | | `defaultIsMaterial` | none | only this package's tests: src/analyst/analyst.test.ts:8 | | `defineCustomAnalyst` | planned | doc: docs/trace-analysis.md | -| `defineTraceAnalyst` | planned | doc: docs/charter.md | +| `defineTraceAnalyst` | planned | doc: docs/feature-guide.md | | `deriveEfficiencyFindings` | none | only this package's tests: src/trace-analyst/behavioral-metrics.test.ts:2 | | `diffFindings` | production | agent-runtime:src/analyst-loop/run-analyst-loop.ts | | `effectiveAnalystProtocolSha256` | production | this package: src/analyst/benchmark-command-persistence.ts:31 | @@ -532,7 +532,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `loadCodeTraceVerificationArtifacts` | production | this package: src/analyst/benchmark-public-data.ts:32 | | `loadPublicBenchmarkRows` | production | this package: scripts/gepa-analyst-campaign.ts:64 | | `makeFinding` | production | agent-dev-container:products/intelligence/api/src/lib/consultant/candidates.ts | -| `makeProposalFinding` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `makeProposalFinding` | production | agent-runtime:examples/improve/improve.ts | | `MAX_INCORRECT_BLOCK_STEPS` | production | this package: src/analyst/benchmark-public-adapters.ts:9 | | `MAX_INCORRECT_BLOCKS` | production | this package: src/analyst/benchmark-public-adapters.ts:9 | | `mergePrimeRawUsage` | planned | doc: docs/prime-analyst.md | @@ -628,7 +628,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./campaign` -122 value exports — 81 production, 15 planned, 26 none. +124 value exports — 80 production, 18 planned, 26 none. | symbol | consumer | evidence | | --- | --- | --- | @@ -637,7 +637,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `assertCampaignDesign` | production | this package: src/campaign/plan-campaign-run.ts:10 | | `assertCampaignSplitIdentity` | production | agent-knowledge:src/memory/experiment/learning-pairs.ts | | `assertCodeSurfaceIdentity` | none | — | -| `assertCompleteSearchHistory` | production | this package: src/campaign/optimization-method.ts:9 | +| `assertCompleteSearchHistory` | production | this package: src/campaign/optimization-method.ts:14 | | `assertSearchHistoryMatchesReplay` | planned | doc: docs/search-history-receipts.md | | `autoevalsScorerJudge` | none | only this package's tests: src/campaign/upstream-evaluators.test.ts:6 | | `buildCellSchedule` | production | this package: src/campaign/plan-campaign-run.ts:9 | @@ -646,9 +646,9 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `buildTraceAnalystSurfaceDispatch` | none | only this package's tests: src/campaign/analyst-surface.test.ts:4 | | `campaignBreakdown` | production | this package: src/campaign/external-text-evaluation.ts:11 | | `campaignMeanComposite` | production | agent-dev-container:products/intelligence/api/src/lib/optimization-engine.ts | -| `campaignMeasurementDigest` | production | this package: src/contract/self-improve-method.ts:12 | -| `campaignScenarioIdentity` | production | agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts | -| `campaignSplitDigest` | production | agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts | +| `campaignMeasurementDigest` | production | this package: src/contract/self-improve-method.ts:13 | +| `campaignScenarioIdentity` | production | agent-runtime:src/improvement/method-identity.ts | +| `campaignSplitDigest` | production | discovery-lab:tools/profile-population.mjs | | `campaignSplitDigestFromIdentities` | production | agent-runtime:src/improvement/method-identity.ts | | `canonicalDigest` | production | agent-dev-container:products/intelligence/api/src/lib/optimization-engine.ts | | `cellCachePath` | production | discovery-lab:tools/strict-screen.mjs | @@ -669,15 +669,15 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS` | none | — | | `DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS` | none | — | | `defaultProductionGate` | production | agent-app:src/eval-campaign/index.ts | -| `detectScale` | production | this package: src/campaign/gates/promotion-policy.ts:39 | +| `detectScale` | production | this package: src/campaign/gates/promotion-policy.ts:25 | | `dimensionRegressions` | production | agent-dev-container:products/intelligence/api/src/lib/optimization-engine.ts | | `discoverEvalFixtures` | planned | doc: docs/eval-fixtures.md | -| `emitLoopProvenance` | production | this package: src/contract/self-improve.ts:26 | +| `emitLoopProvenance` | production | this package: src/contract/self-improve.ts:36 | | `externalTextOptimizationMethod` | production | agent-app:src/eval-campaign/index.ts | | `FileSearchLedger` | none | — | | `finalizeProfileMatrix` | production | discovery-lab:tools/run-profile-confirmation.mjs | | `fsCampaignStorage` | production | agent-dev-container:products/intelligence/api/src/lib/eval-engine.ts | -| `FsLabeledScenarioStore` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `FsLabeledScenarioStore` | planned | named in agent-runtime (bind not in the import graph) | | `gepaOptimizationMethod` | production | agent-app:src/eval-campaign/index.ts | | `gitWorktreeAdapter` | production | agent-runtime:src/improvement/code-execution.ts | | `heldOutGate` | production | agent-dev-container:products/intelligence/api/src/lib/optimization-gate.ts | @@ -690,13 +690,13 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `llmJudge` | production | agent-runtime:examples/agentic-data-creation/offline-fixtures.ts | | `loadEvalFixture` | planned | doc: docs/eval-fixtures.md | | `loadEvalFixtureScenarios` | planned | example: examples/eval-fixtures-quickstart/index.ts:11 | -| `loopProvenanceArgsFromResult` | production | this package: src/contract/self-improve.ts:26 | +| `loopProvenanceArgsFromResult` | production | this package: src/contract/self-improve.ts:36 | | `loopProvenanceSpans` | none | only this package's tests: src/campaign/provenance-integrity.test.ts:5 | | `makePlaybackDispatch` | none | only this package's tests: src/campaign/presets/playback.test.ts:6 | -| `makeProposalFinding` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `makeProposalFinding` | production | agent-runtime:examples/improve/improve.ts | | `neutralizationGate` | planned | consumer tests: tax-agent:tests/eval/benchmarks/taxcalc/evolve.ts | | `neutralizeText` | planned | consumer tests: tax-agent:tests/eval/benchmarks/taxcalc/compile-and-prove/compile-diff.ts | -| `openAutoPr` | production | this package: src/campaign/presets/run-improvement-loop.ts:7 | +| `openAutoPr` | production | this package: src/campaign/presets/run-improvement-loop.ts:8 | | `openSearchLedger` | production | this package: src/campaign/gepa-optimization-method.ts:66 | | `optimizationTokenUsageFromSummary` | production | this package: src/campaign/external-text-optimization.ts:41 | | `pairHoldout` | production | agent-dev-container:products/intelligence/api/src/lib/run-provenance.ts | @@ -728,21 +728,23 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `runOptimization` | production | agent-dev-container:products/sandbox/evals/src/auto-optimization.ts | | `runProfileMatrix` | production | agent-runtime:examples/product-eval/product-eval.ts | | `runProfileMatrixSegment` | production | discovery-lab:tools/run-profile-confirmation.mjs | +| `scopedOptimizationMethod` | planned | doc: docs/campaign-proposers.md | | `scoreboardSummary` | production | blueprint-agent:scripts/experiments/eval/launch-scoreboard/run.ts | | `scoreDiscrimination` | production | supervisor-lab:bench/comms/seat-discrimination.ts | | `scoreUserStory` | production | blueprint-agent:scripts/experiments/eval/launch-scoreboard/scoreboard-core.ts | -| `searchHistoryCoverageRow` | production | this package: src/campaign/optimization-method.ts:9 | -| `SearchHistoryRequiredError` | none | only this package's tests: src/campaign/presets/compare-optimization-methods-history.test.ts:10 | -| `SearchLedgerConflictError` | production | this package: src/campaign/search-ledger.ts:34 | -| `SearchLedgerError` | production | this package: src/campaign/search-ledger.ts:34 | +| `searchHistoryCoverageRow` | production | this package: src/campaign/optimization-method.ts:14 | +| `SearchHistoryRequiredError` | none | only this package's tests: src/campaign/presets/compare-optimization-methods-history.test.ts:14 | +| `SearchLedgerConflictError` | production | this package: src/campaign/search-ledger.ts:32 | +| `SearchLedgerError` | production | this package: src/campaign/search-ledger.ts:32 | | `SearchLedgerIntegrityError` | production | this package: src/campaign/search-ledger-file.ts:10 | | `SearchRecorder` | production | this package: src/campaign/presets/run-optimization.ts:38 | | `selectDiscriminative` | none | only this package's tests: src/campaign/scenario-selection.test.ts:2 | | `sequentialDecide` | planned | doc: docs/experiment.md | +| `sequentialOptimizationMethod` | planned | doc: docs/campaign-proposers.md | | `sequentialPairedGate` | planned | consumer tests: legal-agent:tests/eval/self-improve.ts | | `skillOptOptimizationMethod` | production | agent-app:src/eval-campaign/index.ts | | `surfaceContentHash` | production | discovery-lab:tools/run-profile-confirmation.mjs | -| `surfaceDispatchRef` | production | this package: src/campaign/presets/run-final-comparison.ts:4 | +| `surfaceDispatchRef` | production | this package: src/campaign/presets/run-final-comparison.ts:11 | | `surfaceHash` | production | agent-builder:src/lib/.server/eval/loops/evolution-runner.ts | | `tangleTracesRoot` | none | only this package's tests: src/campaign/run-dir.test.ts:4 | | `traceAnalystQualityJudge` | production | this package: scripts/gepa-analyst-campaign.ts:74 | @@ -751,7 +753,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `validateSearchLedgerEvent` | planned | consumer tests: discovery-lab:tools/search-ledger-planless.test.mjs | | `verifyCodeSurface` | production | agent-runtime:src/candidate-execution/builder.ts | | `verifyLoopProvenanceRecord` | none | only this package's tests: src/campaign/provenance-integrity.test.ts:5 | -| `verifySearchHistoryArtifact` | production | this package: src/campaign/optimization-method.ts:9 | +| `verifySearchHistoryArtifact` | production | this package: src/campaign/optimization-method.ts:14 | | `verifySearchHistoryReceipt` | planned | doc: docs/search-history-receipts.md | | `WorktreeAdapterError` | none | only this package's tests: tests/campaign/worktree.test.ts:17 | @@ -764,7 +766,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `analyzeRuns` | production | agent-builder:eval/scripts/insight-report.ts | | `buildDefaultAnalystRegistry` | production | agent-dev-container:products/intelligence/api/src/lib/deep-trace-analyst.ts | | `buildEvidenceVector` | planned | doc: docs/experiment.md | -| `campaignSplitDigest` | production | agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts | +| `campaignSplitDigest` | production | discovery-lab:tools/profile-population.mjs | | `compareOptimizationMethods` | production | agent-app:src/eval-campaign/index.ts | | `composeGate` | production | discovery-lab:tools/confirmation-gate.mjs | | `createChatClient` | production | agent-dev-container:products/intelligence/api/src/lib/intent-audit-analyst.ts | @@ -792,7 +794,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `inMemoryCampaignStorage` | production | agent-dev-container:products/sandbox/evals/src/auto-optimization.ts | | `InMemoryOutcomeStore` | planned | consumer tests: phony:products/builder/api/src/eval/outcomes-correlate.test.ts | | `llmJudge` | production | agent-runtime:examples/agentic-data-creation/offline-fixtures.ts | -| `makeProposalFinding` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `makeProposalFinding` | production | agent-runtime:examples/improve/improve.ts | | `measuredComparisonFromAgentProfileImprovementExperiment` | production | agent-runtime:src/intelligence/authored-profile-improvement.ts | | `measuredComparisonFromCandidateExperiment` | production | agent-runtime:src/intelligence/improvement-cycle.ts | | `observeCodeAgentSession` | production | this package: src/contract/intake/code-agent-session.ts:13 | @@ -829,12 +831,12 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./experiment` -84 value exports — 50 production, 17 planned, 17 none. +89 value exports — 55 production, 17 planned, 17 none. | symbol | consumer | evidence | | --- | --- | --- | | `amendExperiment` | planned | doc: docs/experiment.md | -| `assertDesignAdequate` | planned | doc: docs/experiment.md | +| `assertDesignAdequate` | planned | doc: docs/eval-surface-map.md | | `assertFunnelReconciles` | none | only this package's tests: tests/experiment/funnel.test.ts:7 | | `assertMatchedBudgets` | planned | doc: docs/experiment.md | | `benjaminiHochberg` | production | agent-runtime:bench/src/corpus-report.mts | @@ -846,30 +848,33 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `clusteredPower` | production | this package: scripts/tb-gated-stop-ab.ts:55 | | `comparePairedArms` | production | agent-knowledge:src/memory/experiment/learning-metrics.ts | | `composeFunnels` | planned | doc: docs/experiment.md | -| `computeEstimand` | production | this package: src/experiment/define.ts:21 | -| `computeInterval` | production | this package: src/experiment/define.ts:21 | +| `computeEstimand` | production | this package: src/experiment/define.ts:20 | +| `computeInterval` | production | this package: src/experiment/define.ts:20 | | `createCampaignEvidenceReceipt` | production | this package: src/campaign/presets/compare-optimization-methods.ts:11 | | `createEvidenceReceipt` | production | this package: src/experiment/campaign-evidence.ts:7 | +| `defineEvaluationClaim` | production | this package: src/campaign/final-evidence.ts:2 | | `defineExperiment` | production | this package: scripts/tb-gated-stop-ab.ts:55 | | `DesignRefusalError` | none | only this package's tests: tests/experiment/power.test.ts:10 | -| `eProcess` | production | this package: src/campaign/gates/sequential.ts:35 | +| `eProcess` | production | this package: src/campaign/gates/sequential.ts:37 | | `evaluateCondition` | planned | named in agent-dev-container (bind not in the import graph) | -| `evaluateHaltRule` | production | this package: src/experiment/define.ts:21 | +| `evaluateHaltRule` | production | this package: src/experiment/define.ts:20 | | `evaluateHypothesis` | none | only this package's tests: tests/tier2.test.ts:8 | -| `evaluateIdentityGate` | production | this package: src/experiment/define.ts:21 | -| `evaluateOracleDeterminismGate` | production | this package: src/experiment/define.ts:21 | -| `evaluatePopulationReproducibilityGate` | production | this package: src/experiment/define.ts:21 | -| `evaluatePowerFloorGate` | production | this package: src/experiment/define.ts:21 | +| `evaluateIdentityGate` | production | this package: src/experiment/define.ts:20 | +| `evaluateOracleDeterminismGate` | production | this package: src/experiment/define.ts:20 | +| `evaluatePopulationReproducibilityGate` | production | this package: src/experiment/define.ts:20 | +| `evaluatePowerFloorGate` | production | this package: src/experiment/define.ts:20 | | `evaluatePredicate` | production | this package: src/experiment/funnel.ts:17 | -| `evaluateProvenanceGate` | production | this package: src/experiment/define.ts:21 | +| `evaluateProvenanceGate` | production | this package: src/experiment/define.ts:20 | | `EVIDENCE_AUTHORITY_KINDS` | none | — | | `EVIDENCE_RECEIPT_VERSION` | none | — | | `EVIDENCE_STATES` | none | only this package's tests: src/experiment/evidence-record.test.ts:4 | | `evidenceRegistryRecordSchema` | none | — | -| `executeAdmissionRule` | production | this package: src/experiment/define.ts:58 | -| `executeDecisionRule` | production | this package: src/experiment/define.ts:21 | +| `executeAdmissionRule` | production | this package: src/experiment/define.ts:65 | +| `executeDecisionRule` | production | this package: src/experiment/define.ts:20 | | `ExperimentTracker` | production | phony:products/builder/api/src/eval/evolution.ts | | `fileExperimentStore` | production | phony:products/builder/api/src/eval/evolution.ts | +| `FinalEvidenceConflictError` | production | this package: src/campaign/final-evidence.ts:7 | +| `FinalEvidenceError` | production | this package: src/campaign/final-evidence.ts:7 | | `FunnelIntegrityError` | none | only this package's tests: tests/experiment/funnel.test.ts:7 | | `hashJson` | production | agent-dev-container:products/intelligence/api/src/lib/ingest-optimization-mapping.ts | | `heldoutSignificance` | production | agent-dev-container:products/intelligence/api/src/lib/eval-engine.ts | @@ -877,44 +882,46 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `INDEPENDENT_EVIDENCE_AUTHORITY_KINDS` | none | — | | `inMemoryExperimentStore` | none | only this package's tests: src/experiment-tracker.test.ts:2 | | `isIndependentEvidence` | planned | consumer tests: agent-runtime:tests/integration/runtime-eval-pursuit-evidence.test.ts | -| `manifestContentDigest` | production | this package: src/campaign/gates/sequential.ts:34 | +| `manifestContentDigest` | production | this package: src/campaign/gates/sequential.ts:36 | | `MatchedBudgetError` | none | only this package's tests: tests/experiment/budget-and-seal.test.ts:8 | -| `mcnemar` | production | agent-runtime:bench/src/swe-arena/analyze.ts | +| `mcnemar` | production | blueprint-agent:scripts/experiments/analyze-convergence.ts | | `mcnemarPower` | production | supervisor-lab:bench/deepswe/headroom.ts | | `mcnemarRequiredN` | production | blueprint-agent:scripts/experiments/analyze-convergence.ts | | `mulberry32` | production | agent-knowledge:src/memory/holdout.ts | +| `openFinalEvidenceLedger` | planned | example: examples/evaluation-integrity/index.ts:10 | | `openSealedExperiment` | production | this package: scripts/tb-gated-stop-ab.ts:55 | | `pairArms` | production | agent-knowledge:src/memory/experiment/learning-pairs.ts | | `pairedBootstrap` | production | agent-builder:frontier/adapters/fleet-eval-runner.ts | | `pairedEvalueSequence` | production | creative-agent:src/lib/experiments/ab-design.ts | | `pairedMde` | production | discovery-lab:tools/design-gate.mjs | | `pairedRiskDifference` | production | blueprint-agent:scripts/experiments/lib/validity-gates.ts | -| `pairedRiskDifferenceExact` | production | this package: src/paired-promotion-decision.ts:51 | -| `pairedRiskDifferenceScore` | production | this package: src/paired-promotion-decision.ts:51 | +| `pairedRiskDifferenceExact` | production | this package: src/paired-promotion-decision.ts:50 | +| `pairedRiskDifferenceScore` | production | this package: src/paired-promotion-decision.ts:50 | | `pairHoldout` | production | agent-dev-container:products/intelligence/api/src/lib/run-provenance.ts | | `pairRunRecords` | production | agent-dev-container:products/intelligence/api/src/lib/eval-engine.ts | | `paretoPolicy` | none | only this package's tests: src/campaign/gates/promotion-policy.test.ts:3 | | `paretoSignificanceGate` | production | agent-app:src/eval-campaign/index.ts | | `parseEvidenceRegistryRecord` | none | only this package's tests: src/experiment/evidence-record.test.ts:4 | | `powerPreflight` | production | discovery-lab:tools/run-sequential-profile-improvement.mjs | -| `projectNLadderBudget` | production | this package: src/experiment/define.ts:21 | -| `readField` | planned | named in agent-dev-container (bind not in the import graph) | +| `projectNLadderBudget` | production | this package: src/experiment/define.ts:20 | +| `readField` | production | this package: src/experiment/claim.ts:4 | | `renderEvidenceIndex` | production | this package: scripts/render-evidence-index.ts:15 | | `renderFunnelTable` | planned | example: examples/sealed-experiment/index.ts:12 | | `requiredPairedSampleSize` | production | discovery-lab:tools/design-gate.mjs | | `requiredSampleSize` | planned | doc: docs/design/statistics-decisions.md | -| `runSelectionRule` | production | this package: src/experiment/define.ts:21 | -| `runUniformPassBudget` | production | this package: src/experiment/define.ts:21 | +| `runSelectionRule` | production | this package: src/experiment/define.ts:20 | +| `runUniformPassBudget` | production | this package: src/experiment/define.ts:20 | | `sealExperiment` | production | this package: scripts/tb-gated-stop-ab.ts:55 | | `SealIntegrityError` | none | only this package's tests: tests/experiment/budget-and-seal.test.ts:8 | | `sequentialCrossingHorizon` | none | only this package's tests: tests/sequential.test.ts:2 | | `sequentialDecide` | planned | doc: docs/experiment.md | | `sequentialPairedGate` | planned | consumer tests: legal-agent:tests/eval/self-improve.ts | | `signManifest` | production | phony:products/builder/api/src/eval/champion-sign.ts | +| `summarizeEvaluationUnits` | production | this package: src/campaign/final-evidence.ts:2 | | `validateEvidenceRegistry` | none | only this package's tests: src/experiment/evidence-record.test.ts:4 | | `verifyEvidenceReceipt` | planned | consumer tests: agent-runtime:tests/integration/runtime-eval-pursuit-evidence.test.ts | | `verifyManifest` | production | phony:products/builder/api/src/eval/champion-sign.ts | -| `verifyMatchedBudgets` | production | this package: src/experiment/define.ts:57 | +| `verifyMatchedBudgets` | production | this package: src/experiment/define.ts:58 | | `verifySealedExperiment` | planned | example: examples/sealed-experiment/index.ts:12 | | `wilson` | production | agent-dev-container:products/intelligence/api/src/lib/project-analysis-coverage-recommendations.ts | @@ -941,7 +948,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./hosted` -11 value exports — 7 production, 3 planned, 1 none. +11 value exports — 7 production, 2 planned, 2 none. | symbol | consumer | evidence | | --- | --- | --- | @@ -953,7 +960,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `IngestEvalRunsRequestSchema` | production | this package: src/hosted/client.ts:19 | | `IngestResponseSchema` | production | this package: src/hosted/client.ts:19 | | `IngestTracesRequestSchema` | production | this package: src/hosted/client.ts:19 | -| `InsightReportSchema` | planned | doc: docs/insight-report.md | +| `InsightReportSchema` | none | only this package's tests: tests/contract-analyze-runs.test.ts:29 | | `TraceSpanEventSchema` | planned | example: examples/hosted-ingest-server/server.ts:33 | | `UnixNanoTimestampSchema` | none | — | @@ -969,7 +976,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `FileLedgerJournal` | production | this package: src/campaign/search-ledger.ts:22 | | `hashCanonical` | production | discovery-lab:tools/provenance.mjs | | `jsonDocument` | production | this package: src/analyst/benchmark-command-artifact.ts:2 | -| `LEDGER_HASH_PATTERN` | production | this package: src/ledger-core/trusted-head.ts:55 | +| `LEDGER_HASH_PATTERN` | production | this package: src/campaign/final-evidence.ts:13 | | `LedgerCanonicalizationError` | none | only this package's tests: src/ledger-core/canonical.test.ts:5 | | `probeAtomicFileLock` | production | this package: src/campaign/single-run-lock.ts:17 | | `readTrustedHeadFile` | production | this package: src/ledger-core/journal.ts:29 | @@ -993,28 +1000,37 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./meta-eval` -18 value exports — 10 production, 6 planned, 2 none. +27 value exports — 13 production, 13 planned, 1 none. | symbol | consumer | evidence | | --- | --- | --- | -| `calibrationCurve` | none | only this package's tests: tests/meta-eval.test.ts:2 | -| `catchRate` | planned | doc: README.md | +| `auditEvaluator` | planned | example: examples/evaluation-integrity/index.ts:15 | +| `calibrateJudge` | production | creative-agent:eval/calibrate-judges.ts | +| `calibrateJudgeContinuous` | planned | consumer tests: tax-agent:tests/eval/lib/judge-sentinel.ts | +| `calibrationCurve` | planned | doc: docs/eval-surface-map.md | +| `calibrationFromPairs` | planned | doc: docs/eval-surface-map.md | +| `catchRate` | planned | doc: docs/eval-surface-map.md | +| `continuousAgreement` | production | agent-builder:frontier/judges/eval-validity.ts | | `correlationStudy` | production | phony:products/builder/api/src/eval/outcomes.ts | -| `definePlant` | planned | doc: docs/plants.md | +| `definePlant` | planned | doc: docs/eval-surface-map.md | | `evalHealthStamp` | production | creative-agent:eval/lib/judge-sentinel.ts | | `fileSentinelStore` | production | creative-agent:eval/lib/judge-sentinel.ts | | `FileSystemOutcomeStore` | production | phony:products/builder/api/src/eval/outcomes.ts | | `InMemoryOutcomeStore` | planned | consumer tests: phony:products/builder/api/src/eval/outcomes-correlate.test.ts | | `inMemorySentinelStore` | none | only this package's tests: tests/judge-sentinel.test.ts:7 | | `judgeSentinelReport` | production | creative-agent:eval/lib/judge-sentinel.ts | +| `OutcomeStoreError` | planned | doc: docs/outcome-validity.md | | `perturbEvidence` | planned | doc: docs/plants.md | -| `plantByPerturbation` | planned | doc: README.md | +| `plantByPerturbation` | planned | doc: docs/plants.md | +| `positionalBias` | planned | doc: docs/concepts.md | | `rubricPredictiveValidity` | production | physim:apps/server/src/scripts/agent-eval/validate-rubrics.ts | -| `seedPlants` | planned | doc: README.md | +| `seedPlants` | planned | doc: docs/eval-surface-map.md | +| `selfPreference` | planned | doc: docs/concepts.md | | `snapshotFromAgreement` | production | physim:apps/server/src/scripts/agent-eval/judge-sentinel.ts | | `snapshotFromCalibration` | production | gtm-agent:eval/calibrate-judges.ts | | `snapshotFromSentinelSet` | production | creative-agent:eval/lib/judge-sentinel.ts | | `validateSentinelSnapshot` | production | creative-agent:eval/lib/judge-sentinel.ts | +| `verbosityBias` | production | blueprint-agent:scripts/experiments/calibration/judge-agreement.ts | ### `./multishot` @@ -1127,7 +1143,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./rl` -65 value exports — 18 production, 15 planned, 32 none. +65 value exports — 18 production, 19 planned, 28 none. | symbol | consumer | evidence | | --- | --- | --- | @@ -1153,7 +1169,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `extractVerifiableRewardsFromRecords` | production | creative-agent:eval/canonical-export.ts | | `FileSystemOutcomeStore` | production | phony:products/builder/api/src/eval/outcomes.ts | | `filterDeterministicallyRewarded` | production | this package: src/rl/reward-hacking.ts:43 | -| `firstPassK` | none | only this package's tests: tests/rl-adaptation-eval.test.ts:3 | +| `firstPassK` | planned | doc: docs/statistical-evidence.md | | `fitBradleyTerry` | none | only this package's tests: tests/rl-tournament.test.ts:2 | | `injectIrrelevantClause` | none | only this package's tests: tests/rl-contamination.test.ts:2 | | `InMemoryOutcomeStore` | planned | consumer tests: phony:products/builder/api/src/eval/outcomes-correlate.test.ts | @@ -1163,16 +1179,16 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `observationsFromRunRecords` | none | only this package's tests: tests/rl-active-curriculum.test.ts:3 | | `offPolicyEstimateAll` | none | only this package's tests: tests/rl-off-policy.test.ts:3 | | `paretoFrontier` | production | agent-dev-container:products/sandbox/evals/run.ts | -| `PredictiveValidityResearcher` | none | only this package's tests: tests/rl-predictive-validity-researcher.test.ts:4 | +| `PredictiveValidityResearcher` | planned | doc: docs/design/mlbenchmarks-book-review.md | | `prmTrainingPairs` | planned | doc: examples/fine-tune-with-prime-rl/README.md | | `quantileEdges` | none | only this package's tests: src/rl/sim-fidelity.test.ts:3 | | `readCorpus` | planned | named in agent-runtime (bind not in the import graph) | | `renameVariables` | none | only this package's tests: tests/rl-contamination.test.ts:2 | -| `runAdaptationCurve` | none | only this package's tests: tests/rl-adaptation-eval.test.ts:3 | +| `runAdaptationCurve` | planned | doc: docs/statistical-evidence.md | | `runComputeCurve` | none | only this package's tests: tests/rl-adversarial-and-compute.test.ts:2 | -| `runContaminationProbe` | none | only this package's tests: tests/rl-contamination.test.ts:2 | +| `runContaminationProbe` | planned | doc: docs/statistical-evidence.md | | `runEvalCampaign` | production | agent-builder:scripts/eval.ts | -| `runRLCampaign` | planned | named in creative-agent (bind not in the import graph) | +| `runRLCampaign` | planned | doc: docs/outcome-validity.md | | `runwiseStepRewardSummary` | none | only this package's tests: tests/rl-process-reward.test.ts:3 | | `selfConsistency` | none | only this package's tests: tests/rl-adversarial-and-compute.test.ts:2 | | `selfNormalizedImportanceWeighting` | planned | consumer tests: agent-knowledge:tests/memory-holdout.test.ts | @@ -1199,26 +1215,26 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./rollout` -67 value exports — 49 production, 8 planned, 10 none. +67 value exports — 42 production, 15 planned, 10 none. | symbol | consumer | evidence | | --- | --- | --- | | `addScrubCounts` | production | this package: src/rollout/release/hf-dataset.ts:34 | -| `appendRolloutLines` | production | agent-runtime:bench/src/rollout-ledger/settle-capture.mts | +| `appendRolloutLines` | planned | doc: docs/rollout.md | | `assertGateReport` | production | this package: src/rollout/release/card.ts:12 | | `assertMinted` | production | blueprint-agent:scripts/experiments/lib/steering-dataset.ts | -| `assertMintedLines` | planned | consumer tests: agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts | +| `assertMintedLines` | planned | doc: docs/rollout.md | | `assertRolloutLine` | production | this package: src/rollout/interchange/harbor.ts:60 | | `ATIF_SCHEMA_VERSION` | none | only this package's tests: src/rollout/interchange/harbor.test.ts:6 | | `buildDatasetCard` | production | this package: src/rollout/release/hf-dataset.ts:25 | | `buildHfDataset` | production | supervisor-lab:bench/vertical-rollout-package.ts | -| `claudeProjectSlug` | planned | consumer tests: agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts | -| `DEFAULT_CLAUDE_PROJECTS_DIR` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | -| `DEFAULT_OPENCODE_DB` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `claudeProjectSlug` | planned | named in agent-runtime (bind not in the import graph) | +| `DEFAULT_CLAUDE_PROJECTS_DIR` | planned | named in agent-runtime (bind not in the import graph) | +| `DEFAULT_OPENCODE_DB` | planned | named in agent-runtime (bind not in the import graph) | | `defaultRolloutScrubber` | production | supervisor-lab:bench/vertical-rollout-package.ts | | `emptyScrubCounts` | production | this package: src/rollout/release/hf-dataset.ts:34 | -| `findClaudeTranscripts` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | -| `findOpencodeSessionsByDirectory` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `findClaudeTranscripts` | planned | doc: docs/rollout.md | +| `findOpencodeSessionsByDirectory` | planned | named in agent-runtime (bind not in the import graph) | | `FORMAT_FILES` | production | this package: src/rollout/release/hf-dataset.ts:25 | | `FORMAT_GATE_DISPOSITION` | production | this package: src/rollout/release/card.ts:12 | | `fromHarborTrajectory` | planned | doc: docs/rollout.md | @@ -1237,11 +1253,11 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `mintRolloutRows` | production | agent-builder:src/lib/.server/eval/loops/auto-research-runner.ts | | `observedScore` | production | this package: src/rl/active-curriculum.ts:34 | | `observedSplitScore` | production | starter-foundry:src/lib/held-out-gate.ts | -| `openOpencodeDb` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `openOpencodeDb` | production | discovery-lab:dashboard/lib/transcript.mjs | | `parseRolloutReleaseArgs` | none | only this package's tests: src/rollout/release/hf-dataset.test.ts:9 | | `planPushCommand` | none | only this package's tests: src/rollout/release/hf-dataset.test.ts:9 | -| `readClaudeTranscript` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | -| `readOpencodeSessionMessages` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `readClaudeTranscript` | planned | doc: docs/rollout.md | +| `readOpencodeSessionMessages` | planned | doc: docs/rollout.md | | `readRolloutJournal` | none | only this package's tests: src/rollout/ledger.test.ts:6 | | `readRolloutLedger` | production | this package: src/rollout/release/hf-dataset.ts:23 | | `relabelImportedSplit` | planned | doc: docs/rollout.md | @@ -1249,7 +1265,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `releaseRowRefs` | production | this package: src/rollout/release/hf-dataset.ts:26 | | `ROLLOUT_CAPTURES` | production | this package: src/rollout/interchange/harbor.ts:60 | | `ROLLOUT_ROLES` | production | this package: src/rollout/interchange/harbor.ts:60 | -| `ROLLOUT_SCHEMA` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `ROLLOUT_SCHEMA` | production | this package: src/rollout/interchange/harbor.ts:60 | | `ROLLOUT_SPLITS` | production | this package: src/rollout/interchange/harbor.ts:60 | | `runRolloutReleaseCli` | production | this package: src/cli.ts:19 | | `scoreOrigin` | production | this package: src/rollout/mint.ts:38 | @@ -1269,7 +1285,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `trainingScore` | production | this package: src/eval-trace-store.ts:20 | | `unmintableReasons` | production | supervisor-lab:bench/mint-audit.ts | | `validateRolloutLine` | production | this package: src/supervisor-run/integrity-tree.ts:1 | -| `writeRolloutLedger` | production | agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts | +| `writeRolloutLedger` | production | this package: src/rollout/release/hf-dataset.ts:23 | ### `./storyboard` @@ -1289,7 +1305,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m ### `./supervisor-run` -30 value exports — 23 production, 2 planned, 5 none. +30 value exports — 20 production, 5 planned, 5 none. | symbol | consumer | evidence | | --- | --- | --- | @@ -1309,8 +1325,8 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `readTerminalRecord` | production | this package: src/supervisor-run/analyze.ts:18 | | `renderSupervisorRollupMarkdown` | production | traces:src/cli.ts | | `renderSupervisorRunHeadline` | production | discovery-lab:experiments/0005-run-visibility/probe.mjs | -| `renderSupervisorRunMarkdown` | production | agent-runtime:bench/src/swe-arena/run-report.mts | -| `reportSupervisorRound` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `renderSupervisorRunMarkdown` | production | discovery-lab:experiments/0005-run-visibility/render-run.mjs | +| `reportSupervisorRound` | planned | named in agent-runtime (bind not in the import graph) | | `rollupSupervisorRuns` | production | discovery-lab:tools/disco.mjs | | `runSupervisorRunCommand` | production | this package: src/cli.ts:20 | | `RUNTIME_FAILED_STATUS` | none | — | @@ -1321,8 +1337,8 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `SUPERVISOR_RUN_SCHEMA` | production | this package: src/supervisor-run/analyze.ts:19 | | `supervisorRunRolloutLines` | planned | consumer tests: discovery-lab:tools/runtime-journal-eval.test.mjs | | `unavailable` | production | this package: src/supervisor-run/analyze.ts:19 | -| `writeSupervisorRunReport` | production | agent-runtime:bench/src/swe-arena/run-report.mts | -| `writeSupervisorRunReportSafe` | production | agent-runtime:bench/src/swe-arena/outer-loop.mts | +| `writeSupervisorRunReport` | planned | named in agent-runtime (bind not in the import graph) | +| `writeSupervisorRunReportSafe` | planned | named in agent-runtime (bind not in the import graph) | ### `./trace-attributes` @@ -1432,7 +1448,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `REPAIR_CONTRACT_LINES` | production | this package: src/trace-repair/arm-completion.ts:37 | | `REPAIR_QUESTION` | production | this package: src/trace-repair/arm-completion.ts:37 | | `REPAIR_REPAIR_CONTRACT_LINES` | production | this package: src/trace-repair/arm-completion.ts:37 | -| `repairArmAsymmetries` | planned | doc: docs/charter.md | +| `repairArmAsymmetries` | planned | doc: docs/trace-repair-analyst-arms.md | | `repairArmPromptSha256` | production | this package: src/trace-repair/analyst-arm.ts:42 | | `repairArmResponse` | planned | doc: docs/trace-repair-analyst-arms.md | | `repairCredit` | production | this package: src/trace-repair/grade.ts:40 | @@ -1572,7 +1588,7 @@ The `none` set was reviewed symbol by symbol on 2026-08-21. Four rules decided m | `TRACE_ANALYST_TOOL_NAMESPACE` | none | only this package's tests: src/trace-analyst/tools.test.ts:4 | | `TRACE_ANALYST_TRUNCATION_MARKER_PREFIX` | production | agent-runtime:src/analyst-loop/iterations-to-trace-store.ts | | `TRACE_SCHEMA_VERSION` | production | skeletal-os:src/eval/agentEvalBridge.ts | -| `TraceAnalysisLimitError` | production | this package: src/trace-analyst/store-bounds.ts:3 | +| `TraceAnalysisLimitError` | production | this package: src/trace-analyst/store-boundary.ts:1 | | `TraceAnalysisStoreContractError` | production | this package: src/trace-analyst/store-boundary.ts:1 | | `TraceAnalysisValidationError` | production | this package: src/trace-analyst/store-boundary.ts:1 | | `traceAnalystFunctionGroup` | none | only this package's tests: src/trace-analyst/tools.test.ts:4 | diff --git a/scripts/public-api-consumers.json b/scripts/public-api-consumers.json index 675e7b64..102bbc4c 100644 --- a/scripts/public-api-consumers.json +++ b/scripts/public-api-consumers.json @@ -517,9 +517,7 @@ ] }, "appendRolloutLines": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -575,13 +573,10 @@ }, "assertCampaignSplitIdentity": { "production": [ - "agent-knowledge:src/memory/experiment/learning-pairs.ts", - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts" + "agent-knowledge:src/memory/experiment/learning-pairs.ts" ], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.test.mts" - ], + "tests": [], "mentions": [ "agent-knowledge", "agent-runtime" @@ -672,9 +667,7 @@ "assertMintedLines": { "production": [], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts" - ], + "tests": [], "mentions": [ "agent-runtime" ] @@ -770,9 +763,7 @@ "assertRolloutLine": { "production": [], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts" - ], + "tests": [], "mentions": [ "agent-runtime" ] @@ -1367,7 +1358,6 @@ }, "campaignScenarioIdentity": { "production": [ - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts", "agent-runtime:src/improvement/method-identity.ts", "discovery-lab:tools/strict-screen.mjs" ], @@ -1384,12 +1374,10 @@ }, "campaignSplitDigest": { "production": [ - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts", "discovery-lab:tools/profile-population.mjs" ], "typeOnly": [], "tests": [ - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.test.mts", "discovery-lab:tools/run-sequential-profile-improvement.test.mjs" ], "mentions": [ @@ -1700,9 +1688,7 @@ "claudeProjectSlug": { "production": [], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts" - ], + "tests": [], "mentions": [ "agent-runtime" ] @@ -2505,8 +2491,6 @@ "agent-knowledge:src/memory/experiment/learning.ts", "agent-knowledge:src/memory/experiment/run.ts", "agent-runtime:bench/src/quant-arena/quant-loop.mts", - "agent-runtime:bench/src/swe-arena/gepa-seat.mts", - "agent-runtime:bench/src/swe-arena/ledger-orphans.mts", "supervisor-lab:bench/drain/seat.ts" ], "typeOnly": [], @@ -2516,8 +2500,7 @@ "agent-knowledge:tests/benchmarks/recovery.test.ts", "agent-knowledge:tests/memory/experiment-cost.test.ts", "agent-knowledge:tests/memory/experiment-learning.test.ts", - "agent-knowledge:tests/memory/experiment-recovery.test.ts", - "agent-runtime:bench/src/swe-arena/ledger-orphans.test.mts" + "agent-knowledge:tests/memory/experiment-recovery.test.ts" ], "mentions": [ "agent-dev-container", @@ -2592,9 +2575,7 @@ ] }, "DEFAULT_CLAUDE_PROJECTS_DIR": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -2620,10 +2601,7 @@ ] }, "DEFAULT_OPENCODE_DB": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts", - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -3748,9 +3726,7 @@ ] }, "findClaudeTranscripts": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -3810,10 +3786,7 @@ ] }, "findOpencodeSessionsByDirectory": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts", - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -3910,16 +3883,13 @@ "agent-knowledge:src/memory/experiment/run.ts", "agent-knowledge:src/memory/improvement/run.ts", "agent-runtime:bench/src/quant-arena/quant-loop.mts", - "agent-runtime:bench/src/swe-arena/gepa-seat.mts", - "agent-runtime:bench/src/swe-arena/ledger-orphans.mts", "ai-trading-blueprint:evals/src/sim/multishot-user-sim.ts", "discovery-lab:tools/strict-screen.mjs", "supervisor-lab:bench/drain/seat.ts" ], "typeOnly": [], "tests": [ - "agent-dev-container:products/intelligence/api/tests/evals.test.ts", - "agent-runtime:bench/src/swe-arena/ledger-orphans.test.mts" + "agent-dev-container:products/intelligence/api/tests/evals.test.ts" ], "mentions": [ "agent-dev-container", @@ -3931,9 +3901,7 @@ ] }, "FsLabeledScenarioStore": { - "production": [ - "agent-runtime:bench/src/swe-arena/outer-loop.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -4011,7 +3979,6 @@ "agent-app:src/eval-campaign/index.ts", "agent-builder:src/lib/.server/eval/loops/evolution-runner.ts", "agent-dev-container:products/intelligence/api/src/lib/optimization-primitives.ts", - "agent-runtime:bench/src/swe-arena/gepa-seat.mts", "agent-runtime:src/improvement/official-optimizers.ts", "creative-agent:eval/scripts/self-improve.ts", "physim:apps/server/src/scripts/agent-eval/self-improve.ts" @@ -5230,7 +5197,6 @@ "production": [ "agent-dev-container:products/intelligence/api/src/lib/consultant/candidates.ts", "agent-dev-container:products/intelligence/api/src/lib/miner/worker.ts", - "agent-runtime:bench/src/swe-arena/diagnosis-ensemble.ts", "agent-runtime:src/runtime/index.ts", "agent-runtime:src/runtime/supervise/authoring.ts", "creative-agent:eval/autoresearch-canonical.ts", @@ -5256,7 +5222,6 @@ }, "makeProposalFinding": { "production": [ - "agent-runtime:bench/src/swe-arena/outer-loop.mts", "agent-runtime:examples/improve/improve.ts", "agent-runtime:examples/intelligence-recommend/intelligence-recommend.ts", "agent-runtime:src/improvement/code-execution.ts", @@ -5268,8 +5233,6 @@ ], "typeOnly": [], "tests": [ - "agent-runtime:bench/src/swe-arena/outer-loop.test.mts", - "agent-runtime:bench/src/swe-arena/proposer-fanout.test.mts", "agent-runtime:src/improvement/improve.test.ts", "agent-runtime:src/improvement/method-identity.test.ts", "agent-runtime:src/improvement/official-optimizers.test.ts", @@ -5332,7 +5295,6 @@ }, "mcnemar": { "production": [ - "agent-runtime:bench/src/swe-arena/analyze.ts", "blueprint-agent:scripts/experiments/analyze-convergence.ts", "blueprint-agent:scripts/experiments/lib/validity-gates.ts", "starter-foundry:src/lib/held-out-gate.ts", @@ -5746,8 +5708,6 @@ }, "openOpencodeDb": { "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts", - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts", "discovery-lab:dashboard/lib/transcript.mjs" ], "typeOnly": [], @@ -6483,9 +6443,7 @@ ] }, "readClaudeTranscript": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -6554,10 +6512,7 @@ ] }, "readOpencodeSessionMessages": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts", - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -6578,9 +6533,7 @@ "readRolloutLedger": { "production": [], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/rollout-ledger/settle-capture.test.mts" - ], + "tests": [], "mentions": [ "agent-runtime" ] @@ -6926,7 +6879,6 @@ }, "renderSupervisorRunMarkdown": { "production": [ - "agent-runtime:bench/src/swe-arena/run-report.mts", "discovery-lab:experiments/0005-run-visibility/render-run.mjs", "traces:src/cli.ts" ], @@ -6986,11 +6938,7 @@ ] }, "reportSupervisorRound": { - "production": [ - "agent-runtime:bench/src/swe-arena/outer-loop.mts", - "agent-runtime:bench/src/swe-arena/run-experiment.mts", - "agent-runtime:bench/src/swe-arena/run-report.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -7147,10 +7095,7 @@ ] }, "ROLLOUT_SCHEMA": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts", - "agent-runtime:bench/src/rollout-ledger/settle-capture.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -8388,8 +8333,6 @@ "agent-knowledge:src/memory/improvement/run.ts", "agent-knowledge:src/optimization.ts", "agent-knowledge:src/retrieval-optimization.ts", - "agent-runtime:bench/src/swe-arena/outer-loop.mts", - "agent-runtime:bench/src/swe-arena/premeasured-from-cells.mts", "blueprint-agent:scripts/experiments/lib/gepa-reflective-proposer.ts", "discovery-lab:tools/run-sequential-profile-improvement.mjs", "discovery-lab:tools/run-strict-100-funnel.mjs", @@ -8537,9 +8480,7 @@ "toSftRows": { "production": [], "typeOnly": [], - "tests": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.test.mts" - ], + "tests": [], "mentions": [ "agent-runtime" ] @@ -9258,9 +9199,7 @@ ] }, "writeRolloutLedger": { - "production": [ - "agent-runtime:bench/src/rollout-ledger/backfill-swe-arena.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -9268,9 +9207,7 @@ ] }, "writeSupervisorRunReport": { - "production": [ - "agent-runtime:bench/src/swe-arena/run-report.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [ @@ -9278,10 +9215,7 @@ ] }, "writeSupervisorRunReportSafe": { - "production": [ - "agent-runtime:bench/src/swe-arena/outer-loop.mts", - "agent-runtime:bench/src/swe-arena/run-experiment.mts" - ], + "production": [], "typeOnly": [], "tests": [], "mentions": [