From bfb86e7eb711c0a1c1326370c0f8f03a96c493a1 Mon Sep 17 00:00:00 2001 From: Josh Owens Date: Sat, 29 Aug 2026 12:11:20 -0400 Subject: [PATCH] feat: turn caps (--max-turns) and raw per-run results (--raw-out) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two additions for token-level experiments (#33, #34): maxTurns (scenario field, per-case override, --max-turns on run/compare/ measure) passes --max-turns to the claude-p runner. A run that stops at the cap (subtype error_max_turns, any exit code) is returned as a measured outcome flagged exhaustedTurns instead of thrown as a crash, and the engine scores it as a failed run without grading — partial output must not pass by accident, and a judge call on a truncated run is money spent on a result that cannot stand. The cap joins the baseline cache key (undefined caps keep their pre-existing keys). --raw-out on compare and measure writes each run's full runner result JSON (usage token categories, modelUsage, subtype) as __.json the moment the run finishes, so a partial invocation still leaves its completed records. Summaries and reports keep digesting to pass/cost; this is the escape hatch for analyses that need the token categories rather than the cost blend. Cache-served baseline arms ran in an earlier invocation and write nothing (noted in progress output). Verified against claude CLI 2.x on macOS: --max-turns is accepted in -p mode and a capped run returns subtype "error_max_turns" with full result JSON. Fixes #33 Fixes #34 Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01EqktLpWgVX5orcGAzx7EK1 --- README.md | 22 ++++++++ SPEC.md | 15 +++++- src/cli.ts | 43 ++++++++++++++-- src/engine/cache.ts | 5 ++ src/engine/compare.ts | 64 +++++++++++++++++++----- src/engine/config.ts | 14 ++++++ src/runner/claude-p.ts | 24 ++++++--- src/types.ts | 8 +++ test/compare.test.ts | 111 +++++++++++++++++++++++++++++++++++++++++ test/config.test.ts | 36 +++++++++++++ test/runner.test.ts | 91 ++++++++++++++++++++++++++++++++- 11 files changed, 408 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index 6c5a6c9..2d85007 100644 --- a/README.md +++ b/README.md @@ -87,6 +87,13 @@ runs need headroom. Set `maxBudgetUsd` to `3` or more for artifact scenarios. A run that hits the cap fails with an explicit `claude hit the $N max budget` error rather than a silent bad sample. +**Turn cap:** `--max-turns ` (or `"maxTurns"` in a scenario file, with a +per-case override) caps agentic turns per run on the claude-p runner. Unlike +a budget abort, a turn-capped run is a *measured outcome*: it is recorded and +scored as a failed run without grading its partial output — an answer the +agent stumbled into at the cap must not count as a cheap pass. Completion +runners (openai) are single-turn and ignore the cap. + ## Runners `--runner claude-p` (default) shells out to headless Claude Code and supports @@ -285,6 +292,21 @@ Reading results: model gets a warning on the summary — a pass on the wrong model validates prompt logic, not production behavior. +### Raw per-run results + +Summaries, receipts, and the ndjson report digest each run down to pass/cost — +the right altitude for verdicts, and the wrong one for token economics. +`--raw-out ` (on `compare` and `measure`) writes each run's **full runner +result JSON** as `__.json` the moment the run finishes: for +claude-p that includes `usage` (input, cache-creation, cache-read, and output +tokens), per-model `modelUsage`, and the result `subtype`. Cost-USD is a +weighted sum of those categories (cache reads are priced far below fresh +input), so experiments whose *question* is token behavior — did a prompt +change cut exploration tokens, or just shift reads into cache? — need the +categories, not the blend. Files are written per run, so a compare that dies +midway still leaves the completed runs' records. Cache-served baseline arms +ran in an earlier invocation and write nothing. + ### Run history `--report ndjson --report-out ./runs.ndjson` appends one record per scenario diff --git a/SPEC.md b/SPEC.md index 95362b4..5243769 100644 --- a/SPEC.md +++ b/SPEC.md @@ -58,7 +58,8 @@ claude -p "" \ --model \ --output-format json \ --tools \ - --max-budget-usd + --max-budget-usd \ + [--max-turns ] ``` The spawned process runs with the per-run sandbox as its actual working @@ -68,6 +69,12 @@ used as a substitute for sandboxing. The runner captures final output text, `total_cost_usd`, turn count, duration, and model usage keys. +A `maxTurns` cap (scenario field, per-case override, or `--max-turns`) is +passed through as `--max-turns`. A run that stops at the cap (result subtype +`error_max_turns`, on any exit code) is returned as a normal result flagged +`exhaustedTurns` — the engine records it and scores it as a failed run +without invoking the grader, so partial output cannot pass by accident. + ### 3.2 `openai` Sends one `POST /chat/completions` request to any OpenAI-compatible @@ -183,7 +190,11 @@ each run) recording the content hash of every prompt file tested plus the verdict — content addressing instead of hand-maintained prompt_version strings, so a consuming repo's CI can require a passing receipt for each shipped prompt's current hash. Reports are history; receipts are current -state. +state. `--raw-out ` (compare and measure) additionally writes each run's +full runner result JSON — token usage categories, modelUsage, subtype — one +file per completed run, for token-level analysis that the digested summaries +cannot support; files land as runs finish, so a partial invocation still +leaves its completed records. `compare` scenarios assert nothing: they exist to report both arms' pass rates and the delta, for comparisons (typically model-vs-model) where neither diff --git a/src/cli.ts b/src/cli.ts index e66401f..25522c4 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -36,6 +36,7 @@ const runSpecs: FlagSpecs = { tools: { arity: "one" }, "timeout-ms": { arity: "one" }, "max-budget-usd": { arity: "one" }, + "max-turns": { arity: "one" }, "keep-sandbox": { arity: "none" }, "clean-sandbox": { arity: "none" }, }; @@ -63,10 +64,12 @@ const compareSpecs: FlagSpecs = { tools: { arity: "one" }, "timeout-ms": { arity: "one" }, "max-budget-usd": { arity: "one" }, + "max-turns": { arity: "one" }, "keep-sandbox": { arity: "none" }, report: { arity: "one" }, "report-out": { arity: "one" }, receipts: { arity: "one" }, + "raw-out": { arity: "one" }, cache: { arity: "none" }, "cache-dir": { arity: "one" }, }; @@ -196,6 +199,7 @@ async function cmdRun(argv: string[]): Promise { tools, timeoutMs: args.number("timeout-ms", DEFAULT_TIMEOUT_MS), maxBudgetUsd: args.number("max-budget-usd", DEFAULT_MAX_BUDGET_USD), + maxTurns: maxTurnsFromArgs(args.has("max-turns") ? args.number("max-turns", 0) : undefined), }; console.error( @@ -233,8 +237,10 @@ const measureSpecs: FlagSpecs = { tools: { arity: "one" }, "timeout-ms": { arity: "one" }, "max-budget-usd": { arity: "one" }, + "max-turns": { arity: "one" }, "keep-sandbox": { arity: "none" }, receipts: { arity: "one" }, + "raw-out": { arity: "one" }, }; async function cmdMeasure(argv: string[]): Promise { @@ -256,6 +262,7 @@ async function cmdMeasure(argv: string[]): Promise { runs: args.has("runs") ? args.number("runs", 0) : undefined, timeoutMs: args.has("timeout-ms") ? args.number("timeout-ms", DEFAULT_TIMEOUT_MS) : undefined, maxBudgetUsd: args.has("max-budget-usd") ? args.number("max-budget-usd", DEFAULT_MAX_BUDGET_USD) : undefined, + maxTurns: maxTurnsFromArgs(args.has("max-turns") ? args.number("max-turns", 0) : undefined), mode: args.one("mode") ? modeFromString(args.one("mode")) : undefined, tools: args.one("tools"), addDirs: args.many("add-dir").length ? args.many("add-dir") : undefined, @@ -264,11 +271,13 @@ async function cmdMeasure(argv: string[]): Promise { keepSandbox: args.has("keep-sandbox") ? true : undefined, }; + const rawOutDir = args.one("raw-out"); const config = loadCompareConfig(scenario, overrides, { singleArm: true }); const summary = await runMeasure({ config, runner: armRunner(config.arms.baseline, config), onProgress: (message) => console.error(`[promptdiff] ${message}`), + rawOut: rawOutDir === undefined ? undefined : { dir: rawOutDir }, }); const receiptsDir = args.one("receipts"); @@ -350,6 +359,7 @@ async function cmdCompare(argv: string[]): Promise { runs: args.has("runs") ? args.number("runs", 0) : undefined, timeoutMs: args.has("timeout-ms") ? args.number("timeout-ms", DEFAULT_TIMEOUT_MS) : undefined, maxBudgetUsd: args.has("max-budget-usd") ? args.number("max-budget-usd", DEFAULT_MAX_BUDGET_USD) : undefined, + maxTurns: maxTurnsFromArgs(args.has("max-turns") ? args.number("max-turns", 0) : undefined), mode: args.one("mode") ? modeFromString(args.one("mode")) : undefined, tools: args.one("tools"), addDirs: args.many("add-dir").length ? args.many("add-dir") : undefined, @@ -358,6 +368,7 @@ async function cmdCompare(argv: string[]): Promise { keepSandbox: args.has("keep-sandbox") ? true : undefined, }; + const rawOutDir = args.one("raw-out"); const config = loadCompareConfig(scenario, overrides); const summary = await runCompare({ config, @@ -367,6 +378,7 @@ async function cmdCompare(argv: string[]): Promise { }, onProgress: (message) => console.error(`[promptdiff] ${message}`), cache, + rawOut: rawOutDir === undefined ? undefined : { dir: rawOutDir }, }); // History is appended before the exit code is decided — failed comparisons @@ -395,6 +407,14 @@ function promptFromArgs(prompt: string | undefined, promptFile: string | undefin throw new CliError(runUsage()); } +function maxTurnsFromArgs(value: number | undefined): number | undefined { + if (value === undefined) return undefined; + if (!Number.isInteger(value) || value < 1) { + throw new CliError("--max-turns must be an integer of at least 1"); + } + return value; +} + function modeFromString(value: string | undefined): RunMode { if (value === "text" || value === "artifact") return value; throw new CliError("--mode must be either text or artifact"); @@ -475,6 +495,7 @@ function runUsage(): string { " --runner --base-url --image ...", " --var ... --sandbox --seed ", " --tools --timeout-ms --max-budget-usd ", + " --max-turns (claude-p: cap agentic turns per run)", "", "templates:", " --var draft=./fixture.md binds {{draft}} in the agent, skills, and prompt;", @@ -522,9 +543,20 @@ function compareUsage(): string { " --baseline-model --proposed-model ", " --baseline-runner --proposed-runner ", " --mode --tools ", - " --timeout-ms --max-budget-usd ", + " --timeout-ms --max-budget-usd --max-turns ", " --report ndjson --report-out ", - " --cache [--cache-dir ]", + " --raw-out --cache [--cache-dir ]", + "", + "turn cap:", + " --max-turns (or scenario \"maxTurns\", per-case override allowed) caps", + " agentic turns per run (claude-p). A run that hits the cap is scored as a", + " FAILED run without grading — partial output must not pass by accident.", + "", + "raw results:", + " --raw-out writes each run's full runner result JSON (token usage,", + " modelUsage, subtype) as __.json, one file per completed", + " run — the token-level record that summaries and reports digest away.", + " Cache-served baseline arms ran earlier and write nothing here.", "", "caching:", " --cache reuses recorded baseline-arm results (default dir .promptdiff/cache)", @@ -631,11 +663,16 @@ function measureUsage(): string { " --runner --base-url --runs ", " --mode --tools ", " --sandbox --seed --keep-sandbox", - " --timeout-ms --max-budget-usd --receipts ", + " --timeout-ms --max-budget-usd --max-turns ", + " --receipts --raw-out ", "", "--receipts writes one .receipt.json per scenario with", "per-file prompt hashes and the measured rates (verdict \"measured\").", "", + "--max-turns caps agentic turns per run (claude-p); a capped run is", + "scored as a failure without grading. --raw-out writes each run's", + "full runner result JSON (token usage included) as _measure_.json.", + "", "Exit code is 0 whenever the runs complete — a measurement has no pass/fail.", ].join("\n"); } diff --git a/src/engine/cache.ts b/src/engine/cache.ts index a01cf43..7378483 100644 --- a/src/engine/cache.ts +++ b/src/engine/cache.ts @@ -20,6 +20,8 @@ export interface CacheKeyInput { arm: ArmConfig; /** Effective run count for the case (case runs ?? config runs). */ runs: number; + /** Effective turn cap for the case (case maxTurns ?? config maxTurns), if any. */ + maxTurns?: number; /** Effective tools string for the case. */ tools: string; /** Effective run mode for the case. */ @@ -53,6 +55,9 @@ export function buildCacheKey(input: CacheKeyInput): string { runner: input.arm.runner, baseUrl: input.arm.baseUrl ?? "none", runs: input.runs, + // stableStringify drops undefined entries, so scenarios without a cap keep + // their pre-maxTurns cache keys. + maxTurns: input.maxTurns, tools: input.tools, mode: input.mode, delivery: input.delivery, diff --git a/src/engine/compare.ts b/src/engine/compare.ts index 35c1ada..4c8ac3e 100644 --- a/src/engine/compare.ts +++ b/src/engine/compare.ts @@ -1,4 +1,6 @@ import { createHash } from "node:crypto"; +import { mkdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; import { assembleSystemPrompt } from "../prompt"; import type { Runner, RunnerRunOptions, RunResult } from "../types"; import { buildCacheKey, loadCachedArm, storeCachedArm } from "./cache"; @@ -59,10 +61,17 @@ export interface CompareRunOptions { onProgress?: (message: string) => void; /** Opt-in baseline-arm result cache; hits skip the baseline runs entirely. */ cache?: { dir: string }; + /** + * Opt-in per-run raw persistence: each completed run's full runner result + * (token usage included) is written to this directory as it finishes. + * Cache-served baseline arms ran in an earlier invocation, so they write + * nothing here. + */ + rawOut?: { dir: string }; } export async function runCompare(options: CompareRunOptions): Promise { - const { config, runners, onProgress, cache } = options; + const { config, runners, onProgress, cache, rawOut } = options; // Validate each arm against its own runner — a mixed comparison fails on the // violating arm before either arm's paid runs. validateRunnerSupport(config, runners.baseline); @@ -80,8 +89,8 @@ export async function runCompare(options: CompareRunOptions): Promise void; + /** Opt-in per-run raw persistence; see CompareRunOptions.rawOut. */ + rawOut?: { dir: string }; } /** @@ -286,7 +297,7 @@ export interface MeasureRunOptions { * nonsense verdicts from sampling noise. */ export async function runMeasure(options: MeasureRunOptions): Promise { - const { config, runner, onProgress } = options; + const { config, runner, onProgress, rawOut } = options; validateRunnerSupport(config, runner); assertJudgeGradersCalibrated(config); const inline = config.delivery !== "install"; @@ -304,7 +315,7 @@ export async function runMeasure(options: MeasureRunOptions): Promise void, + rawOut?: { dir: string }, ): Promise { if (cache === undefined) { - return runArm(config, evalCase, "baseline", systemPrompt, runner, onProgress); + return runArm(config, evalCase, "baseline", systemPrompt, runner, onProgress, "baseline", rawOut); } const key = buildCacheKey({ systemPrompt, casePrompt: evalCase.prompt, arm: config.arms.baseline, runs: evalCase.runs ?? config.runs, + maxTurns: evalCase.maxTurns ?? config.maxTurns, tools: effectiveTools(config, evalCase), mode: effectiveMode(config, evalCase), delivery: config.delivery, @@ -405,9 +418,12 @@ async function runBaselineArm( const hit = loadCachedArm(cache.dir, key); if (hit !== undefined) { onProgress?.(` baseline: cache hit (${key.slice(0, 12)})`); + if (rawOut !== undefined) { + onProgress?.(" baseline: cache hit — no raw results to write"); + } return { ...hit, cached: true }; } - const summary = await runArm(config, evalCase, "baseline", systemPrompt, runner, onProgress); + const summary = await runArm(config, evalCase, "baseline", systemPrompt, runner, onProgress, "baseline", rawOut); storeCachedArm(cache.dir, key, summary); return summary; } @@ -420,6 +436,7 @@ async function runArm( runner: Runner, onProgress?: (message: string) => void, label: string = arm, + rawOut?: { dir: string }, ): Promise { const runs = evalCase.runs ?? config.runs; const summaries: ArmRunSummary[] = []; @@ -445,12 +462,20 @@ async function runArm( } } const run = await runner.run(buildRunnerOptions(config, evalCase, arm, systemPrompt, sandbox.dir)); - const grade = await gradeRun(evalCase.grader, { - run, - sandboxDir: sandbox.dir, - timeoutMs: config.timeoutMs, - maxBudgetUsd: config.maxBudgetUsd, - }); + if (rawOut !== undefined) { + writeRawResult(rawOut.dir, evalCase.name, label, index + 1, run.raw); + } + // A capped run did not finish; grading its partial output would let an + // accidental-looking answer count as a cheap pass (and bill a judge call + // for a result that cannot stand either way). + const grade = run.exhaustedTurns + ? { pass: false, message: `hit the ${evalCase.maxTurns ?? config.maxTurns}-turn cap before finishing` } + : await gradeRun(evalCase.grader, { + run, + sandboxDir: sandbox.dir, + timeoutMs: config.timeoutMs, + maxBudgetUsd: config.maxBudgetUsd, + }); summaries.push(toArmRunSummary(index + 1, run, grade, sandbox.dir)); } finally { sandbox.cleanup(); @@ -488,9 +513,22 @@ function buildRunnerOptions( tools: effectiveTools(config, evalCase), timeoutMs: config.timeoutMs, maxBudgetUsd: config.maxBudgetUsd, + maxTurns: evalCase.maxTurns ?? config.maxTurns, }; } +/** + * One file per completed run, written as the run finishes — the full runner + * result (usage, modelUsage, subtype for claude-p) survives even if a later + * run aborts the invocation. Summaries deliberately keep only the digest; + * this is the escape hatch for token-level analysis. + */ +function writeRawResult(dir: string, caseName: string, label: string, runNumber: number, raw: unknown): void { + mkdirSync(dir, { recursive: true }); + const safeCase = caseName.toLowerCase().replace(/[^a-z0-9_-]+/g, "-").replace(/^-+|-+$/g, "") || "scenario"; + writeFileSync(join(dir, `${safeCase}_${label}_${runNumber}.json`), JSON.stringify(raw, null, 2) + "\n", "utf8"); +} + function effectiveTools(config: CompareConfig, evalCase: EvalCaseConfig): string { // Install delivery always needs tools (the Skill tool does the triggering), // even when the grader is text-only and inline delivery would disable them. diff --git a/src/engine/config.ts b/src/engine/config.ts index c2213ef..f5793ac 100644 --- a/src/engine/config.ts +++ b/src/engine/config.ts @@ -24,6 +24,7 @@ export interface EvalCaseConfig { addDirs: string[]; mode?: RunMode; tools?: string; + maxTurns?: number; } /** Model/runner/endpoint one arm runs against, resolved from per-arm and shared fields. */ @@ -61,6 +62,8 @@ export interface CompareConfig { runs: number; timeoutMs: number; maxBudgetUsd: number; + /** Turn cap per run (claude-p); a run that hits it is scored as a failure, not graded. */ + maxTurns?: number; mode?: RunMode; tools?: string; addDirs: string[]; @@ -85,6 +88,7 @@ export interface CompareOverrides { runs?: number; timeoutMs?: number; maxBudgetUsd?: number; + maxTurns?: number; mode?: RunMode; tools?: string; addDirs?: string[]; @@ -113,6 +117,7 @@ interface RawCompareConfig { runs?: unknown; timeoutMs?: unknown; maxBudgetUsd?: unknown; + maxTurns?: unknown; mode?: unknown; tools?: unknown; addDirs?: unknown; @@ -136,6 +141,7 @@ interface RawCase { addDirs?: unknown; mode?: unknown; tools?: unknown; + maxTurns?: unknown; } export interface LoadOptions { @@ -201,6 +207,7 @@ export function loadCompareConfig( runs: overrides.runs ?? numberValue(raw.runs, 5), timeoutMs: overrides.timeoutMs ?? numberValue(raw.timeoutMs, 600_000), maxBudgetUsd: overrides.maxBudgetUsd ?? numberValue(raw.maxBudgetUsd, 1), + maxTurns: overrides.maxTurns ?? optionalNumber(raw.maxTurns, "maxTurns"), mode: overrides.mode ?? modeValue(raw.mode, undefined), tools: overrides.tools ?? stringValue(raw.tools, undefined), addDirs: (overrides.addDirs ?? stringArray(raw.addDirs, "addDirs")).map((dir) => resolveFrom(baseDir, dir)), @@ -231,6 +238,7 @@ function normalizeCase(baseDir: string, raw: RawCase, index: number): EvalCaseCo addDirs: stringArray(raw.addDirs, `${name}.addDirs`).map((dir) => resolveFrom(baseDir, dir)), mode: modeValue(raw.mode, undefined), tools: stringValue(raw.tools, undefined), + maxTurns: optionalNumber(raw.maxTurns, `${name}.maxTurns`), }; } @@ -250,10 +258,16 @@ function validateCompareConfig(config: CompareConfig): void { if (config.maxBudgetUsd <= 0) { throw new Error("maxBudgetUsd must be positive"); } + if (config.maxTurns !== undefined && (!Number.isInteger(config.maxTurns) || config.maxTurns < 1)) { + throw new Error("maxTurns must be an integer of at least 1"); + } for (const evalCase of config.cases) { if (evalCase.runs !== undefined && evalCase.runs < 1) { throw new Error(`${evalCase.name}.runs must be at least 1`); } + if (evalCase.maxTurns !== undefined && (!Number.isInteger(evalCase.maxTurns) || evalCase.maxTurns < 1)) { + throw new Error(`${evalCase.name}.maxTurns must be an integer of at least 1`); + } for (const image of evalCase.images) { // A missing image must fail at load time, not after the other arm's paid runs. if (!existsSync(image)) { diff --git a/src/runner/claude-p.ts b/src/runner/claude-p.ts index e3bedd4..a8f5ae6 100644 --- a/src/runner/claude-p.ts +++ b/src/runner/claude-p.ts @@ -40,6 +40,7 @@ export function buildClaudeArgs(options: RunnerRunOptions & { systemPromptFile: ...toolArgs, "--max-budget-usd", String(options.maxBudgetUsd), + ...(options.maxTurns === undefined ? [] : ["--max-turns", String(options.maxTurns)]), "--no-session-persistence", // Headless denies file edits without an explicit permission mode, which breaks // artifact-mode agents that must write outputs (e.g. findings.json) into the @@ -93,6 +94,13 @@ export class ClaudePrintRunner implements Runner { throw new Error(`claude timed out after ${options.timeoutMs}ms`); } if (code !== 0) { + // A turn-cap stop is a measured outcome, not a crash: claude may exit + // non-zero with the full result JSON on stdout. Return it so the engine + // can score the run as a failure instead of aborting the comparison. + const capped = tryParseClaudeJson(stdout); + if (capped?.subtype === "error_max_turns") { + return normalizeClaudeResult(capped); + } throw new Error(describeClaudeFailure(code, stdout, stderr, options.maxBudgetUsd)); } @@ -122,12 +130,7 @@ export function describeClaudeFailure( stderr: string, maxBudgetUsd: number, ): string { - let parsed: ClaudeJsonResult | undefined; - try { - parsed = JSON.parse(stdout) as ClaudeJsonResult; - } catch { - // Not JSON — fall through to the raw-stream message. - } + const parsed = tryParseClaudeJson(stdout); if (parsed && typeof parsed === "object") { if (parsed.subtype === "error_max_budget_usd") { @@ -143,6 +146,14 @@ export function describeClaudeFailure( return `claude exited ${code}: ${detail.slice(0, 1_500)}`; } +function tryParseClaudeJson(stdout: string): ClaudeJsonResult | undefined { + try { + return JSON.parse(stdout) as ClaudeJsonResult; + } catch { + return undefined; + } +} + function normalizeClaudeResult(result: ClaudeJsonResult): RunResult { const modelUsage = result.modelUsage; return { @@ -151,6 +162,7 @@ function normalizeClaudeResult(result: ClaudeJsonResult): RunResult { turns: typeof result.num_turns === "number" ? result.num_turns : 0, durationMs: typeof result.duration_ms === "number" ? result.duration_ms : 0, models: modelUsage && typeof modelUsage === "object" ? Object.keys(modelUsage) : [], + ...(result.subtype === "error_max_turns" ? { exhaustedTurns: true } : {}), raw: result, }; } diff --git a/src/types.ts b/src/types.ts index 0feec1f..298e880 100644 --- a/src/types.ts +++ b/src/types.ts @@ -8,6 +8,12 @@ export interface RunResult { turns: number; durationMs: number; models: string[]; + /** + * True when the run stopped at the maxTurns cap instead of finishing. The + * engine fails such runs without grading — partial output that happens to + * satisfy a grader must not count as a pass. + */ + exhaustedTurns?: boolean; raw: unknown; } @@ -56,6 +62,8 @@ export interface RunnerRunOptions { tools: string; addDirs: string[]; maxBudgetUsd: number; + /** Turn cap for agentic runners; completion runners are single-turn and ignore it. */ + maxTurns?: number; } export interface Runner { diff --git a/test/compare.test.ts b/test/compare.test.ts index 273a1c9..1696192 100644 --- a/test/compare.test.ts +++ b/test/compare.test.ts @@ -206,3 +206,114 @@ test("formatCompareSummary labels arms when models differ and flags mixed-runner }; expect(formatCompareSummary(identical)).toContain(" baseline: 3/5 pass"); }); + +import { existsSync, readFileSync } from "node:fs"; +import { runMeasure } from "../src/engine/compare"; + +function fixtureDir(): { dir: string; agent: string; baseline: string; proposed: string } { + const dir = mkdtempSync(join(tmpdir(), "promptdiff-compare-test-")); + const agent = join(dir, "agent.md"); + const baseline = join(dir, "baseline.md"); + const proposed = join(dir, "proposed.md"); + writeFileSync(agent, "Agent", "utf8"); + writeFileSync(baseline, "BASELINE", "utf8"); + writeFileSync(proposed, "PROPOSED", "utf8"); + return { dir, agent, baseline, proposed }; +} + +function cappedConfig(fixture: ReturnType, maxTurns?: number): CompareConfig { + return { + name: "capped compare", + agent: fixture.agent, + baselineSkills: [fixture.baseline], + proposedSkills: [fixture.proposed], + delivery: "inline", + arms: { + baseline: { model: "sonnet", runner: "claude-p" }, + proposed: { model: "sonnet", runner: "claude-p" }, + }, + runs: 1, + timeoutMs: 1_000, + maxBudgetUsd: 1, + maxTurns, + addDirs: [], + sandboxRoot: join(fixture.dir, "runs"), + keepSandbox: false, + cases: [ + { + name: "Case One", + kind: "compare", + prompt: "task", + grader: { type: "text", contains: ["ok"] }, + images: [], + addDirs: [], + }, + ], + }; +} + +test("a run that exhausted its turn cap fails without grading its partial output", async () => { + const fixture = fixtureDir(); + try { + // Output would satisfy the grader — the cap must fail the run anyway. + const runner: Runner = { + name: "mock", + capabilities: { sandboxTools: true, skillRegistry: true, images: false }, + async run(options: RunnerRunOptions) { + expect(options.maxTurns).toBe(7); + const capped = options.systemPrompt.includes("BASELINE"); + return { + output: "ok", + costUsd: 0.1, + turns: capped ? 7 : 3, + durationMs: 10, + models: ["sonnet"], + exhaustedTurns: capped ? true : undefined, + raw: {}, + }; + }, + }; + + const summary = await runCompare({ config: cappedConfig(fixture, 7), runners: { baseline: runner, proposed: runner } }); + expect(summary.cases[0]?.baseline.passes).toBe(0); + expect(summary.cases[0]?.baseline.runs[0]?.grade.message).toBe("hit the 7-turn cap before finishing"); + expect(summary.cases[0]?.proposed.passes).toBe(1); + } finally { + rmSync(fixture.dir, { recursive: true, force: true }); + } +}); + +test("rawOut persists each run's full runner result for both compare arms and for measure", async () => { + const fixture = fixtureDir(); + try { + const usage = { input_tokens: 100, cache_read_input_tokens: 900, output_tokens: 50 }; + const runner: Runner = { + name: "mock", + capabilities: { sandboxTools: true, skillRegistry: true, images: false }, + async run(options: RunnerRunOptions) { + const arm = options.systemPrompt.includes("BASELINE") ? "baseline" : "proposed"; + return { output: "ok", costUsd: 0.1, turns: 1, durationMs: 10, models: ["sonnet"], raw: { arm, usage } }; + }, + }; + + const rawDir = join(fixture.dir, "raw"); + const config = cappedConfig(fixture); + await runCompare({ config, runners: { baseline: runner, proposed: runner }, rawOut: { dir: rawDir } }); + + // Case names are sanitized the same way receipts sanitize scenario names. + const baselineRaw = JSON.parse(readFileSync(join(rawDir, "case-one_baseline_1.json"), "utf8")); + expect(baselineRaw.arm).toBe("baseline"); + expect(baselineRaw.usage).toEqual(usage); + expect(JSON.parse(readFileSync(join(rawDir, "case-one_proposed_1.json"), "utf8")).arm).toBe("proposed"); + + const measureDir = join(fixture.dir, "raw-measure"); + await runMeasure({ config, runner, rawOut: { dir: measureDir } }); + expect(existsSync(join(measureDir, "case-one_measure_1.json"))).toBe(true); + + // Without rawOut nothing extra is written. + const plain = await runCompare({ config, runners: { baseline: runner, proposed: runner } }); + expect(plain.cases[0]?.baseline.passes).toBe(1); + } finally { + rmSync(fixture.dir, { recursive: true, force: true }); + } +}); diff --git a/test/config.test.ts b/test/config.test.ts index a9fbfa6..8e8ff5c 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -213,3 +213,39 @@ test("pricing parses per-model rates and rejects gaps for openai arms", () => { rmSync(dir, { recursive: true, force: true }); } }); + +test("loadCompareConfig parses maxTurns at both levels and rejects bad caps", () => { + const dir = mkdtempSync(join(tmpdir(), "promptdiff-config-test-")); + try { + writeFileSync(join(dir, "agent.md"), "Agent", "utf8"); + writeFileSync(join(dir, "baseline.md"), "Baseline", "utf8"); + writeFileSync(join(dir, "proposed.md"), "Proposed", "utf8"); + const scenario = (extra: object, cases: object = {}) => + JSON.stringify({ + agent: "./agent.md", + baselineSkills: ["./baseline.md"], + proposedSkills: ["./proposed.md"], + model: "sonnet", + scenarios: [{ name: "t", prompt: "p", grader: { type: "text", contains: ["ok"] }, ...cases }], + ...extra, + }); + + writeFileSync(join(dir, "capped.json"), scenario({ maxTurns: 25 }, { maxTurns: 10 }), "utf8"); + const config = loadCompareConfig(join(dir, "capped.json")); + expect(config.maxTurns).toBe(25); + expect(config.cases[0]?.maxTurns).toBe(10); + // CLI override wins over the scenario file. + expect(loadCompareConfig(join(dir, "capped.json"), { maxTurns: 5 }).maxTurns).toBe(5); + + writeFileSync(join(dir, "uncapped.json"), scenario({}), "utf8"); + expect(loadCompareConfig(join(dir, "uncapped.json")).maxTurns).toBeUndefined(); + + writeFileSync(join(dir, "zero.json"), scenario({ maxTurns: 0 }), "utf8"); + expect(() => loadCompareConfig(join(dir, "zero.json"))).toThrow("maxTurns must be an integer of at least 1"); + + writeFileSync(join(dir, "fractional-case.json"), scenario({}, { maxTurns: 2.5 }), "utf8"); + expect(() => loadCompareConfig(join(dir, "fractional-case.json"))).toThrow("t.maxTurns must be an integer of at least 1"); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); diff --git a/test/runner.test.ts b/test/runner.test.ts index 7de2c29..261085c 100644 --- a/test/runner.test.ts +++ b/test/runner.test.ts @@ -1,5 +1,8 @@ +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; import { expect, test } from "bun:test"; -import { buildClaudeArgs, describeClaudeFailure } from "../src/runner/claude-p"; +import { buildClaudeArgs, ClaudePrintRunner, describeClaudeFailure } from "../src/runner/claude-p"; test("buildClaudeArgs uses a system prompt file, budget, tools, and add-dir", () => { const args = buildClaudeArgs({ @@ -147,3 +150,89 @@ test("describeClaudeFailure surfaces structured errors and falls back to raw str expect(describeClaudeFailure(2, "not json", "boom", 1)).toBe("claude exited 2: boom"); expect(describeClaudeFailure(2, "partial output", "", 1)).toBe("claude exited 2: partial output"); }); + +test("buildClaudeArgs passes --max-turns only when a cap is set", () => { + const base = { + systemPrompt: "system", + systemPromptFile: "/tmp/system.md", + userPrompt: "do work", + model: "sonnet", + cwd: "/tmp/sandbox", + addDirs: [], + tools: "", + timeoutMs: 1_000, + maxBudgetUsd: 0.25, + }; + + const capped = buildClaudeArgs({ ...base, maxTurns: 25 }); + const i = capped.indexOf("--max-turns"); + expect(i).toBeGreaterThan(-1); + expect(capped[i + 1]).toBe("25"); + + expect(buildClaudeArgs(base)).not.toContain("--max-turns"); +}); + +// Captured shape of a turn-cap stop: full result JSON on stdout with +// subtype "error_max_turns" (exit code varies by claude version). +const MAX_TURNS_STDOUT = JSON.stringify({ + type: "result", + subtype: "error_max_turns", + is_error: true, + result: "partial answer", + num_turns: 25, + total_cost_usd: 0.42, + duration_ms: 60_000, + modelUsage: { "claude-sonnet-5": { costUSD: 0.42 } }, +}); + +test("a turn-capped run is returned as a measured outcome, not thrown as a crash", async () => { + const dir = mkdtempSync(join(tmpdir(), "promptdiff-runner-test-")); + try { + const bin = join(dir, "fake-claude"); + writeFileSync(bin, `#!/bin/sh\necho '${MAX_TURNS_STDOUT}'\nexit 1\n`, { mode: 0o755 }); + + const runner = new ClaudePrintRunner(bin); + const result = await runner.run({ + systemPrompt: "system", + userPrompt: "do work", + model: "sonnet", + cwd: dir, + addDirs: [], + tools: "", + timeoutMs: 10_000, + maxBudgetUsd: 1, + maxTurns: 25, + }); + + expect(result.exhaustedTurns).toBe(true); + expect(result.output).toBe("partial answer"); + expect(result.turns).toBe(25); + expect(result.costUsd).toBeCloseTo(0.42); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); + +test("non-zero exits without a turn-cap subtype still throw", async () => { + const dir = mkdtempSync(join(tmpdir(), "promptdiff-runner-test-")); + try { + const bin = join(dir, "fake-claude"); + writeFileSync(bin, `#!/bin/sh\necho boom >&2\nexit 2\n`, { mode: 0o755 }); + + const runner = new ClaudePrintRunner(bin); + await expect( + runner.run({ + systemPrompt: "system", + userPrompt: "do work", + model: "sonnet", + cwd: dir, + addDirs: [], + tools: "", + timeoutMs: 10_000, + maxBudgetUsd: 1, + }), + ).rejects.toThrow("claude exited 2: boom"); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +});