diff --git a/docs/project-configuration.md b/docs/project-configuration.md index 1d131fec6..4371b64db 100644 --- a/docs/project-configuration.md +++ b/docs/project-configuration.md @@ -348,3 +348,11 @@ and applies its severity policy to recovered findings. Workflow identity records explicitly requested deep settings, not ambient values or shipped defaults, so changing those defaults does not prevent resumption. Changing the explicit request still requires a different workflow ID. + +### Native executable selection + +Native plugin sessions use `CODEX_CLI_PATH` when supplied, or a real `codex` +executable on `PATH` (`codex.exe` on Windows). The selected executable must be +outside the scan target. Windows npm shims, managed-package directories and +desktop cache directories are no longer searched; set `CODEX_CLI_PATH` to the +installed executable when it is not directly on `PATH`. diff --git a/plugins/codex-security/mcp-app/package.json b/plugins/codex-security/mcp-app/package.json index 1de6e78bb..46703d6c7 100644 --- a/plugins/codex-security/mcp-app/package.json +++ b/plugins/codex-security/mcp-app/package.json @@ -11,14 +11,13 @@ }, "dependencies": { "@modelcontextprotocol/sdk": "^1.30.0", - "@openai/codex-sdk": "0.158.0", - "smol-toml": "1.8.0", "zod": "^4.6.5" }, "devDependencies": { "@types/node": "^26.6.2", "esbuild": "^0.28.2", "prettier": "3.9.8", + "smol-toml": "1.8.0", "typescript": "^7.0.2" } } diff --git a/plugins/codex-security/mcp-app/pnpm-lock.yaml b/plugins/codex-security/mcp-app/pnpm-lock.yaml index 2f3e90c19..a74909ce9 100644 --- a/plugins/codex-security/mcp-app/pnpm-lock.yaml +++ b/plugins/codex-security/mcp-app/pnpm-lock.yaml @@ -11,12 +11,6 @@ importers: '@modelcontextprotocol/sdk': specifier: ^1.30.0 version: 1.30.0(zod@4.6.5) - '@openai/codex-sdk': - specifier: 0.158.0 - version: 0.158.0 - smol-toml: - specifier: 1.8.0 - version: 1.8.0 zod: specifier: ^4.6.5 version: 4.6.5 @@ -30,6 +24,9 @@ importers: prettier: specifier: 3.9.8 version: 3.9.8 + smol-toml: + specifier: 1.8.0 + version: 1.8.0 typescript: specifier: ^7.0.2 version: 7.0.2 @@ -208,51 +205,6 @@ packages: '@cfworker/json-schema': optional: true - '@openai/codex-sdk@0.158.0': - resolution: {integrity: sha512-lZIsVxPjTkgaz5nBlbUy5qd/cB7kjqXjtKMqvQJ+gaWkXevp9bWcyk3Meyix07GFE8+P8cYH3JxdgBikzaPvdg==} - engines: {node: '>=18'} - - '@openai/codex@0.158.0': - resolution: {integrity: sha512-GBhcKpQmVLsCtEP5mUf6WFye6QQgTKotwsXrYPM0GFmEsoOSahik6hkf2FHabX9kZBl0Y0/PQowLu7uYMKT8dg==} - engines: {node: '>=16'} - hasBin: true - - '@openai/codex@0.158.0-darwin-arm64': - resolution: {integrity: sha512-0OKSjlWY1j4Ld1fT87QttNw3Y2SthcXi4GcrWSHjleZg1n86eG3+shJl4Pv+siUmsBJhWlNmm6rRO4Yv8ZyQLg==} - engines: {node: '>=16'} - cpu: [arm64] - os: [darwin] - - '@openai/codex@0.158.0-darwin-x64': - resolution: {integrity: sha512-FrX1o3APrL7F6QkO8z08Rq8lJitH2sNI7pkebA02eYA103hDs3fyy9d32zeLLQJUeAhmZA4pV9xJR4m7cVyNpQ==} - engines: {node: '>=16'} - cpu: [x64] - os: [darwin] - - '@openai/codex@0.158.0-linux-arm64': - resolution: {integrity: sha512-T9AgGcoU7HNsxJ4prT/R9YvztyEmz9kWP/VlT2cbCop4PVwiqfG9YMJtLo4j2ScewvSaTl4sQFwla+fwGpoRZg==} - engines: {node: '>=16'} - cpu: [arm64] - os: [linux] - - '@openai/codex@0.158.0-linux-x64': - resolution: {integrity: sha512-mY12GZPM8TuOWVGxCyNV2NSA8t1uIupm9Nhu9VeQVEvvxBKN08U8bLH+vn7oISUHZPC8bwhROIbo/ZUxqV6XMg==} - engines: {node: '>=16'} - cpu: [x64] - os: [linux] - - '@openai/codex@0.158.0-win32-arm64': - resolution: {integrity: sha512-Jw1u0q0+5PG97jPkINxE3UCFtsYBN8Af+IjjM0zlCO675Sv5lNsXU2K9aIaXwVeQIKw8q2lbPAR4SEkq2rSOoA==} - engines: {node: '>=16'} - cpu: [arm64] - os: [win32] - - '@openai/codex@0.158.0-win32-x64': - resolution: {integrity: sha512-IaUmY11Zdqa/Zok6kE0Z5375pXtClKRbS8P1Gh2C73Aq65CaGn9N8gT6BXGC3+XD6yamAIiEQW5xM842UCCGow==} - engines: {node: '>=16'} - cpu: [x64] - os: [win32] - '@types/node@26.6.2': resolution: {integrity: sha512-X1P21scMv4zGKLYqjdGjaKa7COa0RKVYYZZN/NfvLQ1JegxFhdhpZG/Lyn8AXx6CDUavKAd11v6BvfpkDByK8g==} @@ -858,37 +810,6 @@ snapshots: transitivePeerDependencies: - supports-color - '@openai/codex-sdk@0.158.0': - dependencies: - '@openai/codex': 0.158.0 - - '@openai/codex@0.158.0': - optionalDependencies: - '@openai/codex-darwin-arm64': '@openai/codex@0.158.0-darwin-arm64' - '@openai/codex-darwin-x64': '@openai/codex@0.158.0-darwin-x64' - '@openai/codex-linux-arm64': '@openai/codex@0.158.0-linux-arm64' - '@openai/codex-linux-x64': '@openai/codex@0.158.0-linux-x64' - '@openai/codex-win32-arm64': '@openai/codex@0.158.0-win32-arm64' - '@openai/codex-win32-x64': '@openai/codex@0.158.0-win32-x64' - - '@openai/codex@0.158.0-darwin-arm64': - optional: true - - '@openai/codex@0.158.0-darwin-x64': - optional: true - - '@openai/codex@0.158.0-linux-arm64': - optional: true - - '@openai/codex@0.158.0-linux-x64': - optional: true - - '@openai/codex@0.158.0-win32-arm64': - optional: true - - '@openai/codex@0.158.0-win32-x64': - optional: true - '@types/node@26.6.2': dependencies: undici-types: 8.9.0 diff --git a/plugins/codex-security/mcp-app/scripts/build_mcp_app.mjs b/plugins/codex-security/mcp-app/scripts/build_mcp_app.mjs index d0e08b7da..6c61c2470 100644 --- a/plugins/codex-security/mcp-app/scripts/build_mcp_app.mjs +++ b/plugins/codex-security/mcp-app/scripts/build_mcp_app.mjs @@ -6,6 +6,7 @@ import { pathToFileURL } from "node:url"; import { brotliCompressSync, constants as zlibConstants } from "node:zlib"; import { execFileSync } from "node:child_process"; import { build } from "esbuild"; +import { mcpBundleOptions } from "./bundle_options.mjs"; const root = resolve(import.meta.dirname, ".."); const sdkRequire = createRequire( @@ -70,24 +71,14 @@ export async function buildMcpApp({ output, native = "universal" }) { const bundle = join(mcpDir, name + ".bundle.cjs"); try { await build({ - bundle: true, - banner: { - js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", - }, - define: { "import.meta.url": "__codexSecurityModuleUrl" }, + ...mcpBundleOptions, entryPoints: [join(root, entryPoint)], inject: name === "server" ? [sdkRequire.resolve("pdfjs-dist/legacy/build/pdf.worker.mjs")] : [], - external: ["fsevents"], - format: "cjs", - loader: { ".md": "text" }, logLevel: "info", - logOverride: { "empty-import-meta": "silent" }, outfile: bundle, - platform: "node", - target: "node20", }); const runtime = brotliCompressSync(await readFile(bundle), { params: { [zlibConstants.BROTLI_PARAM_QUALITY]: 10 }, diff --git a/plugins/codex-security/mcp-app/scripts/bundle_options.mjs b/plugins/codex-security/mcp-app/scripts/bundle_options.mjs new file mode 100644 index 000000000..eeb281226 --- /dev/null +++ b/plugins/codex-security/mcp-app/scripts/bundle_options.mjs @@ -0,0 +1,20 @@ +import { createRequire } from "node:module"; +import { dirname } from "node:path"; + +// Source tests and shipped runtimes use the same CommonJS and dependency resolution. +export const mcpBundleOptions = { + bundle: true, + alias: { + zod: dirname(createRequire(import.meta.url).resolve("zod/package.json")), + }, + banner: { + js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", + }, + define: { "import.meta.url": "__codexSecurityModuleUrl" }, + external: ["fsevents"], + format: "cjs", + loader: { ".md": "text" }, + logOverride: { "empty-import-meta": "silent" }, + platform: "node", + target: "node20", +}; diff --git a/plugins/codex-security/mcp-app/server.ts b/plugins/codex-security/mcp-app/server.ts index ac2ee942c..46dc6164c 100644 --- a/plugins/codex-security/mcp-app/server.ts +++ b/plugins/codex-security/mcp-app/server.ts @@ -9,6 +9,7 @@ import * as z from "zod/v4"; import { missingPythonHelperMessage, resolvePythonCommand, + workbenchCommandTimeout, } from "./src/python_command.js"; import type { ScanResults } from "./src/types.js"; import { MCP_APP_VERSION } from "./src/version.js"; @@ -1194,7 +1195,7 @@ export function createCodexSecurityServer(): McpServer { }, abortSignalFromExtra(extra), ); - return nativeScanCompletedResult(scan); + return nativeScanCompletedResult(scan, "get-scan"); } catch (error: unknown) { if (error instanceof ScanPermissionError) return toolErrorResult(deepScanInvocationFailureMessage(error)); @@ -2379,14 +2380,19 @@ function boundedErrorData(error: unknown): { message: string; name: string } { }; } -async function nativeScanCompletedResult(scan: ScanResults) { +async function nativeScanCompletedResult( + scan: ScanResults, + command: "complete-scan" | "get-scan" = "complete-scan", +) { let completed: JsonObject; try { completed = await runWorkbench([ - "complete-scan", + command, "--scan-id", scan.scanId, - ...optionalArg("--claim-token", scan.handoffClaimToken), + ...(command === "complete-scan" + ? optionalArg("--claim-token", scan.handoffClaimToken) + : []), ]); } catch (error) { return toolErrorResult(completionFailureMessage(error)); @@ -2600,27 +2606,7 @@ async function executeWorkbench( encoding: "utf8" as const, // Artifact bytes are base64-encoded here; retain the existing file-size behavior. maxBuffer: args[0] === "read-artifact" ? Infinity : 4 * 1024 * 1024, - timeout: [ - "begin-deep-scan", - "complete-scan", - "export-findings", - "get-scan", - "get-workspace", - "inspect-setup", - "list-findings", - "preserve-scan-results", - "recover-scan-results", - "request-finding-remediation", - "request-finding-remediation-action", - "save-workspace", - "set-finding-triage", - "set-finding-remediation", - "start-headless-standard-scan", - "start-prompt-only-scan", - "start-scan", - ].includes(args[0] ?? "") - ? 300_000 - : 30_000, + timeout: workbenchCommandTimeout(args[0]), }, ); if (workbenchInput !== undefined) { diff --git a/plugins/codex-security/mcp-app/src/artifact-attack-path.ts b/plugins/codex-security/mcp-app/src/artifact-attack-path.ts index 1666a4be2..8236d191c 100644 --- a/plugins/codex-security/mcp-app/src/artifact-attack-path.ts +++ b/plugins/codex-security/mcp-app/src/artifact-attack-path.ts @@ -86,7 +86,7 @@ export async function recordCodexSecurityCandidateAttackPaths( operation: "replace"; rowsWritten: number; }> { - if (context.layout !== "scan") { + if (!context.scanId) { throw new Error( "Candidate attack-path analysis requires a scan-bound artifact context.", ); diff --git a/plugins/codex-security/mcp-app/src/artifact-context.ts b/plugins/codex-security/mcp-app/src/artifact-context.ts index 3d61d8c9e..9ac640f8f 100644 --- a/plugins/codex-security/mcp-app/src/artifact-context.ts +++ b/plugins/codex-security/mcp-app/src/artifact-context.ts @@ -84,7 +84,6 @@ export async function createScanArtifactContext( rawRepoRoot, "Codex Security scan target root", ), - layout: "scan", scanId, ...defined("scope", optionalString(scan.scope)), ...defined("pluginRoot", options.pluginRoot), diff --git a/plugins/codex-security/mcp-app/src/artifact-inventory.ts b/plugins/codex-security/mcp-app/src/artifact-inventory.ts index 007720791..e3574e7ca 100644 --- a/plugins/codex-security/mcp-app/src/artifact-inventory.ts +++ b/plugins/codex-security/mcp-app/src/artifact-inventory.ts @@ -78,7 +78,7 @@ const reviewItemSchema = loadArtifactZodSchema( export async function prepareCodexSecurityReviewItems( context: ArtifactContext, ): Promise { - if (context.layout !== "scan") { + if (!context.scanId) { throw new Error( `${label}: only a parent scan can prepare its shared inventory.`, ); diff --git a/plugins/codex-security/mcp-app/src/artifact-io.ts b/plugins/codex-security/mcp-app/src/artifact-io.ts index 9325c1146..4c5ee4746 100644 --- a/plugins/codex-security/mcp-app/src/artifact-io.ts +++ b/plugins/codex-security/mcp-app/src/artifact-io.ts @@ -8,7 +8,6 @@ import { dirname, isAbsolute, join, resolve, sep } from "node:path"; export interface ArtifactContext { root: string; repoRoot: string; - layout: "scan"; scanId?: string; scope?: string; pluginRoot?: string; diff --git a/plugins/codex-security/mcp-app/src/artifact-scan-draft.ts b/plugins/codex-security/mcp-app/src/artifact-scan-draft.ts index c009d6c7b..8f51b58e7 100644 --- a/plugins/codex-security/mcp-app/src/artifact-scan-draft.ts +++ b/plugins/codex-security/mcp-app/src/artifact-scan-draft.ts @@ -14,10 +14,6 @@ import { type SemanticScan, type SemanticCoverage, } from "../../../../sdk/typescript/src/scan-semantics.js"; -export { - preserveFindingDetails, - scanFindingIdentity, -} from "../../../../sdk/typescript/src/scan-semantics.js"; import { createHash, randomUUID } from "node:crypto"; import { promises as fs } from "node:fs"; import { join, sep } from "node:path"; @@ -85,12 +81,11 @@ export const completedScanInputSchema = loadArtifactZodSchema( export async function recordCodexSecurityScanDraft( context: ArtifactContext, input: ScanDraftInput, - publishDraft?: PublishScanDraft, + publishDraft: PublishScanDraft, signal?: AbortSignal, ): Promise { const parsed = parseScanDraft(input); requireBoundScan(context, parsed, true); - if (!publishDraft) await saveScanDraftCheckpoint(context, parsed); for (;;) { signal?.throwIfAborted(); @@ -104,30 +99,7 @@ export async function recordCodexSecurityScanDraft( const hardening = await readExistingHardeningPortfolio(context); const draft = prepareSemanticScanDraft(context, reconciled, hardening); try { - if (publishDraft) { - await publishDraft(draft, preserved.previousDigest, parsed); - } else { - const destinations = await Promise.all([ - artifactDestination( - context, - ["findings.json"], - "scan draft findings", - ), - artifactDestination( - context, - ["coverage.json"], - "scan draft coverage", - ), - artifactDestination( - context, - ["scan-manifest.json"], - "scan draft manifest", - ), - ]); - await replaceArtifactJson(destinations[0], draft.findings); - await replaceArtifactJson(destinations[1], draft.coverage); - await replaceArtifactJson(destinations[2], draft.manifest); - } + await publishDraft(draft, preserved.previousDigest, parsed); return { scanId: reconciled.scanId, findingCount: draft.findings.findings.length, @@ -206,31 +178,6 @@ export async function recordCodexSecurityScanDraftViaWorkbench( ); } -/** Keep the semantic input before replacing canonical artifacts. */ -export async function saveScanDraftCheckpoint( - context: ArtifactContext, - input: ScanDraftInput, -): Promise { - const { handoffClaimToken: _claim, ...snapshot } = input; - const contents = JSON.stringify(snapshot, null, 2) + "\n"; - const name = scanDraftCheckpointName(input); - const destination = await artifactDestination( - context, - ["checkpoints", name], - "scan checkpoint", - ); - try { - const existing = await fs.readFile(destination, "utf8"); - if (existing !== contents) - throw new Error( - "scan checkpoint: existing content does not match its digest.", - ); - } catch (error) { - if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error; - await replaceArtifactJson(destination, snapshot); - } -} - async function preserveScanDraft( context: ArtifactContext, input: ScanDraftInput, @@ -480,7 +427,7 @@ async function readCurrentCheckpoints( "scan checkpoint: current checkpoint is not a safe file.", ); } - const input = parsePersistedCheckpoint( + const input = parsePersistedScanDraft( parseJsonObject( await readArtifactText( context, @@ -838,254 +785,28 @@ export function parseScanDraft(input: unknown): ScanDraftInput { return parsed; } -/** Re-admit results persisted by older plugin versions without loosening live tool input. */ -export function parsePersistedScanDraft( +/** Project current canonical metadata without coercing persisted finding details. */ +function parsePersistedScanDraft( input: Record, ): ScanDraftInput { - const compatible = structuredClone(input); - if (!Array.isArray(compatible.findings)) { - return parseScanDraft(compatible); - } - for (const finding of compatible.findings) { - if (!isObject(finding)) continue; - normalizePersistedFindingDetails(finding); - } - return parseScanDraft(compatible); -} - -function parsePersistedCheckpoint( - input: Record, -): ScanDraftInput { - const compatible = structuredClone(input); - if (isObject(compatible.scope)) { - delete compatible.scope.includePaths; - delete compatible.scope.excludePaths; - if (Object.keys(compatible.scope).length === 0) delete compatible.scope; - } - if (isObject(compatible.coverage)) { - for (const field of [ - "documentType", - "schemaVersion", - "scanId", - "mode", - "includePaths", - "excludePaths", - "receiptRefs", - "inventoryStrategy", - ]) - delete compatible.coverage[field]; - } - if (Array.isArray(compatible.findings)) { - for (const finding of compatible.findings) { - if (!isObject(finding)) continue; - delete finding.findingId; - delete finding.occurrenceId; - delete finding.fingerprints; - } - } - return parsePersistedScanDraft(compatible); -} - -function normalizePersistedFindingDetails(finding: JsonObject): void { - const canonicalEvidence = Array.isArray(finding.codeEvidence) - ? finding.codeEvidence - : []; - const evidenceIds = new Set( - canonicalEvidence.flatMap((evidence) => { - if (!isObject(evidence)) return []; - const id = evidence.id; - return typeof id === "string" && id.trim().length > 0 ? [id] : []; - }), - ); - if (Array.isArray(finding.code_evidence)) { - const compatibleEvidence: JsonObject[] = []; - for (const evidence of finding.code_evidence) { - if (!isObject(evidence)) continue; - const id = evidence.id; - const code = evidence.code; - if ( - typeof id !== "string" || - id.trim().length === 0 || - typeof code !== "string" || - code.trim().length === 0 || - evidenceIds.has(id) - ) { - continue; - } - evidenceIds.add(id); - compatibleEvidence.push(evidence); - } - finding.code_evidence = compatibleEvidence; - } else if ("code_evidence" in finding) { - delete finding.code_evidence; - } - - for (const [sectionName, listFields] of [ - ["rootCause", ["evidenceRefs", "evidence_refs"]], - ["root_cause", ["evidenceRefs", "evidence_refs"]], - [ - "validation", - [ - "assertions", - "counterEvidence", - "evidence", - "evidenceRefs", - "evidence_refs", - "limitations", - ], - ], - [ - "attackPath", - [ - "assumptions", - "blindspots", - "controls", - "evidenceRefs", - "evidence_refs", - "limitations", - "preconditions", - "steps", - ], - ], - ] satisfies Array<[string, string[]]>) { - const section = finding[sectionName]; - if (!isObject(section)) continue; - normalizePersistedStringLists(section, listFields); - filterPersistedEvidenceRefs(section, evidenceIds); - } - - const rootCause = finding.rootCause; - if (isObject(rootCause)) { - if ( - typeof rootCause.summary !== "string" || - rootCause.summary.trim().length === 0 - ) { - delete finding.rootCause; - } else { - removeUnsupportedPersistedStrings(rootCause, ["code", "language"]); - } - } - const legacyRootCause = finding.root_cause; - if (isObject(legacyRootCause)) { - removeUnsupportedPersistedStrings(legacyRootCause, [ - "summary", - "code", - "language", - ]); - } else if ( - "root_cause" in finding && - (typeof legacyRootCause !== "string" || legacyRootCause.trim().length === 0) - ) { - delete finding.root_cause; - } - - const validation = finding.validation; - if (isObject(validation)) { - removeUnsupportedPersistedStrings(validation, [ - "method", - "status", - "summary", - "disposition", - "result", - ]); - } - - const attackPath = finding.attackPath; - if (!isObject(attackPath)) return; - removeUnsupportedPersistedStrings(attackPath, ["summary"]); - for (const field of ["dataFlow", "data_flow", "dataflow", "reachability"]) { - const detail = attackPath[field]; - if (detail === null) { - delete attackPath[field]; - continue; - } - if (typeof detail === "string") { - if (detail.trim().length === 0) delete attackPath[field]; - continue; - } - if (!isObject(detail)) { - if (field in attackPath) delete attackPath[field]; - continue; - } - removeUnsupportedPersistedStrings(detail, [ - "summary", - "source", - "sink", - "outcome", - ...(field === "reachability" ? ["attacker", "entrypoint"] : []), - ]); - normalizePersistedStringLists(detail, [ - "evidenceRefs", - "evidence_refs", - "transformations", - ...(field === "reachability" ? ["preconditions"] : []), - ]); - filterPersistedEvidenceRefs(detail, evidenceIds); - } - for (const field of ["impact", "likelihood"]) { - const detail = attackPath[field]; - if (isObject(detail)) { - removeUnsupportedPersistedStrings(detail, ["level", "rationale", "why"]); - } else if ( - detail !== undefined && - detail !== null && - (typeof detail !== "string" || detail.trim().length === 0) - ) { - delete attackPath[field]; - } - } -} - -function normalizePersistedStringLists( - section: JsonObject, - fields: string[], -): void { - for (const field of fields) { - if (!(field in section)) continue; - const value = section[field]; - const normalized = - typeof value === "string" - ? value.trim().length > 0 - ? [value] - : [] - : Array.isArray(value) - ? value.filter( - (item): item is string => - typeof item === "string" && item.trim().length > 0, - ) - : []; - if (normalized.length > 0) section[field] = normalized; - else delete section[field]; - } -} - -function filterPersistedEvidenceRefs( - section: JsonObject, - evidenceIds: Set, -): void { - for (const field of ["evidenceRefs", "evidence_refs"]) { - const refs = section[field]; - if (!Array.isArray(refs)) continue; - section[field] = refs.filter( - (ref): ref is string => - typeof ref === "string" && - ref.trim().length > 0 && - evidenceIds.has(ref), + try { + const projected = semanticScanDraft( + input.scanId as string, + input, + input.findings as JsonObject[], + requireObject(input.coverage, "saved scan draft coverage"), + ); + return parseScanDraft({ + ...input, + ...(isObject(input.scope) ? { scope: projected.scope } : {}), + findings: projected.findings, + coverage: projected.coverage, + }); + } catch (cause) { + throw new Error( + "Saved scan draft does not match the current schema. Start a new scan; the saved artifacts remain available.", + { cause }, ); - } -} - -function removeUnsupportedPersistedStrings( - section: JsonObject, - fields: string[], -): void { - for (const field of fields) { - if ( - field in section && - (typeof section[field] !== "string" || section[field].trim().length === 0) - ) { - delete section[field]; - } } } @@ -1094,11 +815,6 @@ function requireBoundScan( input: CompletedScanInput, requireRunning: boolean, ): void { - if (context.layout !== "scan") { - throw new Error( - "scan draft: this operation requires an authoritative parent scan context.", - ); - } requireMatchingScan(context, input); if (requireRunning && context.status !== "running") { throw new Error( diff --git a/plugins/codex-security/mcp-app/src/artifact-storage.ts b/plugins/codex-security/mcp-app/src/artifact-storage.ts index 96f1b8748..757f33791 100644 --- a/plugins/codex-security/mcp-app/src/artifact-storage.ts +++ b/plugins/codex-security/mcp-app/src/artifact-storage.ts @@ -65,7 +65,7 @@ export async function standaloneArtifactContext( if (storage === "temporary") { // Resolve existing ancestors for stable imports without creating or requiring // the persistent collection. storageContext prepares the temporary root. - return { root: await resolveStoragePath(root), repoRoot, layout: "scan" }; + return { root: await resolveStoragePath(root), repoRoot }; } const existingRoot = await fs.realpath(scanRoot).catch(() => scanRoot); if (existingRoot === repoRoot || existingRoot.startsWith(repoRoot + sep)) { @@ -75,7 +75,6 @@ export async function standaloneArtifactContext( return { root: await requireArtifactRoot(root, "Standalone artifacts"), repoRoot, - layout: "scan", }; } diff --git a/plugins/codex-security/mcp-app/src/artifact-validation-phase.ts b/plugins/codex-security/mcp-app/src/artifact-validation-phase.ts index 0d55a8f8e..f5d4346a8 100644 --- a/plugins/codex-security/mcp-app/src/artifact-validation-phase.ts +++ b/plugins/codex-security/mcp-app/src/artifact-validation-phase.ts @@ -78,7 +78,7 @@ export async function recordCodexSecurityCandidateValidations( operation: "replace"; rowsWritten: number; }> { - if (context.layout !== "scan") { + if (!context.scanId) { throw new Error( "Candidate validation requires a scan-bound artifact context.", ); diff --git a/plugins/codex-security/mcp-app/src/native-executable.ts b/plugins/codex-security/mcp-app/src/native-executable.ts index 65146265f..63a91752c 100644 --- a/plugins/codex-security/mcp-app/src/native-executable.ts +++ b/plugins/codex-security/mcp-app/src/native-executable.ts @@ -1,20 +1,10 @@ import { accessSync, constants as fsConstants, - existsSync, promises as fs, - readdirSync, statSync, } from "node:fs"; -import { createRequire } from "node:module"; -import { - delimiter, - dirname, - isAbsolute, - join, - resolve, - win32, -} from "node:path"; +import { delimiter, isAbsolute, join, resolve, win32 } from "node:path"; import { resolveTrustedExecutable, type TrustedExecutable, @@ -24,15 +14,14 @@ export async function resolveTrustedCodex( environment: NodeJS.ProcessEnv, protectedRoot: string, platform: NodeJS.Platform = process.platform, - architecture: NodeJS.Architecture = process.arch, originalCwd: string = process.cwd(), ): Promise { for (const candidate of codexPathCandidates( environment, platform, - architecture, originalCwd, )) { + if (platform === "win32" && isWindowsAppsPath(candidate)) continue; const codex = await resolveTrustedExecutable( candidate, environment, @@ -67,88 +56,41 @@ export async function snapshotNativeEnvironment(): Promise< export function resolveCodexPath( env: NodeJS.ProcessEnv = process.env, platform: NodeJS.Platform = process.platform, - architecture: NodeJS.Architecture = process.arch, originalCwd: string = process.cwd(), ): string { - return codexPathCandidates(env, platform, architecture, originalCwd).next() - .value!; + return codexPathCandidates(env, platform, originalCwd).next().value!; } function* codexPathCandidates( env: NodeJS.ProcessEnv, platform: NodeJS.Platform, - architecture: NodeJS.Architecture, originalCwd: string, ): Generator { - const searchPath = searchPathForPlatform(env, platform); const configured = environmentVariable( env, "CODEX_CLI_PATH", platform, )?.trim(); - if (configured && (platform !== "win32" || !isWindowsAppsPath(configured))) { - if (isBareCommandName(configured)) { - const executableName = - platform === "win32" && !configured.toLowerCase().endsWith(".exe") - ? `${configured}.exe` - : configured; - const fromSearchPath = - platform === "win32" - ? configured === "codex" || configured === "codex.exe" - ? resolveWindowsCodexFromSearchPath( - searchPath, - architecture, - originalCwd, - ) - : resolveWindowsDirectFromSearchPath( - searchPath, - executableName, - originalCwd, - ) - : resolveFromSearchPath(searchPath, executableName, originalCwd); - yield* fromSearchPath; - } - yield absoluteCodexPath(configured, platform, originalCwd); - return; - } - - if (platform !== "win32") { - yield* resolveFromSearchPath(searchPath, "codex", originalCwd); - yield resolve(originalCwd, "codex"); - return; + const command = configured || "codex"; + if (isBareCommandName(command)) { + const executableName = + platform === "win32" && !command.toLowerCase().endsWith(".exe") + ? `${command}.exe` + : command; + const searchPath = searchPathForPlatform(env, platform); + yield* platform === "win32" + ? resolveWindowsDirectFromSearchPath( + searchPath, + executableName, + originalCwd, + ) + : resolveFromSearchPath(searchPath, executableName, originalCwd); } - - const managedPackageRoot = environmentVariable( - env, - "CODEX_MANAGED_PACKAGE_ROOT", + yield absoluteCodexPath( + configured || (platform === "win32" ? "codex.exe" : "codex"), platform, - )?.trim(); - if (managedPackageRoot) { - const managedBinary = resolveWindowsPackageBinary( - absoluteCodexPath(managedPackageRoot, platform, originalCwd), - architecture, - ); - if (managedBinary && !isWindowsAppsPath(managedBinary)) yield managedBinary; - } - - yield* resolveWindowsCodexFromSearchPath( - searchPath, - architecture, originalCwd, ); - - const localAppData = environmentVariable( - env, - "LOCALAPPDATA", - platform, - )?.trim(); - const cachedBinary = resolveWindowsCachedBinary( - localAppData - ? absoluteCodexPath(localAppData, platform, originalCwd) - : undefined, - ); - if (cachedBinary) yield cachedBinary; - yield resolve(originalCwd, "codex.exe"); } function searchPathForPlatform( @@ -201,32 +143,8 @@ function* resolveWindowsDirectFromSearchPath( absoluteWindowsSearchDirectory(directory, originalCwd), executableName, ); - if (!isWindowsAppsPath(candidate) && existsSync(candidate)) yield candidate; - } -} - -function* resolveWindowsCodexFromSearchPath( - searchPath: string | undefined, - architecture: NodeJS.Architecture, - originalCwd: string, -): Generator { - for (const directory of searchPath?.split(delimiter) ?? []) { - const absoluteDirectory = absoluteWindowsSearchDirectory( - directory, - originalCwd, - ); - const directBinary = join(absoluteDirectory, "codex.exe"); - if (!isWindowsAppsPath(directBinary) && existsSync(directBinary)) - yield directBinary; - - const packageRoot = join( - absoluteDirectory, - "node_modules", - "@openai", - "codex", - ); - const nativeBinary = resolveWindowsPackageBinary(packageRoot, architecture); - if (nativeBinary && !isWindowsAppsPath(nativeBinary)) yield nativeBinary; + if (!isWindowsAppsPath(candidate) && isExecutableFile(candidate)) + yield candidate; } } @@ -234,45 +152,6 @@ function isWindowsAppsPath(candidate: string): boolean { return /(?:^|[\\/])windowsapps(?:[\\/]|$)/iu.test(candidate); } -function resolveWindowsCachedBinary( - localAppData: string | undefined, -): string | undefined { - const root = localAppData?.trim(); - if (!root) return undefined; - - const cacheRoot = join(root, "OpenAI", "Codex", "bin"); - let selected: { path: string; modifiedAt: number } | undefined; - try { - for (const entry of readdirSync(cacheRoot, { withFileTypes: true })) { - if (!entry.isDirectory() || !/^[a-f0-9]{8,128}$/iu.test(entry.name)) - continue; - const candidate = join(cacheRoot, entry.name, "codex.exe"); - let metadata: ReturnType; - try { - metadata = statSync(candidate); - } catch { - continue; - } - if ( - !metadata.isFile() || - metadata.size === 0 || - isWindowsAppsPath(candidate) - ) - continue; - if ( - !selected || - metadata.mtimeMs > selected.modifiedAt || - (metadata.mtimeMs === selected.modifiedAt && candidate > selected.path) - ) { - selected = { path: candidate, modifiedAt: metadata.mtimeMs }; - } - } - } catch { - return undefined; - } - return selected?.path; -} - function isExecutableFile(value: string): boolean { try { if (!statSync(value).isFile()) return false; @@ -320,35 +199,3 @@ function isNativeWindowsRootRelativePath(value: string): boolean { const root = win32.parse(value).root; return root === "\\" || root === "/"; } - -function resolveWindowsPackageBinary( - packageRoot: string, - architecture: NodeJS.Architecture, -): string | undefined { - const packageJson = join(packageRoot, "package.json"); - if (!existsSync(packageJson)) return undefined; - - const targetTriple = - architecture === "arm64" - ? "aarch64-pc-windows-msvc" - : architecture === "x64" - ? "x86_64-pc-windows-msvc" - : undefined; - if (!targetTriple) return undefined; - - try { - const platformPackageJson = createRequire(packageJson).resolve( - `@openai/codex-win32-${architecture}/package.json`, - ); - const nativeBinary = join( - dirname(platformPackageJson), - "vendor", - targetTriple, - "bin", - "codex.exe", - ); - return existsSync(nativeBinary) ? nativeBinary : undefined; - } catch { - return undefined; - } -} diff --git a/plugins/codex-security/mcp-app/src/native-permissions.ts b/plugins/codex-security/mcp-app/src/native-permissions.ts index cef74bcf2..0c32cffe2 100644 --- a/plugins/codex-security/mcp-app/src/native-permissions.ts +++ b/plugins/codex-security/mcp-app/src/native-permissions.ts @@ -1,5 +1,4 @@ import { isAbsolute } from "node:path"; -import { fileURLToPath } from "node:url"; import { isDeepStrictEqual } from "node:util"; export const CODEX_SANDBOX_STATE_META_CAPABILITY = "codex/sandbox-state-meta"; @@ -15,7 +14,6 @@ export function resolveNativeParentSandbox( extra: unknown, ): NativeParentSandbox { const state = trustedSandboxState(extra); - validateSandboxCwd(state.sandboxCwd); const profile = record(state.permissionProfile); if (!profile || profile.type !== "managed") { @@ -171,29 +169,6 @@ function trustedSandboxState(extra: unknown): Record { return state; } -function validateSandboxCwd(value: unknown): void { - if (value === undefined) return; - if (typeof value !== "string" || value.trim().length === 0) { - throw unsupportedParentSandbox( - "the parent sandbox working directory is invalid", - ); - } - if (value.startsWith("file:")) { - try { - if (isAbsolute(fileURLToPath(value))) return; - } catch { - throw unsupportedParentSandbox( - "the parent sandbox working directory is invalid", - ); - } - } - if (!isAbsolute(value)) { - throw unsupportedParentSandbox( - "the parent sandbox working directory is invalid", - ); - } -} - function resolveGlobScanMaxDepth( filesystem: Record, ): number | undefined { @@ -220,18 +195,10 @@ function resolveGlobScanMaxDepth( } function validateDenyMissingPathBehavior(entry: Record): void { - const snakeCase = entry.missing_path_behavior; - const camelCase = entry.missingPathBehavior; if ( - snakeCase != null && - camelCase != null && - !isDeepStrictEqual(snakeCase, camelCase) + entry.missing_path_behavior != null || + entry.missingPathBehavior != null ) { - throw unsupportedParentSandbox( - "a parent filesystem denial has conflicting missing_path_behavior", - ); - } - if (snakeCase != null || camelCase != null) { throw unsupportedParentSandbox( "a parent filesystem denial with missing_path_behavior cannot be preserved", ); diff --git a/plugins/codex-security/mcp-app/src/python_command.ts b/plugins/codex-security/mcp-app/src/python_command.ts index e8af5e646..c801a09e1 100644 --- a/plugins/codex-security/mcp-app/src/python_command.ts +++ b/plugins/codex-security/mcp-app/src/python_command.ts @@ -21,7 +21,7 @@ const MISSING_PYTHON_HELPER_MESSAGE = export async function resolvePythonCommand( options: ResolvePythonCommandOptions = {}, ): Promise { - const configuredPython = options.configuredPython ?? process.env.PYTHON; + const configuredPython = options.configuredPython ?? process.env["PYTHON"]; if (configuredPython?.trim()) { return configuredPython.trim(); } @@ -97,3 +97,27 @@ export function missingPythonHelperMessage( } return MISSING_PYTHON_HELPER_MESSAGE; } + +export function workbenchCommandTimeout(command: string | undefined): number { + return [ + "begin-deep-scan", + "complete-scan", + "export-findings", + "get-scan", + "get-workspace", + "inspect-setup", + "list-findings", + "preserve-scan-results", + "recover-scan-results", + "request-finding-remediation", + "request-finding-remediation-action", + "save-workspace", + "set-finding-triage", + "set-finding-remediation", + "start-headless-standard-scan", + "start-prompt-only-scan", + "start-scan", + ].includes(command ?? "") + ? 300_000 + : 30_000; +} diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_attack_path.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_attack_path.mjs index bea7a70e2..c4ac08545 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_attack_path.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_attack_path.mjs @@ -293,7 +293,7 @@ async function testEmptyLedgerAcceptsAnEmptyBatch() { await assert.rejects( recordCodexSecurityCandidateAttackPaths( - { ...fixture.context, layout: "worker" }, + { ...fixture.context, scanId: undefined }, { attackPaths: [] }, ), /scan-bound artifact context/, @@ -323,7 +323,6 @@ async function createFixture(label, originalRows) { context: { root, repoRoot: root, - layout: "scan", scanId, }, ledgerPath, diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_discovery.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_discovery.mjs index 1a663ab60..e8d232441 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_discovery.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_discovery.mjs @@ -114,7 +114,7 @@ try { "first\nsecond\n", ); - const scan = await createContext(root, repoRoot, "scan", "scan"); + const scan = await createContext(root, repoRoot, "scan"); await verifyInputSchema(); await verifyNormalizationAndPagination(scan); await verifyReaderPreservesSharedPhaseRecords(scan); @@ -393,7 +393,7 @@ async function verifyNormalizerFailuresPreserveOutput(context) { } async function verifyDiffInventoryAllowsDeletedFiles(root, repoRoot) { - const context = await createContext(root, repoRoot, "diff-output", "scan"); + const context = await createContext(root, repoRoot, "diff-output"); const inventory = path.join( context.root, "artifacts", @@ -482,7 +482,7 @@ async function verifyEmptyReplacement(context) { } async function verifyWorkerContext(root, repoRoot) { - const worker = await createContext(root, repoRoot, "worker-output", "worker"); + const worker = await createContext(root, repoRoot, "worker-output"); const result = await recordCodexSecurityDiscoveryCandidates( { candidates: [rawCandidate()], @@ -496,12 +496,7 @@ async function verifyWorkerContext(root, repoRoot) { } async function verifyMalformedLedgerIsNotModified(root, repoRoot) { - const context = await createContext( - root, - repoRoot, - "malformed-output", - "scan", - ); + const context = await createContext(root, repoRoot, "malformed-output"); const destination = path.join( context.root, "artifacts", @@ -520,7 +515,7 @@ async function verifyMalformedLedgerIsNotModified(root, repoRoot) { async function verifySymlinkRejection(root, repoRoot) { if (process.platform === "win32") return; - const context = await createContext(root, repoRoot, "unsafe-output", "scan"); + const context = await createContext(root, repoRoot, "unsafe-output"); const outside = path.join(root, "outside.jsonl"); await writeFile(outside, "outside must not change\n"); const destination = path.join( @@ -545,7 +540,7 @@ async function verifySymlinkRejection(root, repoRoot) { ); } -async function createContext(root, repoRoot, name, layout) { +async function createContext(root, repoRoot, name) { const artifactRoot = path.join(root, name); const discoveryDirectory = path.join( artifactRoot, @@ -560,7 +555,6 @@ async function createContext(root, repoRoot, name, layout) { return { root: artifactRoot, repoRoot, - layout, pluginRoot, }; } diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_foundation.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_foundation.mjs index ee675dc27..038a5661e 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_foundation.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_foundation.mjs @@ -196,7 +196,6 @@ async function testScanContext() { assert.deepEqual(calls, [["get-scan", "--scan-id", scanId]]); assert.equal(context.root, await realpath(root)); assert.equal(context.repoRoot, await realpath(repoRoot)); - assert.equal(context.layout, "scan"); assert.equal(context.scope, "."); assert.equal(context.mode, "deep"); assert.deepEqual(context.targetContract, contract); @@ -239,7 +238,6 @@ async function testSafeJsonAndJsonl() { const context = { root: path.join(fixture, "scan"), repoRoot: path.join(fixture, "repository"), - layout: "scan", }; const components = ["artifacts", "02_discovery", "candidate_ledger.jsonl"]; const destination = await io.artifactDestination( @@ -320,7 +318,6 @@ async function testAtomicReplaceAndAppend() { const context = { root: path.join(fixture, "scan"), repoRoot: path.join(fixture, "repository"), - layout: "scan", }; const components = ["artifacts", "02_discovery", "candidate_ledger.jsonl"]; const destination = await io.artifactDestination( @@ -393,7 +390,6 @@ async function testUnsafeArtifacts() { const context = { root: path.join(fixture, "scan"), repoRoot: path.join(fixture, "repository"), - layout: "scan", }; for (const components of [ [], diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_inventory.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_inventory.mjs index 9ff683970..29f48bd49 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_inventory.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_inventory.mjs @@ -424,7 +424,6 @@ async function createFixture(label) { scan: { root: scanRoot, repoRoot, - layout: "scan", scanId: "f84c8312-a602-4660-8e01-518a176cd75a", scope: ".", pluginRoot, @@ -433,7 +432,7 @@ async function createFixture(label) { worker: { root: workerRoot, repoRoot, - layout: "worker", + scanId: undefined, }, }; } diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_scan_draft.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_scan_draft.mjs index 2a6786991..c8f99201c 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_scan_draft.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_scan_draft.mjs @@ -1,4 +1,5 @@ import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; import { createHash } from "node:crypto"; import { mkdir, @@ -32,9 +33,8 @@ const module = await import( const { completedScanInputSchema, getCodexSecurityCompletedScan, - recordCodexSecurityScanDraft, + recordCodexSecurityScanDraft: publishScanDraft, recordCodexSecurityScanDraftViaWorkbench, - saveScanDraftCheckpoint, scanDraftInputSchema, } = module; @@ -46,7 +46,6 @@ try { const context = { root, repoRoot: root, - layout: "scan", scanId, scope: ".", mode: "standard", @@ -148,6 +147,105 @@ try { 1, ); + // This is the exact current parent payload that merge_saved_results persists. + const producerCheckpoint = JSON.parse( + execFileSync( + process.env.PYTHON ?? + (process.platform === "win32" ? "python" : "python3"), + [ + "-B", + "-c", + ` +import json, sys +from pathlib import Path +sys.path.insert(0, sys.argv[1]) +from workbench_saved_results import _read_saved_parent_result +print(json.dumps(_read_saved_parent_result(Path(sys.argv[2]), sys.argv[3])[1])) +`, + path.resolve(import.meta.dirname, "../../scripts"), + parentCheckpointRoot, + scanId, + ], + { encoding: "utf8" }, + ), + ); + assert.deepEqual(producerCheckpoint.scope.includePaths, ["."]); + assert.deepEqual(producerCheckpoint.coverage.includePaths, ["."]); + assert.deepEqual(producerCheckpoint.coverage.excludePaths, []); + assert.equal(producerCheckpoint.coverage.mode, "repository"); + assert.equal(producerCheckpoint.coverage.inventoryStrategy, "repository"); + const producerContext = { ...context, root: parentCheckpointRoot }; + const producerPath = await saveScanDraftCheckpoint( + producerContext, + producerCheckpoint, + ); + const producerBytes = await readFile(producerPath, "utf8"); + await recordCodexSecurityScanDraft(producerContext, { + ...input, + findings: [], + }); + assert.deepEqual( + (await readJson(parentCheckpointRoot, "findings.json")).findings, + producerCheckpoint.findings, + ); + assert.deepEqual( + (await readJson(parentCheckpointRoot, "coverage.json")).includePaths, + ["."], + ); + assert.equal(await readFile(producerPath, "utf8"), producerBytes); + for (const [name, changed, expected] of [ + [ + "scan-identity", + { scanId: "d7caa0cf-b785-47ef-95e7-e753dc288608" }, + /different scan/, + ], + [ + "coverage", + { coverage: { ...producerCheckpoint.coverage, completeness: "invalid" } }, + /current schema/, + ], + ]) { + const invalidContext = { + ...context, + root: path.join(root, "producer-" + name), + }; + await saveScanDraftCheckpoint(invalidContext, { + ...producerCheckpoint, + ...changed, + }); + await assert.rejects( + recordCodexSecurityScanDraft(invalidContext, input), + expected, + ); + } + + const unsupportedCheckpointContext = { + ...context, + root: path.join(root, "unsupported-checkpoint"), + }; + const { handoffClaimToken: _oldClaim, ...oldCheckpoint } = { + ...input, + findings: [ + { ...finding, validation: { assertions: "obsolete string list" } }, + ], + }; + await saveScanDraftCheckpoint(unsupportedCheckpointContext, oldCheckpoint); + await assert.rejects( + recordCodexSecurityScanDraft(unsupportedCheckpointContext, input), + /Saved scan draft does not match the current schema/, + ); + const retained = await readdir( + path.join(unsupportedCheckpointContext.root, "checkpoints"), + ); + assert.equal(retained.length, 1); + assert.deepEqual( + await readJson( + unsupportedCheckpointContext.root, + "checkpoints/" + retained[0], + ), + oldCheckpoint, + ); + const interruptedParentRoot = path.join( root, "interrupted-checkpoint-parent", @@ -2247,8 +2345,8 @@ try { /handoffClaimToken/, ); await assert.rejects( - recordCodexSecurityScanDraft({ ...context, layout: "worker" }, input), - /authoritative parent scan context/, + recordCodexSecurityScanDraft({ ...context, scanId: undefined }, input), + /scanId does not match/, ); await assert.rejects( recordCodexSecurityScanDraft({ ...context, status: "complete" }, input), @@ -2507,3 +2605,37 @@ async function recordFreshScanDraft(context, input) { ]); return recordCodexSecurityScanDraft(context, input); } + +// Exercise reconciliation with explicit fixture publication; production publishes under the workbench lock. +function recordCodexSecurityScanDraft(context, input, publish, signal) { + return publishScanDraft( + context, + input, + publish ?? + (async (draft, _digest, checkpoint) => { + await saveScanDraftCheckpoint(context, checkpoint); + for (const [name, document] of Object.entries({ + "scan-manifest.json": draft.manifest, + "findings.json": draft.findings, + "coverage.json": draft.coverage, + })) + await writeFile( + path.join(context.root, name), + JSON.stringify(document) + "\n", + ); + }), + signal, + ); +} + +async function saveScanDraftCheckpoint(context, input) { + const { handoffClaimToken: _claim, ...snapshot } = input; + const digest = createHash("sha256") + .update(JSON.stringify(snapshot)) + .digest("hex"); + const directory = path.join(context.root, "checkpoints"); + await mkdir(directory, { recursive: true }); + const checkpointPath = path.join(directory, digest + ".json"); + await writeFile(checkpointPath, JSON.stringify(snapshot) + "\n"); + return checkpointPath; +} diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_storage.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_storage.mjs index 906b6b3ae..fc0e7bd6f 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_storage.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_storage.mjs @@ -15,6 +15,7 @@ import { fileURLToPath } from "node:url"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; import { build } from "esbuild"; +import { mcpBundleOptions } from "../scripts/bundle_options.mjs"; const applicationRoot = path.resolve( path.dirname(fileURLToPath(import.meta.url)), @@ -33,21 +34,14 @@ try { await mkdir(repository); await writeFile(path.join(repository, "example.py"), "value = 1\n"); await build({ - bundle: true, - banner: { - js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", - }, + ...mcpBundleOptions, define: { __dirname: JSON.stringify(applicationRoot), - "import.meta.url": "__codexSecurityModuleUrl", + ...mcpBundleOptions.define, }, entryPoints: [path.join(applicationRoot, "main.ts")], - external: ["fsevents"], - format: "cjs", - loader: { ".md": "text" }, logLevel: "silent", outfile: bundle, - platform: "node", }); client = await connect(); const started = await call("start_codex_security_standard_scan", { diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_storage_regressions.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_storage_regressions.mjs index 8810bcfbf..65e05d6bf 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_storage_regressions.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_storage_regressions.mjs @@ -10,6 +10,7 @@ import { promisify } from "node:util"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; import { build } from "esbuild"; +import { mcpBundleOptions } from "../scripts/bundle_options.mjs"; const applicationRoot = path.resolve( path.dirname(fileURLToPath(import.meta.url)), @@ -37,21 +38,14 @@ const temporaryDirectories = []; await fs.mkdir(repository); await fs.writeFile(path.join(repository, "example.py"), "value = 1\n"); await build({ - bundle: true, - banner: { - js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", - }, + ...mcpBundleOptions, define: { __dirname: JSON.stringify(applicationRoot), - "import.meta.url": "__codexSecurityModuleUrl", + ...mcpBundleOptions.define, }, entryPoints: [path.join(applicationRoot, "main.ts")], - external: ["fsevents"], - format: "cjs", - loader: { ".md": "text" }, logLevel: "silent", outfile: bundle, - platform: "node", }); await build({ bundle: true, @@ -125,7 +119,6 @@ try { `deep-scan-rejected-${scanId ?? "standalone"}`, ), repoRoot: repository, - layout: "scan", scanId, }; for (const artifact of [ @@ -162,7 +155,6 @@ try { const context = { root: path.join(fixture, "deep-scan-allowed"), repoRoot: repository, - layout: "scan", }; await fs.mkdir(context.root); const artifact = "artifacts/deep-scan/checkpoint.json"; diff --git a/plugins/codex-security/mcp-app/tests/test_artifact_validation_phase.mjs b/plugins/codex-security/mcp-app/tests/test_artifact_validation_phase.mjs index 7a0049246..3c76d1609 100644 --- a/plugins/codex-security/mcp-app/tests/test_artifact_validation_phase.mjs +++ b/plugins/codex-security/mcp-app/tests/test_artifact_validation_phase.mjs @@ -209,7 +209,7 @@ try { ); await assertNoMutation( - { ...context, layout: "worker" }, + { ...context, scanId: undefined }, ledger, { validations: updates, @@ -276,7 +276,7 @@ async function scanContext(root, directory, scanId) { }), mkdir(repository, { recursive: true }), ]); - return { root: scanRoot, repoRoot: repository, layout: "scan", scanId }; + return { root: scanRoot, repoRoot: repository, scanId }; } function candidate(candidateId, sourcePath) { diff --git a/plugins/codex-security/mcp-app/tests/test_compact_artifact_server.mjs b/plugins/codex-security/mcp-app/tests/test_compact_artifact_server.mjs index 76cbb9060..7e5f1173a 100644 --- a/plugins/codex-security/mcp-app/tests/test_compact_artifact_server.mjs +++ b/plugins/codex-security/mcp-app/tests/test_compact_artifact_server.mjs @@ -8,6 +8,7 @@ import { fileURLToPath, pathToFileURL } from "node:url"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; import { build } from "esbuild"; +import { mcpBundleOptions } from "../scripts/bundle_options.mjs"; const applicationRoot = path.resolve( path.dirname(fileURLToPath(import.meta.url)), @@ -1920,23 +1921,14 @@ async function testParentToolList(bundle) { async function bundleEntrypoint(entrypoint, outfile) { await build({ - bundle: true, - banner: { - js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", - }, + ...mcpBundleOptions, define: { __dirname: JSON.stringify(applicationRoot), - "import.meta.url": "__codexSecurityModuleUrl", + ...mcpBundleOptions.define, }, entryPoints: [path.join(applicationRoot, entrypoint)], - external: ["fsevents"], - format: "cjs", - loader: { ".md": "text" }, logLevel: "silent", - logOverride: { "empty-import-meta": "silent" }, outfile, - platform: "node", - target: "node20", }); } diff --git a/plugins/codex-security/mcp-app/tests/test_mcp_app_smoke.mjs b/plugins/codex-security/mcp-app/tests/test_mcp_app_smoke.mjs index 2fb942918..e910e107f 100644 --- a/plugins/codex-security/mcp-app/tests/test_mcp_app_smoke.mjs +++ b/plugins/codex-security/mcp-app/tests/test_mcp_app_smoke.mjs @@ -112,7 +112,7 @@ const scanHandoffSource = await readFile( const serverSource = await readFile(path.join(mcpAppRoot, "server.ts"), "utf8"); assert.match( serverSource, - /timeout:\s*\[[^\]]*"start-prompt-only-scan"[^\]]*\]\.includes\(args\[0\] \?\? ""\)\s*\?\s*300_000\s*:\s*30_000/, + /timeout: workbenchCommandTimeout\(args\[0\]\)/, "Prompt-only scan startup must use the same five-minute timeout as other scan starts.", ); const authenticatedArtifactClaimSource = serverSource.match( diff --git a/plugins/codex-security/mcp-app/tests/test_native_executable.mjs b/plugins/codex-security/mcp-app/tests/test_native_executable.mjs index 24e52002f..9b7eee6fe 100644 --- a/plugins/codex-security/mcp-app/tests/test_native_executable.mjs +++ b/plugins/codex-security/mcp-app/tests/test_native_executable.mjs @@ -7,7 +7,6 @@ import { realpath, rm, symlink, - utimes, writeFile, } from "node:fs/promises"; import { tmpdir } from "node:os"; @@ -30,10 +29,8 @@ const { resolveCodexPath, resolveTrustedCodex, snapshotNativeEnvironment } = ); const temporaryRoots = []; try { - await testWindowsAppsCodexFallsBackToRelocatedBinary(); - await testWindowsNpmPackageResolution(); - await testWindowsNpmPackageResolution("managed"); await testCodexHomePathsStayBoundToOriginalDirectory(); + await testExplicitAndPathExecutables(); if (process.platform === "win32") { await testWindowsWorkerEnvironmentPreservesMixedCaseKeys(); await testWindowsLauncherSkipsExtensionlessNpmShim(); @@ -48,12 +45,7 @@ try { ); const originalCwd = path.join(process.cwd(), "fixture-root"); assert.equal( - resolveCodexPath( - { CODEX_CLI_PATH: "fixture/codex" }, - "linux", - process.arch, - originalCwd, - ), + resolveCodexPath({ CODEX_CLI_PATH: "fixture/codex" }, "linux", originalCwd), path.join(originalCwd, "fixture/codex"), ); } finally { @@ -62,13 +54,43 @@ try { ); } +async function testExplicitAndPathExecutables() { + const root = await mkdtemp(path.join(tmpdir(), "codex-security-executable-")); + temporaryRoots.push(root); + const repository = path.join(root, "repository"); + const bin = path.join(repository, "bin"); + const external = path.join(root, "external"); + await Promise.all([mkdir(bin, { recursive: true }), mkdir(external)]); + const name = process.platform === "win32" ? "codex.exe" : "codex"; + const selected = path.join(external, name); + await Promise.all([ + copyFile(process.execPath, path.join(bin, name)), + copyFile(process.execPath, selected), + ]); + for (const configured of [undefined, "codex", selected]) { + const environment = { + PATH: [bin, external].join(path.delimiter), + ...(configured === undefined ? {} : { CODEX_CLI_PATH: configured }), + }; + const trusted = await resolveTrustedCodex(environment, repository); + assert.equal(await realpath(trusted.executable), await realpath(selected)); + assert.equal(trusted.environment.PATH, await realpath(external)); + } + assert.equal( + await resolveTrustedCodex( + { CODEX_CLI_PATH: path.join(bin, name) }, + repository, + ), + null, + ); +} + async function testCodexHomePathsStayBoundToOriginalDirectory() { if (process.platform === "win32") { assert.equal( resolveCodexPath( { CODEX_CLI_PATH: "\\Tools\\codex.exe" }, "win32", - process.arch, "C:\\original\\cwd", ), "C:\\Tools\\codex.exe", @@ -77,7 +99,6 @@ async function testCodexHomePathsStayBoundToOriginalDirectory() { resolveCodexPath( { CODEX_CLI_PATH: "/Tools/codex.exe" }, "win32", - process.arch, "D:\\original\\cwd", ), "D:\\Tools\\codex.exe", @@ -183,161 +204,6 @@ async function testWindowsWorkerEnvironmentPreservesMixedCaseKeys() { } } -async function testWindowsAppsCodexFallsBackToRelocatedBinary() { - const root = await mkdtemp( - path.join(tmpdir(), "codex-security-windows-cache-"), - ); - temporaryRoots.push(root); - const localAppData = path.join(root, "LocalAppData"); - const olderBinary = path.join( - localAppData, - "OpenAI", - "Codex", - "bin", - "11111111", - "codex.exe", - ); - const currentBinary = path.join( - localAppData, - "OpenAI", - "Codex", - "bin", - "22222222", - "codex.exe", - ); - const emptyBinary = path.join( - localAppData, - "OpenAI", - "Codex", - "bin", - "33333333", - "codex.exe", - ); - const protectedDirectory = path.join( - root, - "WindowsApps", - "OpenAI.Codex_fixture", - "resources", - ); - const architecture = process.arch === "arm64" ? "arm64" : "x64"; - const targetTriple = - architecture === "arm64" - ? "aarch64-pc-windows-msvc" - : "x86_64-pc-windows-msvc"; - const managedPackage = path.join( - protectedDirectory, - "node_modules", - "@openai", - "codex", - ); - const platformPackage = path.join( - managedPackage, - "node_modules", - "@openai", - `codex-win32-${architecture}`, - ); - const protectedPackageBinary = path.join( - platformPackage, - "vendor", - targetTriple, - "bin", - "codex.exe", - ); - await Promise.all([ - mkdir(path.dirname(olderBinary), { recursive: true }), - mkdir(path.dirname(currentBinary), { recursive: true }), - mkdir(path.dirname(emptyBinary), { recursive: true }), - mkdir(path.dirname(protectedPackageBinary), { recursive: true }), - ]); - await Promise.all([ - copyFile(process.execPath, olderBinary), - copyFile(process.execPath, currentBinary), - writeFile(emptyBinary, ""), - writeFile( - path.join(protectedDirectory, "codex.exe"), - "protected direct binary", - ), - writeFile( - path.join(managedPackage, "package.json"), - JSON.stringify({ name: "@openai/codex" }), - ), - writeFile( - path.join(platformPackage, "package.json"), - JSON.stringify({ name: `@openai/codex-win32-${architecture}` }), - ), - writeFile(protectedPackageBinary, "protected package binary"), - ]); - await Promise.all([ - utimes(olderBinary, new Date(1_000), new Date(1_000)), - utimes(currentBinary, new Date(2_000), new Date(2_000)), - utimes(emptyBinary, new Date(3_000), new Date(3_000)), - ]); - - const resolved = resolveCodexPath( - { - CODEX_CLI_PATH: - "C:\\Program Files\\WindowsApps\\OpenAI.Codex_fixture\\resources\\codex.exe", - LOCALAPPDATA: localAppData, - }, - "win32", - ); - assert.equal(resolved, currentBinary); - assert.equal( - resolveCodexPath( - { - CODEX_MANAGED_PACKAGE_ROOT: managedPackage, - Path: protectedDirectory, - LOCALAPPDATA: localAppData, - }, - "win32", - architecture, - ), - currentBinary, - ); - assert.equal( - resolveCodexPath( - { - LOCALAPPDATA: path.relative(root, localAppData), - }, - "win32", - architecture, - root, - ), - currentBinary, - ); - assert.equal( - resolveCodexPath( - { - localappdata: localAppData, - }, - "win32", - architecture, - ), - currentBinary, - ); - const explicitOverride = path.join(root, "custom-codex.exe"); - assert.equal( - resolveCodexPath( - { - CODEX_CLI_PATH: explicitOverride, - CODEX_MANAGED_PACKAGE_ROOT: managedPackage, - Path: protectedDirectory, - LOCALAPPDATA: localAppData, - }, - "win32", - architecture, - ), - explicitOverride, - ); - - if (process.platform === "win32") { - const launched = spawnSync(resolved, ["--version"], { encoding: "utf8" }); - assert.equal(launched.error, undefined); - assert.equal(launched.status, 0); - assert.equal(launched.stdout.trim(), process.version); - } -} - async function testWindowsLauncherSkipsExtensionlessNpmShim() { const root = await mkdtemp( path.join(tmpdir(), "codex-security-windows-launcher-"), @@ -373,152 +239,6 @@ async function testWindowsLauncherSkipsExtensionlessNpmShim() { assert.equal(fixed.stdout.trim(), process.version); } -async function testWindowsNpmPackageResolution(installation = "global") { - const root = await mkdtemp( - path.join(tmpdir(), "codex-security windows-npm-"), - ); - temporaryRoots.push(root); - const architecture = process.arch === "arm64" ? "arm64" : "x64"; - const targetTriple = - architecture === "arm64" - ? "aarch64-pc-windows-msvc" - : "x86_64-pc-windows-msvc"; - const packageDirectory = - installation === "managed" - ? path.join(root, "node_modules") - : path.join(root, "npm", "node_modules"); - const shimDirectory = - installation === "managed" - ? path.join(packageDirectory, ".bin") - : path.join(root, "npm"); - const codexPackage = path.join(packageDirectory, "@openai", "codex"); - const platformPackage = path.join( - codexPackage, - "node_modules", - "@openai", - `codex-win32-${architecture}`, - ); - const nativeBinary = path.join( - platformPackage, - "vendor", - targetTriple, - "bin", - "codex.exe", - ); - await Promise.all([ - mkdir(path.dirname(nativeBinary), { recursive: true }), - mkdir(shimDirectory, { recursive: true }), - ]); - await Promise.all([ - writeFile(path.join(shimDirectory, "codex"), "#!/bin/sh\nexit 1\n"), - writeFile( - path.join(codexPackage, "package.json"), - JSON.stringify({ name: "@openai/codex" }), - ), - writeFile( - path.join(platformPackage, "package.json"), - JSON.stringify({ name: `@openai/codex-win32-${architecture}` }), - ), - copyFile(process.execPath, nativeBinary), - ]); - - const environment = windowsLauncherEnvironment(shimDirectory); - if (installation === "managed") { - environment.CODEX_MANAGED_PACKAGE_ROOT = codexPackage; - } - assert.equal( - await realpath(resolveCodexPath(environment, "win32", architecture)), - await realpath(nativeBinary), - ); - const repository = path.join(root, "repository"); - await mkdir(repository); - const selected = await resolveTrustedCodex( - environment, - repository, - "win32", - architecture, - ); - assert.ok(selected); - assert.equal( - await realpath(selected.executable), - await realpath(nativeBinary), - ); - if (installation === "global") { - const repositoryBin = path.join(repository, "bin"); - const alias = path.join(root, "repository-bin-alias"); - await mkdir(repositoryBin, { recursive: true }); - await copyFile(process.execPath, path.join(repositoryBin, "codex.exe")); - await symlink(repositoryBin, alias, "junction"); - for (const configured of [undefined, "codex", "codex.exe"]) { - for (const quoted of [false, true]) { - const directories = [repositoryBin, alias, shimDirectory].map( - (directory) => (quoted ? `"${directory}"` : directory), - ); - const search = windowsLauncherEnvironment(...directories); - if (configured !== undefined) search.CODEX_CLI_PATH = configured; - const trusted = await resolveTrustedCodex( - search, - repository, - "win32", - architecture, - ); - assert.ok( - trusted, - "a later trusted npm installation must remain discoverable", - ); - assert.equal( - await realpath(trusted.executable), - await realpath(nativeBinary), - ); - if (!quoted || process.platform === "win32") { - assert.equal(trusted.environment.PATH, await realpath(shimDirectory)); - } - assert.equal(search.Path, directories.join(path.delimiter)); - } - } - const explicit = windowsLauncherEnvironment(repositoryBin, shimDirectory); - explicit.CODEX_CLI_PATH = path.join(repositoryBin, "codex.exe"); - assert.equal( - await resolveTrustedCodex(explicit, repository, "win32", architecture), - null, - ); - } - if (installation === "managed") { - const mixedCaseEnvironment = { - ...environment, - codex_managed_package_root: codexPackage, - }; - delete mixedCaseEnvironment.CODEX_MANAGED_PACKAGE_ROOT; - assert.equal( - await realpath( - resolveCodexPath(mixedCaseEnvironment, "win32", architecture), - ), - await realpath(nativeBinary), - ); - } - if (process.platform === "win32") { - assert.equal( - spawnSync("codex.exe", ["--version"], { - encoding: "utf8", - env: environment, - }).error?.code, - "ENOENT", - ); - - const fixed = spawnSync( - resolveCodexPath(environment, "win32", architecture), - ["--version"], - { - encoding: "utf8", - env: environment, - }, - ); - assert.equal(fixed.error, undefined); - assert.equal(fixed.status, 0); - assert.equal(fixed.stdout.trim(), process.version); - } -} - function windowsLauncherEnvironment(...directories) { const environment = { ...process.env }; for (const key of Object.keys(environment)) { diff --git a/plugins/codex-security/mcp-app/tests/test_native_permissions.mjs b/plugins/codex-security/mcp-app/tests/test_native_permissions.mjs index b6b4e19d4..e48e499ec 100644 --- a/plugins/codex-security/mcp-app/tests/test_native_permissions.mjs +++ b/plugins/codex-security/mcp-app/tests/test_native_permissions.mjs @@ -142,24 +142,10 @@ assert.throws( /symbolic project-roots denial metadata/i.test(error.message), ); -const pinnedFileUri = extra( - pinnedReadOnly, - "file:///tmp/codex-security-parent", -); -assert.deepEqual(resolveNativeParentSandbox(pinnedFileUri), { - filesystemDenies: [], -}); -assert.deepEqual( - resolveNativeParentSandbox( - extra(pinnedReadOnly, "/tmp/codex-security-parent"), - ), - { - filesystemDenies: [], - }, -); +const parentMetadata = extra(pinnedReadOnly); assert.deepEqual( resolveNativeParentSandbox({ - requestInfo: pinnedFileUri, + requestInfo: parentMetadata, }), { filesystemDenies: [], @@ -167,8 +153,8 @@ assert.deepEqual( ); assert.deepEqual( resolveNativeParentSandbox({ - _meta: pinnedFileUri._meta, - requestInfo: pinnedFileUri, + _meta: parentMetadata._meta, + requestInfo: parentMetadata, }), { filesystemDenies: [], @@ -336,8 +322,6 @@ for (const invalid of [ entries: [{ path: { type: "path", path: "" }, access: "read" }], }, }), - extra(pinnedReadOnly, "relative/working-directory"), - extra(pinnedReadOnly, "file://remote-host/tmp/codex-security-parent"), { _meta: extra(pinnedReadOnly)._meta, requestInfo: extra({ ...pinnedReadOnly, network: "enabled" }), diff --git a/plugins/codex-security/mcp-app/tests/test_native_scan.mjs b/plugins/codex-security/mcp-app/tests/test_native_scan.mjs index 866e36ec0..549ba09f2 100644 --- a/plugins/codex-security/mcp-app/tests/test_native_scan.mjs +++ b/plugins/codex-security/mcp-app/tests/test_native_scan.mjs @@ -13,7 +13,6 @@ import { createRequire } from "node:module"; import { tmpdir } from "node:os"; import { delimiter, dirname, join, sep } from "node:path"; import { after, test } from "node:test"; -import { setTimeout as delay } from "node:timers/promises"; import { fileURLToPath } from "node:url"; import { build } from "esbuild"; import { parse as parseToml } from "smol-toml"; @@ -183,39 +182,6 @@ test("native preparation requires an external executable for fresh and resumed c const installations = [ { bin: externalBin, executable: externalExecutable }, ]; - if (process.platform === "win32") { - const npmBin = join(root, "global npm"); - const codexPackage = join(npmBin, "node_modules", "@openai", "codex"); - const architecture = process.arch === "arm64" ? "arm64" : "x64"; - const platformPackage = join( - codexPackage, - "node_modules", - "@openai", - `codex-win32-${architecture}`, - ); - const triple = - architecture === "arm64" - ? "aarch64-pc-windows-msvc" - : "x86_64-pc-windows-msvc"; - const npmExecutable = join( - platformPackage, - "vendor", - triple, - "bin", - "codex.exe", - ); - await mkdir(dirname(npmExecutable), { recursive: true }); - await writeFile( - join(codexPackage, "package.json"), - JSON.stringify({ name: "@openai/codex" }), - ); - await writeFile( - join(platformPackage, "package.json"), - JSON.stringify({ name: `@openai/codex-win32-${architecture}` }), - ); - await writeFile(npmExecutable, "inert executable fixture"); - installations.push({ bin: npmBin, executable: npmExecutable }); - } await symlink( bin, alias, @@ -793,223 +759,6 @@ test("native launches snapshot safety identifiers and prefer saved recipes", asy } }); -test( - "native worker turns verify the selected permissions before fresh and resumed execution", - { - skip: - process.platform === "win32" - ? "Synthetic executable uses a POSIX shebang." - : false, - }, - async () => { - const root = await realpath( - await mkdtemp(join(tmpdir(), "native-permission-child-")), - ); - const executable = join(root, "codex"); - const capture = join(root, "capture.jsonl"); - const scenario = join(root, "scenario"); - const fallback = - "Configured value for `permission_profile` is disallowed by requirements; falling back from `codex_security_scan` to required value `:read-only`."; - const observations = async () => - (await readFile(capture, "utf8")) - .trim() - .split("\n") - .filter(Boolean) - .map(JSON.parse); - const rawConfig = (argv) => - argv.flatMap((value, index) => - value === "--config" || value === "-c" ? [argv[index + 1]] : [], - ); - const permissionError = (error) => { - assert.equal(error.constructor.name, "ScanPermissionError"); - return true; - }; - try { - await writeFile( - executable, - `#!${process.execPath} -const fs = require("node:fs"); -${syntheticPermissionAppServer()} -if (process.argv.includes("app-server")) { - servePermissionProfiles(); -} else { - const capture = (value) => fs.appendFileSync(process.env.NATIVE_PROFILE_CAPTURE, JSON.stringify(value) + "\\n"); - capture({ kind: "exec", argv: process.argv.slice(2), marker: process.env.NATIVE_PROFILE_MARKER, - codex: process.env.CODEX_API_KEY, openai: process.env.OPENAI_API_KEY, - gitEnvironment: Object.fromEntries(["PATH", "CODEX_SECURITY_GIT", "GIT_SSH_COMMAND", "GIT_CONFIG_GLOBAL"].map(name => [name, process.env[name]])) }); - console.log(JSON.stringify({ type: "thread.started", thread_id: "synthetic-worker-thread" })); - const scenario = fs.readFileSync(process.env.NATIVE_PROFILE_SCENARIO, "utf8").trim(); - if (scenario.startsWith("late-fallback")) { - process.on("SIGTERM", () => { capture({ kind: "aborted" }); process.exit(0); }); - console.log(JSON.stringify(scenario === "late-fallback-error" - ? { type: "error", message: ${JSON.stringify(fallback)} } - : { type: "item.completed", item: { type: "error", message: ${JSON.stringify(fallback)} } })); - setTimeout(() => { - console.log(JSON.stringify({ type: "item.completed", item: { type: "agent_message", text: "must not be consumed" } })); - console.log(JSON.stringify({ type: "turn.completed", usage: { input_tokens: 0, cached_input_tokens: 0, output_tokens: 0 } })); - process.exit(0); - }, 500); - } else { - console.log(JSON.stringify({ type: "turn.completed", usage: { input_tokens: 0, cached_input_tokens: 0, output_tokens: 0 } })); - } -} -`, - { mode: 0o700 }, - ); - for (const role of ["discovery", "merge", "comparison"]) { - const cwd = join(root, role); - await mkdir(cwd); - const config = scanRuntimeCodexConfig( - { approval_policy: "on-request" }, - root, - { - filesystem: { - [join(root, "private")]: "deny", - glob_scan_max_depth: 3, - }, - network: { enabled: false }, - }, - ); - const profileId = - role === "comparison" - ? "codex_security_comparison" - : "codex_security_scan"; - const configOverrides = [ - `default_permissions=${JSON.stringify(profileId)}`, - ]; - if (role === "comparison") { - configOverrides.push( - `permissions.codex_security_comparison={extends=":read-only",filesystem={${JSON.stringify(join(root, "private"))}="deny"},network={enabled=false}}`, - ); - } - const gitEnvironment = { - PATH: join(root, "selected tools"), - CODEX_SECURITY_GIT: join(root, "selected tools", "git"), - GIT_SSH_COMMAND: "synthetic-ssh --fixture", - GIT_CONFIG_GLOBAL: join(root, "operator.gitconfig"), - }; - const sdk = createPermissionCheckedCodex({ - codexPathOverride: executable, - config: { ...config, default_permissions: ":read-only" }, - configOverrides, - apiKey: "synthetic-final-key", - env: { - CODEX_HOME: root, - CODEX_API_KEY: "synthetic-stale-key", - NATIVE_PROFILE_CAPTURE: capture, - NATIVE_PROFILE_SCENARIO: scenario, - NATIVE_PROFILE_MARKER: role, - ...gitEnvironment, - }, - }); - for (const resumed of [false, true]) { - const options = { - workingDirectory: cwd, - skipGitRepoCheck: true, - approvalPolicy: "never", - ...(role === "comparison" - ? { networkAccessEnabled: false, webSearchMode: "disabled" } - : {}), - }; - const thread = resumed - ? sdk.resumeThread(`synthetic-${role}-thread`, options) - : sdk.startThread(options); - await writeFile(scenario, "valid"); - await writeFile(capture, ""); - const events = await collectNativeEvents( - thread, - "Synthetic permission verification.", - ); - assert.equal(events.at(-1).type, "turn.completed"); - assert.equal(thread.id, "synthetic-worker-thread"); - const observed = await observations(); - const preflight = observed.find( - (entry) => entry.kind === "preflight", - ); - const executed = observed.find((entry) => entry.kind === "exec"); - assert.equal(preflight.cwd, cwd); - assert.equal(executed.argv[executed.argv.indexOf("--cd") + 1], cwd); - assert.equal(executed.argv.includes("resume"), resumed); - assert.deepEqual(rawConfig(preflight.argv), rawConfig(executed.argv)); - assert.equal( - rawConfig(preflight.argv).at(-1), - 'approval_policy="never"', - ); - for (const process of [preflight, executed]) { - assert.deepEqual(process.gitEnvironment, gitEnvironment); - assert.equal(process.marker, role); - assert.equal(process.codex, "synthetic-final-key"); - assert.equal(process.openai, undefined); - } - const requests = observed.filter((entry) => entry.kind === "request"); - assert.deepEqual( - requests.map((entry) => entry.method), - [ - "initialize", - "initialized", - "config/read", - "permissionProfile/list", - "permissionProfile/list", - ], - ); - for (const request of requests.slice(2)) - assert.equal(request.params.cwd, cwd); - assert.equal(requests.at(-1).params.cursor, "selected-page"); - if (role === "comparison") break; - - for (const rejectedScenario of [ - "disallowed", - "substituted-default", - "substituted-profile", - ]) { - await writeFile(scenario, rejectedScenario); - await writeFile(capture, ""); - await assert.rejects( - collectNativeEvents(thread, "Recheck the same worker turn."), - permissionError, - ); - assert.equal( - (await observations()).filter((entry) => entry.kind === "exec") - .length, - 0, - ); - } - for (const lateScenario of [ - "late-fallback-item", - "late-fallback-error", - ]) { - await writeFile(scenario, lateScenario); - await writeFile(capture, ""); - const streamed = await thread.runStreamed( - "Stop on a late permission fallback.", - ); - const yielded = []; - await assert.rejects(async () => { - for await (const event of streamed.events) yielded.push(event); - }, permissionError); - assert.deepEqual( - yielded.map((event) => event.type), - ["thread.started"], - ); - for (let attempt = 0; attempt < 100; attempt += 1) { - if ( - (await observations()).some((entry) => entry.kind === "aborted") - ) - break; - await delay(10); - } - assert.ok( - (await observations()).some((entry) => entry.kind === "aborted"), - ); - } - } - } - } finally { - await rm(root, { recursive: true, force: true }); - } - }, -); - test( "native-selected credentials reach actual SDK child processes", { diff --git a/plugins/codex-security/mcp-app/tests/test_native_scan_stop.mjs b/plugins/codex-security/mcp-app/tests/test_native_scan_stop.mjs index 6509c3d64..cc3fd8f17 100644 --- a/plugins/codex-security/mcp-app/tests/test_native_scan_stop.mjs +++ b/plugins/codex-security/mcp-app/tests/test_native_scan_stop.mjs @@ -33,7 +33,8 @@ const bundle = await build({ }`, "./src/python_command.js": ` export async function resolvePythonCommand() { return "fixture-python"; } - export function missingPythonHelperMessage() {}`, + export function missingPythonHelperMessage() {} + export function workbenchCommandTimeout() { return 30000; }`, "../../../sdk/typescript/src/scan-execution.js": ` export const ScanPermissionError = fixture.ScanPermissionError;`, "node:child_process": ` @@ -69,7 +70,7 @@ function serverFor(fixture) { for (const entry of [ "completed", - "run-success", + "read-error", "run-error", "permission-error", ]) { @@ -91,6 +92,8 @@ for (const entry of [ if (command === "begin-deep-scan") return { scan }; assert.equal(args[args.indexOf("--scan-id") + 1], scan.scanId); if (command === "get-scan") { + if (entry === "read-error") + throw new Error("synthetic saved metadata unavailable"); return { scan: { ...scan, progress: { status: "complete" } } }; } assert.equal(command, "complete-scan"); @@ -135,11 +138,16 @@ for (const entry of [ result.content[0].text, entry === "permission-error" ? /synthetic permission rejection/ - : /synthetic sealed artifact mismatch/, + : entry === "read-error" + ? /synthetic saved metadata unavailable/ + : /synthetic sealed artifact mismatch/, ); assert.equal(result.structuredContent, undefined); assert.equal(runs, entry === "completed" ? 0 : 1); - assert.equal(validations, entry === "permission-error" ? 0 : 1); + assert.equal( + validations, + ["permission-error", "read-error"].includes(entry) ? 0 : 1, + ); if (entry === "permission-error") assert.equal(scan.progress.status, "complete"); }); @@ -343,21 +351,27 @@ for (const completed of [false, true]) { handoffClaimToken: "synthetic-claim", progress: { status: completed ? "complete" : "running" }, }; - const server = serverFor({ - async workbench([command]) { + const commands = []; + const fixture = { + async workbench([command, ...args]) { if (command === "list-scans") return {}; if (command === "resolve-scan-root") return { scanRoot: "/synthetic/scans" }; + commands.push(command); if (command === "begin-deep-scan") return { scan }; - assert.equal(command, "complete-scan"); + assert.ok(["complete-scan", "get-scan"].includes(command)); + assert.equal(args[args.indexOf("--scan-id") + 1], scan.scanId); return { scan: { ...scan, ...metadata, recipe: { private: "not public" } }, }; }, async run() { - return {}; + // A successful SDK run has already completed the scan; saved metadata is authoritative. + await this.workbench(["complete-scan", "--scan-id", scan.scanId]); + return { cost: { estimatedUsd: 99 }, turnResult: { usage: null } }; }, - }); + }; + const server = serverFor(fixture); const result = await server.tools.get("start_codex_security_deep_scan")( { scanId: scan.scanId, handoffClaimToken: scan.handoffClaimToken }, nativeMeta, @@ -366,5 +380,11 @@ for (const completed of [false, true]) { for (const key of Object.keys(metadata)) assert.deepEqual(result.structuredContent[key], metadata[key]); assert.equal(result.structuredContent.recipe, undefined); + assert.deepEqual( + commands, + completed + ? ["begin-deep-scan", "complete-scan"] + : ["begin-deep-scan", "complete-scan", "get-scan"], + ); }); } diff --git a/plugins/codex-security/mcp-app/tests/test_workbench_state_fallback.mjs b/plugins/codex-security/mcp-app/tests/test_workbench_state_fallback.mjs index 7a6314e5c..46ad9c527 100644 --- a/plugins/codex-security/mcp-app/tests/test_workbench_state_fallback.mjs +++ b/plugins/codex-security/mcp-app/tests/test_workbench_state_fallback.mjs @@ -15,6 +15,7 @@ import { tmpdir } from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { build } from "esbuild"; +import { mcpBundleOptions } from "../scripts/bundle_options.mjs"; if (process.platform !== "win32") { await testWorkbenchStateFallback(); @@ -50,19 +51,10 @@ async function testWorkbenchStateFallback() { await writeFile(path.join(targetPath, "fixture.py"), "print('fixture')\n"); await writeFakePython(fakePythonPath); await build({ - bundle: true, - banner: { - js: "const __codexSecurityModuleUrl = require('node:url').pathToFileURL(__filename).href;", - }, - define: { "import.meta.url": "__codexSecurityModuleUrl" }, + ...mcpBundleOptions, entryPoints: [path.join(mcpAppRoot, "main.ts")], - external: ["fsevents"], - format: "cjs", - loader: { ".md": "text" }, logLevel: "silent", outfile: serverBundlePath, - platform: "node", - target: "node20", }); try { diff --git a/plugins/codex-security/scripts/project_scan_artifacts.py b/plugins/codex-security/scripts/project_scan_artifacts.py index 9adbc7534..13308f934 100644 --- a/plugins/codex-security/scripts/project_scan_artifacts.py +++ b/plugins/codex-security/scripts/project_scan_artifacts.py @@ -57,6 +57,18 @@ def _scope_path(value: str) -> str: return normcase(value).replace("\\", "/") +def merge_coverage(target: dict[str, Any], source: dict[str, Any]) -> None: + """Retain distinct coverage rows in their original order.""" + for field in ("surfaces", "explicitExclusions", "deferred", "openQuestions"): + rows = target.setdefault(field, []) + seen = {json.dumps(row, sort_keys=True) for row in rows} + for row in source.get(field, []): + key = json.dumps(row, sort_keys=True) + if key not in seen: + rows.append(row) + seen.add(key) + + def project_scan_artifacts( parent_scan_id: str, source_scan_id: str, diff --git a/plugins/codex-security/scripts/report_projection.py b/plugins/codex-security/scripts/report_projection.py index dac44ecf5..5900f9213 100644 --- a/plugins/codex-security/scripts/report_projection.py +++ b/plugins/codex-security/scripts/report_projection.py @@ -6,6 +6,7 @@ import argparse import re from collections import Counter +from collections.abc import Iterator from typing import Any SEVERITY_ORDER = {"critical": 0, "high": 1, "medium": 2, "low": 3, "informational": 4} @@ -523,11 +524,8 @@ def _surface_notes(surface: dict[str, Any]) -> str: return _cell(f"{notes} Evidence: {evidence}") -def _remediation_section(finding: dict[str, Any]) -> list[str]: - remediation = _text(finding.get("remediation"), "No canonical remediation was recorded.") - lines = ["", "#### Remediation", "", remediation] - seen = {remediation} - originals: list[tuple[str, dict[str, Any]]] = [] +def retained_findings(finding: dict[str, Any]) -> Iterator[tuple[Any, dict[str, Any]]]: + """Yield canonical and retained findings in source order, visiting shared objects once.""" pending = [("finding", finding)] seen_findings: set[int] = set() while pending: @@ -535,7 +533,7 @@ def _remediation_section(finding: dict[str, Any]) -> list[str]: if id(original) in seen_findings: continue seen_findings.add(id(original)) - originals.append((source_id, original)) + yield source_id, original provenance = original.get("provenance") if not isinstance(provenance, dict): continue @@ -547,15 +545,22 @@ def _remediation_section(finding: dict[str, Any]) -> list[str]: sources = provenance.get("sourceFindings") if isinstance(sources, list): pending.extend( - (_text(source.get("id"), "finding"), source["finding"]) + (source.get("id"), source["finding"]) for source in reversed(sources) if isinstance(source, dict) and isinstance(source.get("finding"), dict) ) + + +def _remediation_section(finding: dict[str, Any]) -> list[str]: + remediation = _text(finding.get("remediation"), "No canonical remediation was recorded.") + lines = ["", "#### Remediation", "", remediation] + seen = {remediation} + originals = list(retained_findings(finding)) for source_id, original in originals[1:]: text = _text(original.get("remediation"), "") if text and text not in seen: seen.add(text) - lines.extend(["", f"Source {source_id}: {text}"]) + lines.extend(["", f"Source {_text(source_id, 'finding')}: {text}"]) for field, label in ( ("remediationTests", "Tests"), ("preventiveControls", "Preventive controls"), @@ -570,6 +575,25 @@ def _remediation_section(finding: dict[str, Any]) -> list[str]: return lines +def _finding_header(number: int, finding: dict[str, Any]) -> list[str]: + cwes = ", ".join(finding["taxonomy"]["cwe"]) or "none" + title = _text(finding["title"], "Untitled finding") + return [ + f'', + "", + f"### [{number}] {title}", + "", + "| Field | Value |", + "| --- | --- |", + f"| Severity | {_cell(finding['severity']['level'])} |", + f"| Confidence | {_cell(finding['confidence']['level'])} |", + f"| Confidence rationale | {_cell(finding['confidence']['rationale'])} |", + f"| Category | {_cell(finding['taxonomy']['category'])} |", + f"| CWE | {_cell(cwes)} |", + f"| Affected lines | {_cell(_locations(finding))} |", + ] + + def _finding_section(number: int, finding: dict[str, Any]) -> list[str]: validation = finding.get("validation") if isinstance(finding.get("validation"), dict) else {} _, raw_root_cause = merged_root_cause(finding) @@ -632,10 +656,6 @@ def _finding_section(number: int, finding: dict[str, Any]) -> list[str]: if validation_outcomes else f"{finding['confidence']['rationale']} Validation details were not recorded separately.", ) - validation_evidence = _strings(validation.get("evidence")) - validation_assertions = _strings(validation.get("assertions")) - validation_counterevidence = _strings(validation.get("counterEvidence")) - validation_limitations = _strings(validation.get("limitations")) root_cause_summary = _text( raw_root_cause if isinstance(raw_root_cause, str) else root_cause.get("summary"), "", @@ -664,21 +684,8 @@ def _finding_section(number: int, finding: dict[str, Any]) -> list[str]: "Additional runtime or deployment evidence could raise or lower this severity.", ) attack_steps = _strings(attack_path.get("steps")) - cwes = ", ".join(finding["taxonomy"]["cwe"]) or "none" - title = _text(finding["title"], "Untitled finding") lines = [ - f'', - "", - f"### [{number}] {title}", - "", - "| Field | Value |", - "| --- | --- |", - f"| Severity | {_cell(severity['level'])} |", - f"| Confidence | {_cell(finding['confidence']['level'])} |", - f"| Confidence rationale | {_cell(finding['confidence']['rationale'])} |", - f"| Category | {_cell(finding['taxonomy']['category'])} |", - f"| CWE | {_cell(cwes)} |", - f"| Affected lines | {_cell(_locations(finding))} |", + *_finding_header(number, finding), "", "#### Summary", "", @@ -695,20 +702,15 @@ def _finding_section(number: int, finding: dict[str, Any]) -> list[str]: if validation_outcomes: lines.extend(["", *(f"- **{label}:** {value}" for label, value in validation_outcomes)]) lines.extend(_code_evidence_lines(validation_code_evidence)) - if validation_assertions: - lines.extend(["", "Assertions:", *_bullets(validation_assertions, "None recorded.")]) - if validation_evidence: - lines.extend(["", "Evidence:", *_bullets(validation_evidence, "No evidence recorded.")]) - if validation_counterevidence: - lines.extend( - [ - "", - "Counterevidence and remaining uncertainty:", - *_bullets(validation_counterevidence, "None recorded."), - ] - ) - if validation_limitations: - lines.extend(["", "Limitations:", *_bullets(validation_limitations, "None recorded.")]) + for label, key in ( + ("Assertions", "assertions"), + ("Evidence", "evidence"), + ("Counterevidence and remaining uncertainty", "counterEvidence"), + ("Limitations", "limitations"), + ): + values = _strings(validation.get(key)) + if values: + lines.extend(["", f"{label}:", *_bullets(values, "None recorded.")]) lines.extend(["", "#### Dataflow", "", dataflow_summary]) if attack_steps: lines.extend(["", "Attack steps:", *_bullets(attack_steps, "None recorded.")]) @@ -786,23 +788,8 @@ def _finding_section(number: int, finding: dict[str, Any]) -> list[str]: def _linked_finding_section(number: int, finding: dict[str, Any], report_path: str) -> list[str]: - cwes = ", ".join(finding["taxonomy"]["cwe"]) or "none" - title = _text(finding["title"], "Untitled finding") link = f"[detailed technical write-up]({report_path})" - lines = [ - f'', - "", - f"### [{number}] {title}", - "", - "| Field | Value |", - "| --- | --- |", - f"| Severity | {_cell(finding['severity']['level'])} |", - f"| Confidence | {_cell(finding['confidence']['level'])} |", - f"| Confidence rationale | {_cell(finding['confidence']['rationale'])} |", - f"| Category | {_cell(finding['taxonomy']['category'])} |", - f"| CWE | {_cell(cwes)} |", - f"| Affected lines | {_cell(_locations(finding))} |", - ] + lines = _finding_header(number, finding) for heading in ("Summary", "Validation", "Dataflow", "Reachability", "Severity"): lines.extend(["", f"#### {heading}", "", f"See the {link}."]) if any( diff --git a/plugins/codex-security/scripts/workbench/handoff.py b/plugins/codex-security/scripts/workbench/handoff.py index a0e5ee7d1..d5f63408c 100644 --- a/plugins/codex-security/scripts/workbench/handoff.py +++ b/plugins/codex-security/scripts/workbench/handoff.py @@ -12,6 +12,15 @@ RECOVERY_HANDOFF_TOKEN_PREFIX = "recovery_" +def durable_owner_thread_id(scan: sqlite3.Row, workspace: sqlite3.Row) -> str | None: + """Keep the native owner when a separate execution session has been recorded.""" + return ( + scan["deep_scan_owner_thread_id"] + or scan["continuation_thread_id"] + or workspace["thread_id"] + ) + + def require_handoff_claim_token(value: str) -> str: recovery_token = value.startswith(RECOVERY_HANDOFF_TOKEN_PREFIX) token = value.removeprefix(RECOVERY_HANDOFF_TOKEN_PREFIX) if recovery_token else value @@ -196,9 +205,7 @@ def mark_handoff_delivered( if thread_id is not None: workspace = require_workspace(connection, scan["workspace_id"]) validate_handoff_delivery_thread( - scan["deep_scan_owner_thread_id"] - or scan["continuation_thread_id"] - or workspace["thread_id"], + durable_owner_thread_id(scan, workspace), thread_id, claim_token, ) diff --git a/plugins/codex-security/scripts/workbench_cli.py b/plugins/codex-security/scripts/workbench_cli.py index 127548139..7584a9d09 100644 --- a/plugins/codex-security/scripts/workbench_cli.py +++ b/plugins/codex-security/scripts/workbench_cli.py @@ -191,7 +191,6 @@ def parse_args(description: str) -> argparse.Namespace: get_cli_scan_resume = subparsers.add_parser("get-cli-scan-resume") get_cli_scan_resume.add_argument("--scan-id", required=True) get_cli_scan_resume.add_argument("--claim-token") - get_cli_scan_resume.add_argument("--migrate", action="store_true") get_cli_scan_resume.add_argument("--allow-unavailable", action="store_true") compare_scans = subparsers.add_parser("compare-scans") diff --git a/plugins/codex-security/scripts/workbench_composition.py b/plugins/codex-security/scripts/workbench_composition.py index a1670f733..bd6731f6e 100644 --- a/plugins/codex-security/scripts/workbench_composition.py +++ b/plugins/codex-security/scripts/workbench_composition.py @@ -84,12 +84,6 @@ def read_composition_checkpoint(scan: sqlite3.Row) -> CompositionCheckpoint | No return cast(CompositionCheckpoint, checkpoint) -def encode_composition_checkpoint(checkpoint: CompositionCheckpoint) -> bytes: - return json.dumps( - checkpoint, ensure_ascii=True, allow_nan=False, sort_keys=True, separators=(",", ":") - ).encode() - - def composition_children(connection: sqlite3.Connection, scan: sqlite3.Row) -> list[sqlite3.Row]: return connection.execute( "SELECT * FROM scans WHERE parent_scan_id = ? AND parent_scan_role = 'deep_pass' " @@ -98,13 +92,6 @@ def composition_children(connection: sqlite3.Connection, scan: sqlite3.Row) -> l ).fetchall() -def composition_child_ids(connection: sqlite3.Connection) -> set[str]: - return { - row["id"] - for row in connection.execute("SELECT id FROM scans WHERE parent_scan_role = 'deep_pass'") - } - - def composition_execution_threads(scan: sqlite3.Row) -> tuple[str, ...]: scan_dir = Path(scan["scan_dir"]) try: diff --git a/plugins/codex-security/scripts/workbench_db.py b/plugins/codex-security/scripts/workbench_db.py index f110a60fd..929e57b58 100644 --- a/plugins/codex-security/scripts/workbench_db.py +++ b/plugins/codex-security/scripts/workbench_db.py @@ -46,6 +46,7 @@ write_scan_local_bytes, ) from finding_preview import bounded_finding_details +from project_scan_artifacts import merge_coverage from workbench import handoff from workbench.storage import ( create_private_directory, @@ -1331,21 +1332,13 @@ def add_warning() -> None: ) coverage = read_json_object(scan_dir / ARTIFACTS["coverage"]) coverage["completeness"] = "partial" - for field in ("surfaces", "explicitExclusions", "deferred", "openQuestions"): - rows = coverage.setdefault(field, []) - for row in recovered["coverage"].get(field, []): - if row not in rows: - rows.append(row) + merge_coverage(coverage, recovered["coverage"]) documents = current_manifest, {"findings": recovered["findings"]}, coverage elif scan["mode"] != "deep": documents = saved_results.merge_saved_results( scan_dir, scan["id"], completion_binding, - connection.execute( - "SELECT * FROM deep_scan_workers WHERE scan_id = ? ORDER BY created_at, id", - (scan["id"],), - ).fetchall(), warnings, stopped=False, reason="", @@ -1516,6 +1509,15 @@ def register_cli_scan(connection: sqlite3.Connection, args: argparse.Namespace) raise SystemExit( "Cannot resume: the original checkout revision or contents changed." ) + sealed_version = scan_history.sealed_scan_producer_version( + scan, + scan_dir, + artifact_path=artifact_path, + read_json_object=read_json_object, + workbench_completion_binding=workbench_completion_binding, + ) + if sealed_version is None: + scan_history.require_current_deep_runtime(connection, scan) saved_recipe = json.loads(scan["recipe_json"]) if scan["recipe_json"] else None if saved_recipe is not None and saved_recipe["target"] != recipe["target"]: raise SystemExit("Saved scan registration must preserve the original scope.") @@ -1524,13 +1526,16 @@ def register_cli_scan(connection: sqlite3.Connection, args: argparse.Namespace) if recipe["target"]["paths"] != expected_paths: raise SystemExit("Saved scan registration must preserve the original scope.") connection.execute( - "UPDATE scans SET recipe_json = ?, continuation_thread_id = NULL, " + "UPDATE scans SET recipe_json = ?, continuation_thread_id = ?, " "updated_at = ? WHERE id = ?", - (json.dumps(recipe, allow_nan=False), now(), scan_id), + ( + json.dumps(recipe, allow_nan=False), + scan["continuation_thread_id"] if sealed_version is not None else None, + now(), + scan_id, + ), ) scan = require_scan(connection, scan_id) - with scan_completion_lock(scan_id): - scan = saved_results.migrate_legacy_scan(_WORKBENCH_DB_CONTEXT, connection, scan) return scan_history.scan_registration(connection, scan, scan_contract) if next(scan_dir.iterdir(), None) is not None: raise SystemExit("The scan artifact directory must be empty before the scan starts.") @@ -2762,7 +2767,7 @@ def scan_result( "remediationUnavailableReason": remediation_unavailable_reason, "reportAvailable": "markdownReport" in artifacts, "resultsRecoveryNeeded": saved_results.scan_results_recovery_needed( - _WORKBENCH_DB_CONTEXT, connection, scan, composition + _WORKBENCH_DB_CONTEXT, connection, scan ), "scanDir": scan["scan_dir"], "scanId": scan["id"], @@ -3383,12 +3388,6 @@ def main() -> None: workbench_completion_binding=workbench_completion_binding, claim_token=args.claim_token, ) - if args.migrate and "sealedProducerVersion" not in result: - with scan_completion_lock(scan["id"]): - scan = saved_results.migrate_legacy_scan( - _WORKBENCH_DB_CONTEXT, connection, scan - ) - result = scan_history.scan_registration(connection, scan, scan_contract) except SystemExit as exc: if not args.allow_unavailable: raise diff --git a/plugins/codex-security/scripts/workbench_progress.py b/plugins/codex-security/scripts/workbench_progress.py index 80c3e6701..2982a7343 100644 --- a/plugins/codex-security/scripts/workbench_progress.py +++ b/plugins/codex-security/scripts/workbench_progress.py @@ -8,7 +8,7 @@ from typing import Any, Callable sys.path.insert(0, str(Path(__file__).resolve().parent)) -from workbench.handoff import require_current_continuation +from workbench.handoff import durable_owner_thread_id, require_current_continuation from workbench_constants import PHASES from workbench_validation import optional_text, require_uuid, user_context_argument @@ -96,11 +96,7 @@ def update_context( raise SystemExit("This scan does not belong to the selected workspace.") else: thread_id = optional_text(args.thread_id, maximum=512) - owning_thread_id = ( - scan["deep_scan_owner_thread_id"] - or scan["continuation_thread_id"] - or workspace["thread_id"] - ) + owning_thread_id = durable_owner_thread_id(scan, workspace) if thread_id is None or thread_id != owning_thread_id: raise SystemExit("This scan does not belong to the current Codex thread.") require_current_continuation( diff --git a/plugins/codex-security/scripts/workbench_saved_results.py b/plugins/codex-security/scripts/workbench_saved_results.py index 0c03a0fa1..28bb87d13 100644 --- a/plugins/codex-security/scripts/workbench_saved_results.py +++ b/plugins/codex-security/scripts/workbench_saved_results.py @@ -14,7 +14,6 @@ import sys from collections.abc import Callable, Iterator from dataclasses import dataclass -from datetime import timedelta from pathlib import Path from types import ModuleType from typing import Any @@ -37,17 +36,15 @@ open_scan_local_file_descriptor, write_scan_local_bytes, ) -from project_scan_artifacts import project_scan_artifacts +from project_scan_artifacts import merge_coverage, project_scan_artifacts +from report_projection import retained_findings from workbench_composition import ( COMPOSITION_CHECKPOINT, - CompositionCheckpoint, - CompositionView, composition_children, - encode_composition_checkpoint, read_composition_checkpoint, ) from workbench_constants import PHASES -from workbench_scan_usage import _timestamp, stored_scan_cost_fields +from workbench_scan_usage import merge_scan_cost from workbench_target import committed_diff_snapshot_digest from workbench_validation import path_within_scope @@ -117,71 +114,17 @@ def _children(scan_dir: Path, relative: str) -> list[str]: return sorted(child.name for child in cursor.iterdir()) -def _latest_successful_reducer(workers: list[Any]) -> Any | None: - return max( - ( - worker - for worker in workers - if worker["kind"] == "dedup" - and worker["status"] == "succeeded" - and worker["result_manifest_path"] - ), - key=lambda worker: (worker["completed_at"] or "", worker["id"]), - default=None, - ) - +def _saved_result_paths(scan_dir: Path) -> Iterator[str]: + for name in _children(scan_dir, "checkpoints"): + if re.fullmatch(r"[0-9a-f]{64}\.json", name): + yield f"checkpoints/{name}" -def _saved_result_paths(scan_dir: Path, workers: list[Any]) -> Iterator[tuple[str, str | None]]: - latest_reducer = _latest_successful_reducer(workers) - def checkpoints(directory: str, kind: str | None = None) -> Iterator[tuple[str, str | None]]: - for name in _children(scan_dir, directory): - if re.fullmatch(r"[0-9a-f]{64}\.json", name): - yield f"{directory}/{name}", kind - - yield from checkpoints("checkpoints") - for worker in workers: - if worker["kind"] not in {"dedup", "discovery"}: - continue - try: - output = Path(worker["artifact_dir"]).relative_to(scan_dir).as_posix() - except (TypeError, ValueError): - continue - attempts = (Path(output).parent if Path(output).name == "output" else Path(output)) / ( - "attempts" - ) - directories = [output] + [ - (attempts / name).as_posix() - for name in _children(scan_dir, attempts.as_posix()) - if re.fullmatch(r"attempt-\d+", name) - ] - for directory in directories: - checkpoint_paths = list(checkpoints(f"{directory}/checkpoints", worker["kind"])) - if worker["kind"] == "discovery" or checkpoint_paths: - yield f"{directory}/result.json", worker["kind"] - yield from checkpoint_paths - if worker["result_manifest_path"] and ( - worker["kind"] == "discovery" - or (latest_reducer is not None and worker["id"] == latest_reducer["id"]) - ): - try: - yield ( - Path(worker["result_manifest_path"]).relative_to(scan_dir).as_posix(), - worker["kind"], - ) - except ValueError: - continue - - -def _read_saved_result( - scan_dir: Path, relative: str, scan_id: str, *, kind: str | None = None -) -> tuple[dict[str, Any], str]: +def _read_saved_result(scan_dir: Path, relative: str, scan_id: str) -> tuple[dict[str, Any], str]: draft = _read_scan_local_json(scan_dir, relative, "Saved scan checkpoint") if draft.get("scanId") != scan_id: raise ContractError("checkpoint belongs to a different scan") - if not isinstance(draft.get("findings"), list) or not isinstance( - draft.get("coverage", {} if kind == "dedup" else None), dict - ): + if not isinstance(draft.get("findings"), list) or not isinstance(draft.get("coverage"), dict): raise ContractError("checkpoint has no semantic findings or coverage") return draft, _digest(draft) @@ -224,27 +167,17 @@ def _source_digests(value: Any, label: str) -> dict[str, str]: return value -def _saved_results_changed( - db: Any, connection: Any, scan: Any, composition: CompositionView | None = None -) -> bool: +def _saved_results_changed(db: Any, scan: Any) -> bool: try: scan_dir = db.require_canonical_scan_directory(Path(scan["scan_dir"])) manifest_path = db.artifact_path(scan_dir, db.ARTIFACTS["manifest"], required=False) - workers = connection.execute( - "SELECT id, kind, status, completed_at, artifact_dir, result_manifest_path " - "FROM deep_scan_workers WHERE scan_id = ?", - (scan["id"],), - ).fetchall() - checkpoint = ( - composition.checkpoint if composition is not None else read_composition_checkpoint(scan) - ) - paths = dict(_saved_result_paths(scan_dir, workers if checkpoint is None else [])) + paths = list(_saved_result_paths(scan_dir)) frozen_sources = scan["retained_source_digests_json"] def has_saved_source() -> bool: for path in paths: try: - _read_saved_result(scan_dir, path, scan["id"], kind=paths[path]) + _read_saved_result(scan_dir, path, scan["id"]) return True except (ContractError, OSError, ValueError): continue @@ -275,9 +208,7 @@ def has_saved_source() -> bool: current_sources = dict(published_sources) for path in paths: try: - _, current_sources[path] = _read_saved_result( - scan_dir, path, scan["id"], kind=paths[path] - ) + _, current_sources[path] = _read_saved_result(scan_dir, path, scan["id"]) except (ContractError, OSError, ValueError): continue return current_sources != published_sources @@ -326,36 +257,25 @@ def _recovery_source_digests(db: Any, connection: Any, scan: Any) -> tuple[dict[ if frozen_sources is None: save_composed_checkpoint(db, connection, scan, scan_dir) - workers = connection.execute( - "SELECT id, kind, status, completed_at, artifact_dir, result_manifest_path " - "FROM deep_scan_workers WHERE scan_id = ?", - (scan["id"],), - ).fetchall() - paths = dict( - _saved_result_paths(scan_dir, workers if read_composition_checkpoint(scan) is None else []) - ) + paths = set(_saved_result_paths(scan_dir)) recovery_sources = dict(frozen_sources or {}) for relative, expected_digest in recovery_sources.items(): try: - _, digest = _read_saved_result(scan_dir, relative, scan["id"], kind=paths.get(relative)) + _, digest = _read_saved_result(scan_dir, relative, scan["id"]) except (ContractError, OSError, ValueError) as exc: raise ContractError("Frozen stopped-scan checkpoint set is incomplete.") from exc if digest != expected_digest: raise ContractError("checkpoint changed after the scan stopped") - for relative in paths.keys() - recovery_sources.keys(): + for relative in paths - recovery_sources.keys(): try: - _, recovery_sources[relative] = _read_saved_result( - scan_dir, relative, scan["id"], kind=paths[relative] - ) + _, recovery_sources[relative] = _read_saved_result(scan_dir, relative, scan["id"]) except (ContractError, OSError, ValueError): continue return recovery_sources, include_parent -def scan_results_recovery_needed( - db: Any, connection: Any, scan: Any, composition: CompositionView | None = None -) -> bool: +def scan_results_recovery_needed(db: Any, connection: Any, scan: Any) -> bool: if scan["status"] != "failed" or scan["canceled_at"] is not None: return False warnings = json.loads(scan["completion_warnings_json"]) @@ -370,12 +290,12 @@ def scan_results_recovery_needed( ).fetchone() if publication_error is not None and publication_error["publication_error_message"]: return True - return _saved_results_changed(db, connection, scan, composition) + return _saved_results_changed(db, scan) def _finding_key(finding: dict[str, Any]) -> str: # Wording and evidence may improve between checkpoints; distinct source locations - # must not collide merely because two workers chose the same semantic identity. + # must not collide merely because checkpoints reused the same semantic identity. provenance = finding.get("provenance") identity = ( provenance.get("preservedIdentity", finding.get("identity")) @@ -415,25 +335,6 @@ def _finding_key(finding: dict[str, Any]) -> str: ) -def _worker_candidate_key( - worker_id: str, candidate_id: str, finding: dict[str, Any] -) -> tuple[str, str, Any, Any, Any]: - """Identify one worker-local candidate without merging unrelated locations.""" - provenance = finding.get("provenance") - identity = ( - provenance.get("preservedIdentity", finding.get("identity")) - if isinstance(provenance, dict) - else finding.get("identity") - ) - if not isinstance(identity, dict): - normalized = dict(finding) - _ensure_finding_identity(normalized) - identity = normalized.get("identity") - anchor = identity.get("anchor") if isinstance(identity, dict) else None - instance = identity.get("instance") if isinstance(identity, dict) else None - return worker_id, candidate_id, finding.get("ruleId"), anchor, instance - - def _finding_content(finding: dict[str, Any]) -> dict[str, Any]: """Return substantive finding content without generated identity or provenance.""" return { @@ -458,37 +359,10 @@ def _ensure_finding_identity(finding: Any, *, candidate_only: bool = False) -> N finding["identity"] = {"anchor": anchor} -def _retained_findings(finding: dict[str, Any]) -> Iterator[dict[str, Any]]: - """Yield canonical and historical findings without trusting candidate IDs.""" - pending = [finding] - seen: set[int] = set() - while pending: - current = pending.pop() - marker = id(current) - if marker in seen: - continue - seen.add(marker) - yield current - provenance = current.get("provenance") - if not isinstance(provenance, dict): - continue - previous = provenance.get("previousFindings") - if isinstance(previous, list): - pending.extend(item for item in reversed(previous) if isinstance(item, dict)) - sources = provenance.get("sourceFindings") - if isinstance(sources, list): - pending.extend( - source["finding"] - for source in reversed(sources) - if isinstance(source, dict) and isinstance(source.get("finding"), dict) - ) - - def merge_saved_results( scan_dir: Path, scan_id: str, binding: dict[str, Any], - workers: list[Any], warnings: list[str], *, stopped: bool, @@ -496,7 +370,7 @@ def merge_saved_results( frozen_source_digests: dict[str, str] | None = None, allow_frozen_legacy_parent: bool = False, ) -> tuple[dict[str, Any], dict[str, Any], dict[str, Any]] | None: - """Read only bound parent/worker files; return an unsealed loss-preserving union.""" + """Read bound parent drafts and checkpoints; return an unsealed loss-preserving union.""" initial_warnings = set(warnings) parent: dict[str, Any] | None = None parent_manifest: dict[str, Any] | None = None @@ -523,7 +397,7 @@ def merge_saved_results( parent_checkpoint: parent_digest, } - sources: list[tuple[str, dict[str, Any], str | None]] = [] + sources: list[tuple[str, dict[str, Any]]] = [] parent_preserved_sources: dict[str, str] = {} source_digests: dict[str, str] = {} if parent_manifest: @@ -531,100 +405,17 @@ def merge_saved_results( if isinstance(recorded, dict): parent_preserved_sources = recorded source_digests.update(parent_preserved_sources) - paths: dict[str, str | None] = {} - reducer_paths: set[str] = set() - current_results: set[str] = set() - reducer_outputs: list[tuple[Any, str, list[str], int]] = [] - reducer = _latest_successful_reducer(workers) - latest_reducer: str | None = None - if reducer is not None: - try: - latest_reducer = Path(reducer["result_manifest_path"]).relative_to(scan_dir).as_posix() - paths[latest_reducer] = None - reducer_paths.add(latest_reducer) - except ValueError: - warnings.append("Skipped a reducer result outside the scan directory.") - - def checkpoints(directory: str, worker_id: str | None) -> None: - for name in _children(scan_dir, directory): - if re.fullmatch(r"[0-9a-f]{64}\.json", name): - paths[f"{directory}/{name}"] = worker_id - - checkpoints("checkpoints", None) - for worker in workers: - try: - output = Path(worker["artifact_dir"]).relative_to(scan_dir).as_posix() - except (TypeError, ValueError): - warnings.append("Skipped a worker checkpoint outside the scan directory.") - continue - if worker["kind"] == "dedup": - - def reducer_output(directory: str, attempt: int, reducer_worker: Any) -> None: - result_path = f"{directory}/result.json" - checkpoint_paths = [ - f"{directory}/checkpoints/{name}" - for name in _children(scan_dir, f"{directory}/checkpoints") - if re.fullmatch(r"[0-9a-f]{64}\.json", name) - ] - if not checkpoint_paths: - return - paths[result_path] = None - for checkpoint_path in checkpoint_paths: - paths[checkpoint_path] = None - reducer_paths.update([result_path, *checkpoint_paths]) - reducer_outputs.append((reducer_worker, result_path, checkpoint_paths, attempt)) - - reducer_output(output, int(worker["attempt"] or 0), worker) - attempts = ( - Path(output).parent if Path(output).name == "output" else Path(output) - ) / "attempts" - for name in _children(scan_dir, attempts.as_posix()): - match = re.fullmatch(r"attempt-(\d+)", name) - if match: - reducer_output((attempts / name).as_posix(), int(match.group(1)), worker) - continue - if worker["kind"] != "discovery": - continue - paths[f"{output}/result.json"] = worker["id"] - current_results.add(f"{output}/result.json") - checkpoints(f"{output}/checkpoints", worker["id"]) - attempts = ( - Path(output).parent if Path(output).name == "output" else Path(output) - ) / "attempts" - for name in _children(scan_dir, attempts.as_posix()): - if re.fullmatch(r"attempt-\d+", name): - archived = (attempts / name).as_posix() - paths[f"{archived}/result.json"] = worker["id"] - checkpoints(f"{archived}/checkpoints", worker["id"]) - if worker["result_manifest_path"]: - try: - current_path = Path(worker["result_manifest_path"]).relative_to(scan_dir).as_posix() - paths[current_path] = worker["id"] - current_results.add(current_path) - except ValueError: - warnings.append("Skipped a worker result outside the scan directory.") - + paths = list(_saved_result_paths(scan_dir)) if frozen_source_digests is not None: - paths = { - relative: worker_id - for relative, worker_id in paths.items() - if relative in frozen_source_digests - } - current_results.intersection_update(frozen_source_digests) - if latest_reducer not in frozen_source_digests: - latest_reducer = None + paths = [relative for relative in paths if relative in frozen_source_digests] - for relative, worker_id in paths.items(): + for relative in paths: try: - draft, digest = _read_saved_result( - scan_dir, relative, scan_id, kind="dedup" if relative in reducer_paths else None - ) + draft, digest = _read_saved_result(scan_dir, relative, scan_id) if frozen_source_digests is not None and frozen_source_digests[relative] != digest: raise ContractError("checkpoint changed after the scan stopped") source_digests[relative] = digest - # Recovery expects coverage, but reducer results only contain findings - # and context. Add an empty value after hashing the original result. - sources.append((relative, {"coverage": {}, **draft}, worker_id)) + sources.append((relative, draft)) except (ContractError, OSError, ValueError) as exc: if (scan_dir / relative).exists(): warnings.append(f"Preserved unreadable checkpoint {relative}: {exc}") @@ -632,29 +423,6 @@ def reducer_output(directory: str, attempt: int, reducer_worker: Any) -> None: if frozen_source_digests.keys() - source_digests.keys(): raise ContractError("Frozen stopped-scan checkpoint set is incomplete.") - drafts_by_path = {relative: draft for relative, draft, _ in sources} - latest_reducer_key = ( - (reducer["completed_at"] or "", reducer["id"], int(reducer["attempt"] or 0)) - if reducer is not None and latest_reducer in drafts_by_path - else None - ) - if latest_reducer_key is None: - latest_reducer = None - for worker, result_path, checkpoint_paths, attempt in reducer_outputs: - result = drafts_by_path.get(result_path) - if result is None or not any( - drafts_by_path.get(checkpoint_path) == result for checkpoint_path in checkpoint_paths - ): - continue - current_results.add(result_path) - candidate_key = (worker["completed_at"] or "", worker["id"], attempt) - if latest_reducer_key is None or candidate_key > latest_reducer_key: - latest_reducer_key = candidate_key - latest_reducer = result_path - - if parent is None and latest_reducer is not None: - parent = next((draft for relative, draft, _ in sources if relative == latest_reducer), None) - if parent is None and not sources: if not stopped or frozen_source_digests is not None: return None @@ -734,10 +502,7 @@ def reducer_output(directory: str, attempt: int, reducer_worker: Any) -> None: findings: list[dict[str, Any]] = [] finding_positions: dict[str, int] = {} represented: dict[str, str | None] = {} - represented_candidates: dict[tuple[str, str, Any, Any, Any], str | None] = {} represented_history: dict[str, set[str]] = {} - represented_candidate_history: dict[tuple[str, str, Any, Any, Any], set[str]] = {} - rejected_history: dict[tuple[str, str], list[dict[str, Any]]] = {} stopped_parent_seal = bool( stopped and parent_manifest and parent_manifest["scan"].get("sealedAt") ) @@ -757,19 +522,16 @@ def valid_finding(value: Any) -> bool: ) return bool(document["findings"]) - all_sources = ([("parent", parent, None)] if parent else []) + sources - current_drafts = ([(None, parent)] if parent else []) + [ - (worker_id, draft) for relative, draft, worker_id in sources if relative in current_results - ] - resolved: dict[tuple[str | None, str], str] = {} - for owner, draft in current_drafts: + all_sources = ([("parent", parent)] if parent else []) + sources + resolved: dict[str, str] = {} + for draft in [parent] if parent else []: for finding in draft["findings"]: if ( isinstance(finding, dict) and valid_finding(finding) and (candidate_id := finding_candidate_id(finding)) ): - resolved.setdefault((owner, candidate_id), "reported") + resolved.setdefault(candidate_id, "reported") for field in ("surfaces", "explicitExclusions"): items = draft["coverage"].get(field, []) for item in items if isinstance(items, list) else []: @@ -778,14 +540,13 @@ def valid_finding(value: Any) -> bool: and isinstance(item.get("candidateId"), str) and item.get("disposition") in {"reported", "rejected", "not_applicable"} ): - resolved.setdefault((owner, item["candidateId"]), item["disposition"]) - # Only the current parent may claim that another worker finding was absorbed. - # A superseded checkpoint must not suppress a newer independent result. + resolved.setdefault(item["candidateId"], item["disposition"]) + # Only the current parent can supersede findings from earlier checkpoints. for draft in [parent] if parent else []: for finding in draft["findings"]: if valid_finding(finding): canonical_key = _finding_key(finding) - for retained in _retained_findings(finding): + for _, retained in retained_findings(finding): retained_key = _finding_key(retained) if retained is not finding: represented_history.setdefault(retained_key, set()).add( @@ -797,44 +558,12 @@ def valid_finding(value: Any) -> bool: elif previous_key != canonical_key: # Ambiguous history cannot suppress an independent source. represented[retained_key] = None - originals = finding["provenance"].get("sourceFindings", []) - for original in originals if isinstance(originals, list) else []: - if isinstance(original, dict) and isinstance(original.get("finding"), dict): - source_id = original.get("id") - candidate_id = finding_candidate_id(original["finding"]) - if isinstance(source_id, str) and ":" in source_id and candidate_id: - candidate_key = _worker_candidate_key( - source_id.rsplit(":", 1)[0], - candidate_id, - original["finding"], - ) - previous_key = represented_candidates.get(candidate_key) - if candidate_key not in represented_candidates: - represented_candidates[candidate_key] = canonical_key - elif previous_key != canonical_key: - # Candidate ids are only authoritative within one - # logical worker. Multiple canonical owners make - # that worker-local identity ambiguous. - represented_candidates[candidate_key] = None - represented_candidate_history.setdefault(candidate_key, set()).add( - _digest(_finding_content(original["finding"])) - ) - resolved.setdefault(candidate_key, "reported") - for relative, draft, worker_id in all_sources: + for relative, draft in all_sources: superseded = ( - worker_id is None - and parent is not None + parent is not None and parent.get("complete") is not False and relative != "parent" and (not stopped_parent_seal or relative in parent_preserved_sources) - ) or ( - relative not in current_results - and any( - saved_worker == worker_id - and saved_path in current_results - and current.get("complete") is not False - for saved_path, current, saved_worker in sources - ) ) if ( (relative != "parent" or not parent_manifest) @@ -858,17 +587,6 @@ def valid_finding(value: Any) -> bool: if relative == "parent" and parent_manifest: finding = copy.deepcopy(value) _ensure_finding_identity(finding, candidate_only=True) - provenance = finding.get("provenance") if isinstance(finding, dict) else None - owner = provenance.get("workerId") if isinstance(provenance, dict) else None - candidate_id = finding_candidate_id(finding) if isinstance(finding, dict) else None - if ( - stopped_parent_seal - and isinstance(owner, str) - and candidate_id - and resolved.get((owner, candidate_id)) in {"rejected", "not_applicable"} - ): - rejected_history.setdefault((owner, candidate_id), []).append(finding) - continue if valid_finding(finding): finding_positions.setdefault(_finding_key(finding), len(findings)) findings.append(finding) @@ -881,7 +599,7 @@ def valid_finding(value: Any) -> bool: source_value = copy.deepcopy(value) finding = copy.deepcopy(value) candidate_id = finding_candidate_id(finding) - if relative != "parent" and resolved.get((worker_id, candidate_id)) in { + if relative != "parent" and resolved.get(candidate_id) in { "rejected", "not_applicable", }: @@ -918,8 +636,6 @@ def valid_finding(value: Any) -> bool: if not isinstance(provenance, dict): findings.append(finding) continue - if worker_id: - provenance.setdefault("workerId", worker_id) _ensure_finding_identity(finding) if not valid_finding(finding): findings.append(finding) @@ -930,12 +646,6 @@ def valid_finding(value: Any) -> bool: if key in represented: mapped_key = represented[key] historical_contents = represented_history.get(key, set()) - elif worker_id and candidate_id: - candidate_key = _worker_candidate_key(worker_id, candidate_id, finding) - if candidate_key not in represented_candidates: - represented_candidates[candidate_key] = key - mapped_key = represented_candidates[candidate_key] - historical_contents = represented_candidate_history.get(candidate_key, set()) else: mapped_key = None historical_contents = set() @@ -981,7 +691,7 @@ def valid_finding(value: Any) -> bool: already_retained = any( source_key == _finding_key(historical) and source_content == _finding_content(historical) - for historical in _retained_findings(retained) + for _, historical in retained_findings(retained) ) if ( not already_retained @@ -1006,28 +716,9 @@ def valid_finding(value: Any) -> bool: for item in items: if field == "openQuestions" and isinstance(item, str): item = {"question": item.strip()} - if ( - field == "surfaces" - and isinstance(item, dict) - and item.get("disposition") in {"rejected", "not_applicable"} - and isinstance(item.get("candidateId"), str) - and (history_findings := rejected_history.get((worker_id, item["candidateId"]))) - ): - item = copy.deepcopy(item) - if not isinstance(item.get("previousFindings"), list): - item["previousFindings"] = [] - history = item["previousFindings"] - for finding in history_findings: - if not any( - isinstance(previous, dict) - and _finding_key(previous) == _finding_key(finding) - and _finding_content(previous) == _finding_content(finding) - for previous in history - ): - history.append(copy.deepcopy(finding)) if ( isinstance(item, dict) - and (worker_id, item.get("candidateId")) in resolved + and item.get("candidateId") in resolved and (field == "deferred" or item.get("disposition") == "needs_follow_up") ): continue @@ -1131,206 +822,6 @@ def _restore_published_outputs(scan_dir: Path, snapshots: dict[str, bytes | None write_scan_local_bytes(scan_dir, relative, contents) -def retire_legacy_run(connection: Any, scan_id: str) -> None: - with connection: - connection.execute( - "UPDATE deep_scan_runs SET status = 'interrupted', phase = 'terminal', cancel_requested = 1, " - "coordinator_generation = coordinator_generation + 1 WHERE scan_id = ? AND status = 'running'", - (scan_id,), - ) - - -def migrate_legacy_scan(db: Any, connection: Any, scan: Any) -> Any: - """Import completed v1 results once; ordinary scans own all subsequent execution.""" - scan = db.require_scan(connection, scan["id"]) - if scan["mode"] != "deep": - return scan - checkpoint = read_composition_checkpoint(scan) - if checkpoint is not None: - if checkpoint.get("legacy") is not None: - retire_legacy_run(connection, scan["id"]) - return scan - run = connection.execute( - "SELECT * FROM deep_scan_runs WHERE scan_id = ?", (scan["id"],) - ).fetchone() - if run is None: - return scan - if ( - scan["status"] != "running" - or scan["canceled_at"] is not None - or run["status"] not in {"running", "succeeded"} - or run["cancel_requested"] - ): - raise SystemExit("This Deep Scan has stopped and cannot resume.") - scan_dir = db.require_canonical_scan_directory(Path(scan["scan_dir"])) - workers = connection.execute( - "SELECT * FROM deep_scan_workers WHERE scan_id = ? ORDER BY created_at, id", (scan["id"],) - ).fetchall() - if run["status"] == "running": - # These are the retired coordinator's existing leases, read only during conversion. - last_seen = _timestamp(run["updated_at"]) - grace = 120 if run["coordinator_generation"] == 1 else 30 - if run["coordinator_generation"] != 1: - relative = f"artifacts/deep_discovery/coordinator-heartbeat-{run['coordinator_generation']}.json" - if (scan_dir / relative).exists(): - heartbeat = _read_scan_local_json( - scan_dir, relative, "Previous coordinator heartbeat" - ) - if heartbeat.get("coordinatorGeneration") == run["coordinator_generation"]: - timestamp = _timestamp(heartbeat.get("updatedAt")) - if timestamp is not None: - last_seen = ( - max(last_seen, timestamp) if last_seen is not None else timestamp - ) - active = run["coordinator_generation"] != 1 or any( - worker["status"] in {"queued", "running"} for worker in workers - ) - if ( - active - and last_seen is not None - and last_seen > _timestamp(db.now()) - timedelta(seconds=grace) - ): - raise SystemExit( - "The previous Deep Scan owner is still active; stop it before resuming with this version." - ) - reducer = _latest_successful_reducer(workers) - accepted = [ - worker - for worker in workers - if worker == reducer or (worker["kind"] == "discovery" and worker["status"] == "succeeded") - ] - warnings: list[str] = [] - binding = db.workbench_completion_binding(scan, db.now()) - documents = merge_saved_results( - scan_dir, - scan["id"], - binding, - accepted, - warnings, - stopped=run["status"] != "succeeded", - reason="Saved Deep Scan execution migrated to ordinary scans.", - ) - aggregate = { - "scanId": scan["id"], - "findings": [], - "coverage": { - "completeness": "partial", - "surfaces": [], - "explicitExclusions": [], - "deferred": [], - }, - } - if documents is not None: - documents[2]["deferred"] = [ - item for item in documents[2]["deferred"] if item.get("id") != "scan-stopped" - ] - _, _, manifest, findings, coverage, _, _ = _prepare_scan_finalization( - scan_dir, - expected_coverage_mode=binding["coverageMode"], - completion_binding=binding, - draft_documents=documents, - ) - aggregate.update(findings=findings["findings"], coverage=coverage) - for field in ("scope", "threatModel"): - if field in manifest["scan"]: - aggregate[field] = manifest["scan"][field] - for index, finding in enumerate(aggregate["findings"]): - original = copy.deepcopy(finding) - for field in ("findingId", "occurrenceId", "fingerprints"): - finding.pop(field, None) - source_id = f"legacy:{index}" - finding.setdefault("provenance", {}).update( - sourceFindingIds=[source_id], - sourceFindings=[{"id": source_id, "finding": original}], - ) - coverage = aggregate["coverage"] - for field in ( - "documentType", - "schemaVersion", - "scanId", - "mode", - "includePaths", - "excludePaths", - "receiptRefs", - "inventoryStrategy", - ): - coverage.pop(field, None) - for field in ("includePaths", "excludePaths"): - aggregate.get("scope", {}).pop(field, None) - merge_failures = 0 - for worker in workers: - if worker["kind"] == "dedup": - if worker["status"] == "succeeded": - merge_failures = 0 - elif worker["status"] == "failed": - merge_failures += 1 - if worker["kind"] != "discovery" or worker in accepted: - continue - directory = Path(worker["artifact_dir"]).relative_to(scan_dir).as_posix() - coverage["deferred"].append( - { - "id": f"legacy-{worker['id']}", - "reason": f"Earlier independent scan did not merge. Saved work: {directory}.", - } - ) - coverage["deferred"].extend({"reason": warning} for warning in warnings) - # The retired coordinator refunded these abandoned attempts when its lease expired. - # Preserve that budget so migration can dispatch their replacements. - interrupted_discoveries = sum( - worker["kind"] == "discovery" - and ( - worker["status"] in {"queued", "running"} - or ( - worker["status"] == "canceled" - and ( - (worker["error_message"] or "").startswith("coordinator_shutdown:") - or (run["coordinator_generation"] == 1 and worker["error_message"] is None) - ) - ) - ) - for worker in workers - if run["status"] == "running" - ) - checkpoint: CompositionCheckpoint = { - "version": 2, - "startedAt": run["created_at"], - "passes": [], - "mergedScanIds": [], - "aggregate": aggregate, - "noNewStreak": run["consecutive_no_new"], - "consecutiveErrors": run["consecutive_errors"], - "mergeFailures": merge_failures, - "legacy": { - "originThreadId": scan["continuation_thread_id"] or scan["deep_scan_owner_thread_id"], - "discoveryRuns": run["discovery_runs_dispatched"] - interrupted_discoveries, - "coverage": copy.deepcopy(coverage), - **( - {"cost": cost} - if (cost := stored_scan_cost_fields(scan["cost_json"]).get("cost")) - else {} - ), - }, - **( - {"terminalReason": run["terminal_reason"]} - if run["status"] == "succeeded" and run["terminal_reason"] is not None - else {} - ), - } - # Clear the old execution thread before publishing the checkpoint. A crash between - # these writes safely repeats conversion and never resumes the retired conversation. - with connection: - connection.execute( - "UPDATE scans SET deep_scan_owner_thread_id = COALESCE(deep_scan_owner_thread_id, " - "continuation_thread_id), continuation_thread_id = NULL WHERE id = ?", - (scan["id"],), - ) - write_scan_local_bytes( - scan_dir, COMPOSITION_CHECKPOINT, encode_composition_checkpoint(checkpoint) - ) - retire_legacy_run(connection, scan["id"]) - return db.require_scan(connection, scan["id"]) - - def _stopped_child_draft(db: Any, child: Any, scan_dir: Path) -> dict[str, Any] | None: """Read one ordinary saved scan through its normal validation and recovery path.""" child_dir = db.require_canonical_scan_directory(Path(child["scan_dir"])) @@ -1348,7 +839,6 @@ def _stopped_child_draft(db: Any, child: Any, scan_dir: Path) -> dict[str, Any] child_dir, child["id"], binding, - [], warnings, stopped=True, reason="Independent scan stopped before aggregation.", @@ -1392,16 +882,18 @@ def save_composed_checkpoint( aggregate = {"findings": [], "coverage": {}} represented = set() for finding in aggregate["findings"]: - for retained in _retained_findings(finding): + for _, retained in retained_findings(finding): provenance = retained.get("provenance", {}) represented.update(provenance.get("sourceFindingIds", [])) represented.update(source["id"] for source in provenance.get("sourceFindings", [])) + recovery_errors = {} for child in children.values(): if child["id"] in merged_ids: continue try: draft = _stopped_child_draft(db, child, scan_dir) - except (ContractError, OSError, SystemExit, ValueError): + except (ContractError, OSError, SystemExit, ValueError) as exc: + recovery_errors[child["id"]] = str(exc) continue if draft is None: continue @@ -1410,11 +902,7 @@ def save_composed_checkpoint( for finding in draft["findings"] if not represented.intersection(finding["provenance"]["sourceFindingIds"]) ) - for field in ("surfaces", "explicitExclusions", "deferred", "openQuestions"): - rows = aggregate.setdefault("coverage", {}).setdefault(field, []) - for row in draft["coverage"].get(field, []): - if row not in rows: - rows.append(row) + merge_coverage(aggregate.setdefault("coverage", {}), draft["coverage"]) aggregate["scanId"] = scan["id"] aggregate["complete"] = False coverage = aggregate.setdefault("coverage", {}) @@ -1437,6 +925,8 @@ def save_composed_checkpoint( else f"unmerged-{_digest(relative)[:16]}", "reason": f"Independent scan did not complete and merge. Saved work: {relative}.", } + if child is not None and child["id"] in recovery_errors: + note["reason"] += f" Recovery failed: {recovery_errors[child['id']]}" if note not in deferred: deferred.append(note) payload = _encoded(aggregate) @@ -1581,12 +1071,6 @@ def record_publication(manifest: dict[str, Any], findings: dict[str, Any]) -> No scan_dir, scan_id, binding, - connection.execute( - "SELECT * FROM deep_scan_workers WHERE scan_id = ? ORDER BY created_at, id", - (scan_id,), - ).fetchall() - if read_composition_checkpoint(scan) is None - else [], warnings, stopped=True, reason=( @@ -1683,11 +1167,7 @@ def preserve_scan_results(db: Any, connection: Any, args: Any) -> dict[str, Any] if scan["status"] == "complete": raise SystemExit("A completed scan cannot preserve new results.") workspace = db.require_workspace(connection, scan["workspace_id"]) - owner = ( - scan["deep_scan_owner_thread_id"] - or scan["continuation_thread_id"] - or workspace["thread_id"] - ) + owner = db.handoff.durable_owner_thread_id(scan, workspace) if args.thread_id is not None and args.thread_id != owner: raise SystemExit("Saved results can only be published from the owning Codex thread.") # The app can cancel before a continuation has claimed the scan. @@ -1703,9 +1183,7 @@ def preserve_scan_results(db: Any, connection: Any, args: Any) -> dict[str, Any] error_message="Saved results are owned by another continuation.", ) if cost_json is not None: - stored = stored_scan_cost_fields(scan["cost_json"]) - if "usage" in stored: - cost_json = json.dumps({**stored, "cost": json.loads(cost_json)}) + cost_json = merge_scan_cost(scan["cost_json"], cost_json) with connection: connection.execute( "UPDATE scans SET cost_json = ? WHERE id = ?", (cost_json, scan_id) @@ -1905,12 +1383,10 @@ def fail_scan_locked(db: Any, connection: Any, args: Any) -> dict[str, Any]: args.claim_token, error_message="Scan failure is owned by another continuation.", ) + if cost_json is not None: + cost_json = merge_scan_cost(scan["cost_json"], cost_json) if scan["status"] == "failed": if cost_json is not None: - stored = stored_scan_cost_fields(scan["cost_json"]) - incoming = json.loads(cost_json) - if "usage" in stored and "usage" not in incoming: - cost_json = json.dumps({**stored, "cost": incoming}) connection.execute( "UPDATE scans SET cost_json = ? WHERE id = ?", (cost_json, scan_id) ) diff --git a/plugins/codex-security/scripts/workbench_scan_history.py b/plugins/codex-security/scripts/workbench_scan_history.py index 9a16d42de..ef4298055 100644 --- a/plugins/codex-security/scripts/workbench_scan_history.py +++ b/plugins/codex-security/scripts/workbench_scan_history.py @@ -19,7 +19,6 @@ from workbench.handoff import require_current_continuation from workbench_composition import ( CompositionView, - composition_child_ids, load_composition, read_composition_checkpoint, ) @@ -92,35 +91,58 @@ def cli_scan_resume( scan_dir = require_scan_directory(Path(scan["scan_dir"])) result = scan_registration(connection, scan, scan_contract) result["recipe"] = recipe + sealed_version = sealed_scan_producer_version( + scan, + scan_dir, + artifact_path=artifact_path, + read_json_object=read_json_object, + workbench_completion_binding=workbench_completion_binding, + ) + if sealed_version is None: + require_current_deep_runtime(connection, scan) + else: + result["sealedProducerVersion"] = sealed_version + return result + + +def sealed_scan_producer_version( + scan: sqlite3.Row, + scan_dir: Path, + *, + artifact_path: Callable[..., Path | None], + read_json_object: Callable[[Path], dict[str, Any]], + workbench_completion_binding: Callable[..., dict[str, Any]], +) -> str | None: + # A process can stop after sealing files but before committing completion. + manifest_path = artifact_path(scan_dir, ARTIFACTS["manifest"], required=False) + if manifest_path is None: + return None + manifest = read_json_object(manifest_path) + manifest_scan = manifest.get("scan") + if not isinstance(manifest_scan, dict) or ( + manifest_scan.get("sealedAt") is None and manifest_scan.get("artifacts") in (None, []) + ): + return None + try: + binding = workbench_completion_binding(scan, scan["started_at"], manifest) + _prepare_scan_finalization( + scan_dir, + expected_coverage_mode=binding["coverageMode"], + completion_binding=binding, + ) + return manifest_scan["producer"]["version"] + except ContractError as exc: + raise SystemExit(f"Cannot resume sealed scan: {exc}") from exc + + +def require_current_deep_runtime(connection: sqlite3.Connection, scan: sqlite3.Row) -> None: if ( scan["mode"] == "deep" - and read_composition_checkpoint(scan) is None and connection.execute( "SELECT 1 FROM deep_scan_runs WHERE scan_id = ?", (scan["id"],) ).fetchone() - is not None ): - result["threadId"] = None - # A process can stop after sealing files but before committing completion. - manifest_path = artifact_path(scan_dir, ARTIFACTS["manifest"], required=False) - if manifest_path is not None: - manifest = read_json_object(manifest_path) - manifest_scan = manifest.get("scan") - if isinstance(manifest_scan, dict) and ( - manifest_scan.get("sealedAt") is not None - or manifest_scan.get("artifacts") not in (None, []) - ): - try: - binding = workbench_completion_binding(scan, scan["started_at"], manifest) - _prepare_scan_finalization( - scan_dir, - expected_coverage_mode=binding["coverageMode"], - completion_binding=binding, - ) - result["sealedProducerVersion"] = manifest_scan["producer"]["version"] - except ContractError as exc: - raise SystemExit(f"Cannot resume sealed scan: {exc}") from exc - return result + raise SystemExit("This Deep Scan uses a retired runtime. Start a fresh scan.") def scan_registration( @@ -1059,35 +1081,36 @@ def finding_matches( occurrences.title, matches.reason FROM scan_comparison_matches AS matches JOIN finding_occurrences AS occurrences ON occurrences.id = matches.after_occurrence_id + JOIN scans ON scans.id = matches.after_scan_id WHERE matches.before_occurrence_id = ? + AND (scans.parent_scan_role IS NOT 'deep_pass' OR scans.id = ?) UNION SELECT matches.before_scan_id AS scan_id, occurrences.id AS occurrence_id, occurrences.finding_id, occurrences.title, matches.reason FROM scan_comparison_matches AS matches JOIN finding_occurrences AS occurrences ON occurrences.id = matches.before_occurrence_id + JOIN scans ON scans.id = matches.before_scan_id WHERE matches.after_occurrence_id = ? + AND (scans.parent_scan_role IS NOT 'deep_pass' OR scans.id = ?) ORDER BY scan_id, occurrence_id """, - (occurrence_id, occurrence_id), + (occurrence_id, scan_id, occurrence_id, scan_id), ).fetchall() linked_rows = list( - _rows_for_ids( - connection, + connection.execute( f""" - {_LINKED_FINDINGS_SQL} + {_LINKED_FINDINGS_SQL.format(placeholders="?")} SELECT occurrences.id AS occurrence_id, occurrences.finding_id, occurrences.title, scans.started_at, scans.id AS scan_id FROM linked CROSS JOIN finding_occurrences AS occurrences ON occurrences.finding_id = linked.finding_id CROSS JOIN scans ON scans.id = occurrences.scan_id + WHERE scans.parent_scan_role IS NOT 'deep_pass' OR scans.id = ? """, - (occurrence_id,), + (occurrence_id, scan_id), ) ) - child_ids = composition_child_ids(connection) - {scan_id} - linked_rows = [row for row in linked_rows if row["scan_id"] not in child_ids] - rows = [row for row in rows if row["scan_id"] not in child_ids] known_scans = sorted( {(started_at, scan_id)} | {(row["started_at"], row["scan_id"]) for row in linked_rows} ) diff --git a/plugins/codex-security/scripts/workbench_scan_usage.py b/plugins/codex-security/scripts/workbench_scan_usage.py index b77cebf73..e1b0729ad 100644 --- a/plugins/codex-security/scripts/workbench_scan_usage.py +++ b/plugins/codex-security/scripts/workbench_scan_usage.py @@ -66,13 +66,7 @@ def reconcile_completed_scan_cost( ) -> None: """Persist authoritative SDK cost without discarding measured worker usage.""" - existing = json.loads(scan["cost_json"]) if scan["cost_json"] is not None else {} - if isinstance(existing, dict) and "usage" in existing: - cost_json = json.dumps( - {**existing, "cost": json.loads(cost_json)}, - separators=(",", ":"), - allow_nan=False, - ) + cost_json = merge_scan_cost(scan["cost_json"], cost_json) connection.execute("BEGIN IMMEDIATE") try: connection.execute( @@ -85,6 +79,16 @@ def reconcile_completed_scan_cost( raise +def merge_scan_cost(stored: str | None, incoming: str | None) -> str | None: + """Replace supplied cost/usage fields while retaining the other measured fields.""" + fields = {**stored_scan_cost_fields(stored), **stored_scan_cost_fields(incoming)} + if not fields: + return None + return json.dumps( + fields if "usage" in fields else fields["cost"], separators=(",", ":"), allow_nan=False + ) + + def collect_scan_usage( connection: sqlite3.Connection, scan: sqlite3.Row, diff --git a/plugins/codex-security/scripts/workbench_schema.py b/plugins/codex-security/scripts/workbench_schema.py index e05cae34b..65df4bbe3 100644 --- a/plugins/codex-security/scripts/workbench_schema.py +++ b/plugins/codex-security/scripts/workbench_schema.py @@ -4,7 +4,6 @@ import json import sqlite3 from collections.abc import Callable -from pathlib import Path MIGRATIONS = ( ( @@ -919,25 +918,6 @@ ) -def backfill_composition_children(connection: sqlite3.Connection) -> None: - # Preserve the previous membership rule using stored paths, including archived - # scans and scans whose outputs no longer exist. Do not consult checkpoints. - rows = connection.execute( - "SELECT children.id, children.scan_dir, parents.scan_dir AS parent_scan_dir " - "FROM scans AS children JOIN scans AS parents ON parents.id = children.parent_scan_id " - "WHERE parents.mode = 'deep' AND children.mode = 'standard'" - ).fetchall() - connection.executemany( - "UPDATE scans SET parent_scan_role = 'deep_pass' WHERE id = ?", - ( - (child["id"],) - for child in rows - if Path(child["scan_dir"]).parent - == Path(child["parent_scan_dir"]) / "artifacts/deep-scan/passes" - ), - ) - - def migrate_finding_workflow_review_columns(connection: sqlite3.Connection) -> None: for row in connection.execute( "SELECT workflow_id, review_key, prompt_digest FROM finding_workflow_reviews" @@ -1091,8 +1071,6 @@ def apply_migrations( migrate_finding_workflow_columns(connection) elif version == 39: migrate_finding_workflow_review_columns(connection) - elif version == 43: - backfill_composition_children(connection) connection.execute( "INSERT INTO schema_migrations (version, name, applied_at) VALUES (?, ?, ?)", (version, name, now()), diff --git a/plugins/codex-security/tests/test_deep_scan_successful_publication.py b/plugins/codex-security/tests/test_deep_scan_successful_publication.py index bf7f17f99..0ef3bf86e 100644 --- a/plugins/codex-security/tests/test_deep_scan_successful_publication.py +++ b/plugins/codex-security/tests/test_deep_scan_successful_publication.py @@ -356,27 +356,12 @@ def test_deep_prepare_and_complete_preserve_the_same_aggregate( assert_published_aggregate(scan) -@pytest.mark.parametrize( - ("source", "scope", "has_parent"), - [ - ("standard-worker-checkpoint", ".", True), - ("deep-reducer-checkpoint", ".", True), - ("deep-reducer-archived-checkpoint", ".", True), - ("deep-reducer-result", ".", True), - ("deep-reducer-result", "subdir", False), - ], - ids=[ - "standard-worker-checkpoint", - "deep-reducer-checkpoint", - "deep-reducer-archived-checkpoint", - "deep-reducer-result", - "scoped-reducer-without-parent", - ], -) -def test_stopped_deep_scan_still_salvages_saved_findings( - workbench_api, workbench_db, publication_scan, source, scope, has_parent +@pytest.mark.parametrize("mode", ["standard", "deep"]) +@pytest.mark.parametrize(("scope", "has_parent"), [(".", True), ("subdir", False)]) +def test_stopped_scan_salvages_saved_parent_checkpoints( + workbench_api, workbench_db, publication_scan, mode, scope, has_parent ): - scan = publication_scan(scope=scope) + scan = publication_scan(mode=mode, scope=scope) if not has_parent: for name in ("scan-manifest.json", "findings.json", "coverage.json"): (scan.scan_dir / name).unlink() @@ -386,42 +371,17 @@ def test_stopped_deep_scan_still_salvages_saved_findings( "terminal_reason = NULL, completed_at = NULL WHERE scan_id = ?", (scan.scan_id,), ) - result = add_worker( - workbench_db, scan, status="succeeded" if source == "deep-reducer-result" else "running" - ) - if source.startswith("deep-reducer"): - with workbench_db: - workbench_db.execute( - "UPDATE deep_scan_workers SET kind = 'dedup', merge_state = 'none' WHERE id = ?", - (result.parent.name,), - ) later_finding = copy.deepcopy(scan.findings[0]) later_finding["identity"]["anchor"] = "later-checkpoint-finding" later_finding["summary"] = "Finding saved after the last completed aggregate." saved = { "scanId": scan.scan_id, - "complete": source == "deep-reducer-result", + "complete": False, "findings": [later_finding], + "coverage": scan.coverage, } - if source == "standard-worker-checkpoint": - saved["coverage"] = { - "completeness": "partial", - "surfaces": [], - "explicitExclusions": [], - "deferred": [], - } - if source == "deep-reducer-result": - checkpoint = None - result.write_text(json.dumps(saved)) - else: - checkpoint_root = ( - result.parent / "attempts" / "attempt-01" - if source == "deep-reducer-archived-checkpoint" - else result.parent - ) - checkpoint = write_checkpoint(checkpoint_root / "checkpoints", saved) - result.write_text("{interrupted worker output") - result_bytes = result.read_bytes() + checkpoint = write_checkpoint(scan.scan_dir / "checkpoints", saved) + checkpoint_bytes = checkpoint.read_bytes() stopped = workbench_api["fail_scan"]( workbench_db, @@ -448,9 +408,7 @@ def test_stopped_deep_scan_still_salvages_saved_findings( )["scan"] assert recovered["findingCount"] == len(expected_summaries) assert {name: (scan.scan_dir / name).read_bytes() for name in artifact_names} == published - if checkpoint is not None: - assert json.loads(checkpoint.read_text()) == saved - assert result.read_bytes() == result_bytes + assert checkpoint.read_bytes() == checkpoint_bytes @pytest.mark.parametrize("source", ["result", "checkpoint", "parent-checkpoint"]) diff --git a/plugins/codex-security/tests/test_report_projection.py b/plugins/codex-security/tests/test_report_projection.py index 4fba9fcc8..13be4f8d6 100644 --- a/plugins/codex-security/tests/test_report_projection.py +++ b/plugins/codex-security/tests/test_report_projection.py @@ -132,6 +132,25 @@ def test_projection_retains_distinct_source_fixes(linked_writeup: bool) -> None: assert markdown.count(text) == 1 +def test_retained_findings_visit_sources_before_history_and_handle_cycles() -> None: + previous = {"remediation": "Retain the earlier fix."} + source = {"provenance": {"previousFindings": [previous, None]}} + finding = { + "provenance": { + "sourceFindings": [{"id": "source:0", "finding": source}, {"finding": None}], + "previousFindings": [previous], + } + } + previous["provenance"] = {"previousFindings": [finding]} + assert [ + (source_id, id(value)) for source_id, value in PROJECTION.retained_findings(finding) + ] == [ + ("finding", id(finding)), + ("source:0", id(source)), + ("source:0", id(previous)), + ] + + def test_projection_renders_inline_code_and_section_code_evidence() -> None: manifest, findings, coverage = canonical_documents() finding = findings["findings"][0] diff --git a/plugins/codex-security/tests/test_scan_projection.py b/plugins/codex-security/tests/test_scan_projection.py index 9dde63cca..64581f502 100644 --- a/plugins/codex-security/tests/test_scan_projection.py +++ b/plugins/codex-security/tests/test_scan_projection.py @@ -8,8 +8,32 @@ from pathlib import Path import pytest -from test_workbench_scan_composition import register -from workbench_test_support import run_workbench, write_completed_contract +from workbench_test_support import register, run_workbench, write_completed_contract + + +def test_coverage_union_keeps_distinct_rows_with_the_same_id(workbench_api) -> None: + coverage = { + "completeness": "partial", + "surfaces": [{"id": "surface", "notes": "Earlier observation"}], + } + addition = { + "surfaces": [ + {"notes": "Earlier observation", "id": "surface"}, + {"id": "surface", "notes": "Later observation"}, + ], + "openQuestions": [{"question": "Remaining coverage?"}], + } + workbench_api["saved_results"].merge_coverage(coverage, addition) + assert coverage == { + "completeness": "partial", + "surfaces": [ + {"id": "surface", "notes": "Earlier observation"}, + {"id": "surface", "notes": "Later observation"}, + ], + "explicitExclusions": [], + "deferred": [], + "openQuestions": [{"question": "Remaining coverage?"}], + } @pytest.fixture diff --git a/plugins/codex-security/tests/test_workbench_scan_composition.py b/plugins/codex-security/tests/test_workbench_scan_composition.py index bb6d7af2d..fac134b3d 100644 --- a/plugins/codex-security/tests/test_workbench_scan_composition.py +++ b/plugins/codex-security/tests/test_workbench_scan_composition.py @@ -13,12 +13,51 @@ from unittest import mock import pytest -from workbench_test_support import SCRIPT, run_workbench, write_checkpoint, write_completed_contract +from workbench_test_support import ( + SCRIPT, + checkpoint, + recipe, + register, + run_workbench, + write_checkpoint, + write_completed_contract, +) CHECKPOINT = "artifacts/deep-scan/checkpoint.json" EXECUTION_THREADS = "artifacts/deep-scan/execution-threads.json" +def test_composed_recovery_records_child_failure_and_continues(workbench_api, monkeypatch) -> None: + saved = workbench_api["saved_results"] + root = Path("/synthetic-scan") + children = [ + {"id": "broken", "scan_dir": str(root / "broken")}, + {"id": "retained", "scan_dir": str(root / "retained")}, + ] + monkeypatch.setattr(saved, "read_composition_checkpoint", lambda _: None) + monkeypatch.setattr(saved, "composition_children", lambda *_: children) + monkeypatch.setattr(saved, "write_scan_local_bytes", lambda *_: None) + retained_coverage = {"surfaces": [{"id": "retained/surface", "summary": "Saved work"}]} + with mock.patch.object( + saved, + "_stopped_child_draft", + side_effect=[ + ValueError("Synthetic malformed artifact"), + {"findings": [], "coverage": retained_coverage}, + ], + ): + result = saved.save_composed_checkpoint( + None, None, {"id": "parent", "scan_dir": str(root)}, root + ) + assert result["coverage"]["surfaces"] == retained_coverage["surfaces"] + assert result["coverage"]["deferred"][0] == { + "id": "unmerged-broken", + "reason": "Independent scan did not complete and merge. Saved work: broken. " + "Recovery failed: Synthetic malformed artifact", + } + assert result["complete"] is False + + @pytest.mark.parametrize("alias", ["exact", "case", "directory"]) def test_stopped_projection_retains_report_and_colliding_evidence(tmp_path: Path, alias: str): target = tmp_path / "target" @@ -70,65 +109,6 @@ def test_stopped_projection_retains_report_and_colliding_evidence(tmp_path: Path assert evidence.read_text() == "Synthetic supporting evidence\n" -def recipe(target: Path, mode: str = "standard") -> dict: - return { - "repository": str(target), - "target": {"kind": "repository", "paths": []}, - "mode": mode, - "config": {"model": "synthetic-model", "model_reasoning_effort": "high"}, - **({"deepScan": {"maxDiscoveryRuns": 8}} if mode == "deep" else {}), - } - - -def register( - state: Path, target: Path, directory: Path, *, mode="standard", parent=None, role=None, paths=() -) -> dict: - missing = [] - current = directory - while not current.exists(): - missing.append(current) - current = current.parent - for path in reversed(missing): - path.mkdir(mode=0o700) - saved_recipe = recipe(target, mode) - if paths: - saved_recipe["target"] = {"kind": "paths", "paths": list(paths)} - return run_workbench( - state, - "register-cli-scan", - "--repository", - str(target), - "--scan-dir", - str(directory), - "--registration-json-stdin", - *(("--parent-scan-id", parent) if parent else ()), - input_text=json.dumps({"recipe": saved_recipe, "parentScanRole": role}), - ) - - -def checkpoint(state: Path, scan: dict, *, passes=(), merged=(), terminal=None) -> dict: - value = { - "version": 2, - "startedAt": "2026-01-01T00:00:00Z", - "passes": list(passes), - "mergedScanIds": list(merged), - "aggregate": None, - "noNewStreak": 0, - "consecutiveErrors": 0, - **({"terminalReason": terminal} if terminal else {}), - } - run_workbench( - state, - "save-scan-artifact", - "--scan-id", - scan["scanId"], - "--artifact-path", - CHECKPOINT, - input_text=json.dumps(value), - ) - return value - - @pytest.mark.parametrize("accepted", [False, True]) def test_explicit_recovery_materializes_unfrozen_composition_after_checkpoint_failure( tmp_path: Path, workbench_api, accepted: bool @@ -262,7 +242,7 @@ def test_explicit_recovery_materializes_unfrozen_composition_after_checkpoint_fa @pytest.mark.parametrize("name", ["current", "legacy"]) -def test_checkpoint_roundtrips_shared_sdk_fixtures(tmp_path, workbench_api, monkeypatch, name): +def test_checkpoint_reads_shared_sdk_fixtures(tmp_path, workbench_api, monkeypatch, name): target = tmp_path / "target" target.mkdir() (target / "app.py").write_text("print('fixture')\n") @@ -283,18 +263,6 @@ def test_checkpoint_roundtrips_shared_sdk_fixtures(tmp_path, workbench_api, monk stored = {"id": scan["scanId"], "scan_dir": scan["scanDir"]} loaded = workbench_api["read_composition_checkpoint"](stored) assert loaded == original - encoded = workbench_api["saved_results"].encode_composition_checkpoint(loaded) - assert json.loads(encoded) == original - run_workbench( - state, - "save-scan-artifact", - "--scan-id", - scan["scanId"], - "--artifact-path", - CHECKPOINT, - input_text=encoded.decode(), - ) - assert workbench_api["read_composition_checkpoint"](stored) == original def test_checkpoint_read_blocks_other_threads_and_atomic_writers( @@ -589,6 +557,105 @@ def complete(): return state, target, arguments, started, complete +@pytest.mark.parametrize("saved_recipe", [False, True]) +@pytest.mark.parametrize("artifact_state", ["unsealed", "sealed", "tampered"]) +def test_native_legacy_registration_only_rejoins_validated_sealed_results( + native_scan_completion, saved_recipe: bool, artifact_state: str +) -> None: + state, target, _, started, complete = native_scan_completion + scan = started["scan"] + directory = Path(scan["scanDir"]) + token = scan["handoffClaimToken"] + run_workbench( + state, + "set-scan-thread", + "--scan-id", + scan["scanId"], + "--thread-id", + "saved-execution", + "--claim-token", + token, + ) + if artifact_state != "unsealed": + run_workbench( + state, "prepare-scan-completion", "--scan-id", scan["scanId"], "--claim-token", token + ) + if artifact_state == "tampered": + findings_path = directory / "findings.json" + findings_path.write_bytes(findings_path.read_bytes() + b" ") + checkpoint_path = directory / CHECKPOINT + checkpoint = json.loads(checkpoint_path.read_text()) + checkpoint["legacy"] = {"discoveryRuns": 1, "coverage": {"completeness": "complete"}} + checkpoint_path.write_text(json.dumps(checkpoint)) + identity_query = ( + "SELECT recipe_json, continuation_thread_id, deep_scan_owner_thread_id, handoff_claim_token " + "FROM scans WHERE id = ?" + ) + with sqlite3.connect(state / "workbench.sqlite3") as connection: + if not saved_recipe: + connection.execute( + "UPDATE scans SET recipe_json = NULL WHERE id = ?", (scan["scanId"],) + ) + connection.execute( + "INSERT INTO deep_scan_runs (scan_id, schema_version, workflow_version, status, phase, " + "workers, subagents, stop_after_no_new, max_discovery_runs, created_at, updated_at, " + "terminal_reason, manifest_path) " + "SELECT id, 1, 'synthetic-legacy', 'succeeded', 'terminal', 1, 0, 3, 8, started_at, " + "updated_at, 'saturated', ? FROM scans WHERE id = ?", + (str(directory / "scan-manifest.json"), scan["scanId"]), + ) + original_identity = connection.execute(identity_query, (scan["scanId"],)).fetchone() + originals = { + name: (directory / name).read_bytes() + for name in ("scan-manifest.json", "findings.json", "coverage.json", CHECKPOINT) + } + rebound = run_workbench( + state, + "register-cli-scan", + "--repository", + str(target), + "--scan-dir", + str(directory), + "--registration-json-stdin", + input_text=json.dumps( + { + "scanId": scan["scanId"], + "threadId": "native-owner", + "claimToken": token, + "recipe": recipe(target, "deep"), + } + ), + check=artifact_state == "sealed", + ) + if artifact_state == "sealed": + assert rebound["scanId"] == scan["scanId"] + assert rebound["threadId"] == "saved-execution" + resumed = run_workbench( + state, "get-cli-scan-resume", "--scan-id", scan["scanId"], "--claim-token", token + ) + assert resumed["threadId"] == "saved-execution" + with sqlite3.connect(state / "workbench.sqlite3") as connection: + assert ( + connection.execute(identity_query, (scan["scanId"],)).fetchone()[1:] + == original_identity[1:] + ) + assert ( + resumed["sealedProducerVersion"] + == json.loads(originals["scan-manifest.json"])["scan"]["producer"]["version"] + ) + assert complete()["progress"]["status"] == "complete" + else: + assert ( + "retired runtime" if artifact_state == "unsealed" else "Cannot resume sealed scan" + ) in rebound["stderr"] + with sqlite3.connect(state / "workbench.sqlite3") as connection: + assert ( + connection.execute(identity_query, (scan["scanId"],)).fetchone() + == original_identity + ) + assert {name: (directory / name).read_bytes() for name in originals} == originals + + @pytest.mark.parametrize("during_retry", [False, True]) def test_native_target_retry_reuses_completed_result( native_scan_completion, workbench_api, monkeypatch, during_retry: bool @@ -876,9 +943,8 @@ def test_native_parent_binds_once_and_keeps_native_claim( @pytest.mark.parametrize("target_entry", [False, True]) -@pytest.mark.parametrize("recipe_maximum", [None, 9]) -def test_native_legacy_settings_are_returned_only_without_a_saved_recipe( - tmp_path: Path, target_entry: bool, recipe_maximum: int | None +def test_native_legacy_settings_remain_readable_without_rebinding( + tmp_path: Path, target_entry: bool ) -> None: target = tmp_path / "target" target.mkdir() @@ -962,12 +1028,7 @@ def test_native_legacy_settings_are_returned_only_without_a_saved_recipe( assert connection.execute( "SELECT deep_scan_owner_thread_id, continuation_thread_id, handoff_claim_token FROM scans" ).fetchall() == [("native-owner", None if target_entry else "native-owner", token)] - saved_recipe = recipe(target, "deep") - if recipe_maximum is None: - del saved_recipe["deepScan"] - else: - saved_recipe["deepScan"]["maxDiscoveryRuns"] = recipe_maximum - run_workbench( + rejected = run_workbench( state, "register-cli-scan", "--repository", @@ -977,59 +1038,18 @@ def test_native_legacy_settings_are_returned_only_without_a_saved_recipe( "--registration-json-stdin", input_text=json.dumps( { - "recipe": saved_recipe, + "recipe": recipe(target, "deep"), "scanId": scan["scanId"], "threadId": "native-owner", "claimToken": token, } ), + check=False, ) - joined = run_workbench(state, *joined_args) - assert joined["scan"]["scanId"] == scan["scanId"] - assert joined["recipe"] == saved_recipe - assert "deepScanSettings" not in joined - assert joined["compositionCheckpoint"]["legacy"]["originThreadId"] == "native-owner" - assert joined["compositionCheckpoint"]["legacy"]["discoveryRuns"] == 3 - assert joined["compositionCheckpoint"]["noNewStreak"] == 2 - assert joined["compositionCheckpoint"]["consecutiveErrors"] == 1 - assert joined["scan"]["progress"]["independentReviews"] == { - "active": 0, - "completed": 2, - "maximum": recipe_maximum or 8, - "consolidating": False, - } - scan_dir = Path(scan["scanDir"]) - saved_checkpoint = json.loads((scan_dir / CHECKPOINT).read_text()) - children = [] - for index in range(1, 3): - directory = f"artifacts/deep-scan/passes/pass-{index}" - child = register( - state, target, scan_dir / directory, parent=scan["scanId"], role="deep_pass" - ) - children.append(child) - saved_checkpoint["passes"].append({"directory": directory, "scanId": child["scanId"]}) - run_workbench( - state, - "save-scan-artifact", - "--scan-id", - scan["scanId"], - "--artifact-path", - CHECKPOINT, - *(("--claim-token", token) if token else ()), - input_text=json.dumps(saved_checkpoint), - ) - child = children[0] - write_completed_contract( - Path(child["scanDir"]), child["scanId"], target, relative_path="app.py" - ) - run_workbench(state, "complete-scan", "--scan-id", child["scanId"]) - joined = run_workbench(state, *joined_args) - assert joined["scan"]["progress"]["independentReviews"] == { - "active": 1, - "completed": 3, - "maximum": recipe_maximum or 8, - "consolidating": True, - } + assert "retired runtime. Start a fresh scan" in rejected["stderr"] + with sqlite3.connect(state / "workbench.sqlite3") as connection: + assert connection.execute("SELECT recipe_json FROM scans").fetchone() == (None,) + assert connection.execute("SELECT * FROM deep_scan_runs").fetchone() == legacy @pytest.mark.parametrize( @@ -1313,56 +1333,6 @@ def test_explicit_child_membership_does_not_depend_on_directory_or_checkpoint( ) -@pytest.mark.parametrize("missing_outputs", [False, True]) -def test_membership_migration_backfills_stored_paths_once( - tmp_path: Path, missing_outputs: bool -) -> None: - target = tmp_path / "target" - target.mkdir() - state = tmp_path / "state" - parent_dir = tmp_path / "parent.previous-synthetic" - parent = register(state, target, parent_dir, mode="deep") - child = register( - state, target, parent_dir / "artifacts/deep-scan/passes/pass-1", parent=parent["scanId"] - ) - rerun = register(state, target, tmp_path / "rerun", parent=parent["scanId"]) - database = state / "workbench.sqlite3" - marker = parent_dir / "saved-output.txt" - marker.write_bytes(b"Saved outputs must not change during migration.") - with sqlite3.connect(database) as connection: - connection.execute("DROP INDEX scans_by_composition_parent") - connection.execute("ALTER TABLE scans DROP COLUMN parent_scan_role") - connection.execute("DELETE FROM schema_migrations WHERE version = 43") - before = connection.execute( - "SELECT id, parent_scan_id, scan_dir, status FROM scans ORDER BY id" - ).fetchall() - if missing_outputs: - parent_dir.rename(tmp_path / "removed-output") - run_workbench(state, "database-info") - run_workbench(state, "database-info") - with sqlite3.connect(database) as connection: - assert ( - connection.execute( - "SELECT id, parent_scan_id, scan_dir, status FROM scans ORDER BY id" - ).fetchall() - == before - ) - assert dict(connection.execute("SELECT id, parent_scan_role FROM scans")) == { - parent["scanId"]: None, - child["scanId"]: "deep_pass", - rerun["scanId"]: None, - } - assert connection.execute( - "SELECT COUNT(*) FROM schema_migrations WHERE version = 43" - ).fetchone() == (1,) - saved_marker = tmp_path / "removed-output/saved-output.txt" if missing_outputs else marker - assert saved_marker.read_bytes() == b"Saved outputs must not change during migration." - assert {scan["scanId"] for scan in run_workbench(state, "list-scans")["scans"]} == { - parent["scanId"], - rerun["scanId"], - } - - def test_failed_deep_scan_keeps_followup_thread_before_composition_checkpoint( tmp_path: Path, ) -> None: @@ -2103,9 +2073,12 @@ def test_deferred_stop_retains_drained_child_and_cost_before_freezing( run_workbench(state, "get-scan", "--scan-id", child["scanId"])["scan"]["progress"]["status"] == "running" ) - retained = run_workbench(state, *preserve) + retained = run_workbench( + state, *preserve, "--cost-json", json.dumps({"usage": usage, "cost": cost}) + ) assert retained["scan"]["findingCount"] == 1 assert retained["scan"]["cost"] == cost + assert retained["scan"]["usage"] == usage assert retained["scan"]["progress"]["independentReviews"]["active"] == 0 retained_child = run_workbench(state, "get-scan", "--scan-id", child["scanId"])["scan"] assert retained_child["progress"]["status"] == "failed" diff --git a/plugins/codex-security/tests/test_workbench_scan_history.py b/plugins/codex-security/tests/test_workbench_scan_history.py index b9dbc0074..314d4641b 100644 --- a/plugins/codex-security/tests/test_workbench_scan_history.py +++ b/plugins/codex-security/tests/test_workbench_scan_history.py @@ -26,6 +26,37 @@ FINALIZER = SCRIPT.with_name("finalize_scan_contract.py") +def test_finding_matches_hide_other_children_and_keep_requested_child(workbench_api) -> None: + with sqlite3.connect(":memory:") as connection: + connection.row_factory = sqlite3.Row + connection.executescript( + "CREATE TABLE scans (id TEXT, started_at TEXT, parent_scan_role TEXT);" + "CREATE TABLE finding_occurrences (id TEXT, scan_id TEXT, finding_id TEXT, title TEXT);" + "CREATE TABLE scan_comparison_matches (before_scan_id TEXT, after_scan_id TEXT, " + "before_occurrence_id TEXT, after_occurrence_id TEXT, reason TEXT);" + ) + connection.executemany( + "INSERT INTO scans VALUES (?, ?, ?)", + [("parent", "1", None), ("child", "2", "deep_pass"), ("rerun", "3", None)], + ) + connection.executemany( + "INSERT INTO finding_occurrences VALUES (?, ?, 'stable', 'Synthetic finding')", + [("p", "parent"), ("c", "child"), ("c2", "child"), ("r", "rerun")], + ) + connection.executemany( + "INSERT INTO scan_comparison_matches VALUES (?, ?, ?, ?, 'Confirmed')", + [("parent", "child", "p", "c"), ("child", "rerun", "c", "r")], + ) + matches = workbench_api["scan_history"].finding_matches + for occurrence, scan_id, started in (("p", "parent", "1"), ("r", "rerun", "3")): + rows, known_since, known_scans = matches(connection, occurrence, scan_id, started) + assert {row["scanId"] for row in rows} == ({"parent", "rerun"} - {scan_id}) + assert known_since == "1" + assert known_scans == ["parent", "rerun"] + rows, _, _ = matches(connection, "c", "child", "2") + assert {row["occurrenceId"] for row in rows} == {"p", "c2", "r"} + + def run_workbench(state_dir: Path, *args: str, check: bool = True) -> dict[str, Any]: completed = subprocess.run( [sys.executable, str(SCRIPT), *args], diff --git a/plugins/codex-security/tests/test_workbench_scan_usage.py b/plugins/codex-security/tests/test_workbench_scan_usage.py index 546f80a51..b503961d7 100644 --- a/plugins/codex-security/tests/test_workbench_scan_usage.py +++ b/plugins/codex-security/tests/test_workbench_scan_usage.py @@ -34,6 +34,46 @@ class ScanFixture: diff_target: dict[str, Any] | None = None +@pytest.mark.parametrize("include_cost", [False, True]) +def test_cost_envelopes_preserve_usage_without_nesting(workbench_api, include_cost: bool) -> None: + usage = { + "coverage": "unavailable", + "source": "codex_rollout", + "threadCount": 0, + "warnings": ["scan_thread_unavailable"], + } + measured = { + "coverage": "complete", + "source": "codex_rollout", + **_counts(0, 0, 0), + "threadCount": 1, + } + cost = { + "model": "synthetic-model", + "inputTokens": 0, + "cachedInputTokens": 0, + "cacheWriteInputTokens": 0, + "outputTokens": 0, + "estimatedUsd": 0, + } + merge = workbench_api["scan_usage"].merge_scan_cost + stored = json.dumps({"usage": usage, "cost": cost}) + incoming = json.dumps({"usage": measured, **({"cost": cost} if include_cost else {})}) + assert json.loads(merge(stored, incoming)) == {"usage": measured, "cost": cost} + assert json.loads(merge(stored, json.dumps(cost))) == {"usage": usage, "cost": cost} + assert json.loads(merge(None, json.dumps(cost))) == cost + assert merge(None, None) is None + with sqlite3.connect(":memory:") as connection: + connection.execute("CREATE TABLE scans (id TEXT, status TEXT, cost_json TEXT)") + connection.execute("INSERT INTO scans VALUES ('scan', 'complete', ?)", (stored,)) + connection.commit() + workbench_api["scan_usage"].reconcile_completed_scan_cost( + connection, {"id": "scan", "cost_json": stored}, incoming + ) + receipt = connection.execute("SELECT cost_json FROM scans").fetchone()[0] + assert json.loads(receipt) == {"usage": measured, "cost": cost} + + def _start_scan(tmp_path: Path, *, mode: str = "standard") -> ScanFixture: state_dir = tmp_path / "workbench-state" target = tmp_path / "target" @@ -715,6 +755,47 @@ def test_native_completion_retains_measured_usage_with_sdk_cost( assert repeated["usage"] == expected_usage +@pytest.mark.parametrize("readable_usage", [False, True]) +def test_fresh_completion_does_not_promote_a_running_cost_estimate( + tmp_path: Path, readable_usage: bool +) -> None: + fixture = _start_scan(tmp_path, mode="deep") + cost = { + "model": "synthetic-model", + "inputTokens": 10, + "cachedInputTokens": 0, + "cacheWriteInputTokens": 0, + "outputTokens": 5, + "estimatedUsd": 0.001, + } + with sqlite3.connect(fixture.state_dir / "workbench.sqlite3") as connection: + connection.execute( + "UPDATE scans SET cost_json = ? WHERE id = ?", (json.dumps(cost), fixture.scan_id) + ) + if readable_usage: + _state_graph( + fixture.environment, + { + "scan-parent": _rollout( + tmp_path, + "scan-parent", + [_token_event(fixture.started_at + timedelta(microseconds=1), 30, 10)], + ) + }, + [], + ) + completed = _complete_scan(fixture)["scan"] + assert "cost" not in completed + assert completed["usage"]["coverage"] == ("complete" if readable_usage else "unavailable") + if readable_usage: + assert completed["usage"]["totalTokens"] == 40 + with sqlite3.connect(fixture.state_dir / "workbench.sqlite3") as connection: + receipt = connection.execute( + "SELECT cost_json FROM scans WHERE id = ?", (fixture.scan_id,) + ).fetchone()[0] + assert json.loads(receipt) == {"usage": completed["usage"]} + + def test_usage_is_returned_by_completion_without_an_extra_command(tmp_path: Path) -> None: fixture = _start_scan(tmp_path) counted = fixture.started_at + timedelta(microseconds=1) diff --git a/plugins/codex-security/tests/test_workbench_standard_deep_results.py b/plugins/codex-security/tests/test_workbench_standard_deep_results.py index 4c86e3e8d..f29c0dcb2 100644 --- a/plugins/codex-security/tests/test_workbench_standard_deep_results.py +++ b/plugins/codex-security/tests/test_workbench_standard_deep_results.py @@ -8,24 +8,25 @@ import subprocess import sys import uuid -from datetime import datetime, timedelta, timezone from pathlib import Path import pytest -from workbench_test_support import run_workbench, write_checkpoint, write_completed_contract +from workbench_test_support import ( + finding_fixture, + run_workbench, + write_checkpoint, + write_completed_contract, +) -@pytest.mark.parametrize("termination", ["failed", "interrupted", "canceled"]) -def test_stopped_deep_scan_ignores_late_worker_checkpoints_without_reducer( +@pytest.mark.parametrize("termination", ["failed", "canceled"]) +def test_stopped_scan_ignores_late_checkpoints_until_explicit_recovery( tmp_path: Path, termination: str, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + result_path = write_checkpoint(scan_dir / "checkpoints", checkpoint_draft(scan_id)) + finding = finding_fixture(relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -43,13 +44,9 @@ def test_stopped_deep_scan_ignores_late_worker_checkpoints_without_reducer( ], }, } - write_checkpoint(result_path.parent / "checkpoints", checkpoint) + write_checkpoint(scan_dir / "checkpoints", checkpoint) # The latest incomplete attempt need not be parseable for a saved checkpoint to survive. result_path.write_text("{incomplete") - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_workers SET status = 'running' WHERE id = ?", (worker_id,) - ) environment = {"CODEX_HOME": str(codex_home)} if termination == "canceled": run_workbench( @@ -62,7 +59,7 @@ def test_stopped_deep_scan_ignores_late_worker_checkpoints_without_reducer( environment=environment, ) else: - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status=termination) + stop_scan(state_dir, scan_id, "Worker stopped.", status=termination) stopped = run_workbench(state_dir, "get-scan", "--scan-id", scan_id[:12])["scan"] assert stopped["progress"]["status"] == ("canceled" if termination == "canceled" else "failed") @@ -78,7 +75,7 @@ def test_stopped_deep_scan_ignores_late_worker_checkpoints_without_reducer( late = copy.deepcopy(checkpoint) late["findings"][0]["locations"][0]["startLine"] = 91 late["findings"][0]["locations"][0]["endLine"] = 92 - archived = result_path.parent / "attempts" / "attempt-01" / "checkpoints" + archived = scan_dir / "checkpoints" write_checkpoint(archived, late) recovery_needed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert recovery_needed["resultsRecoveryNeeded"] is (termination != "canceled") @@ -126,12 +123,8 @@ def test_stopped_deep_scan_ignores_late_worker_checkpoints_without_reducer( def test_scan_reads_require_explicit_late_result_recovery(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + finding = finding_fixture(relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -143,16 +136,14 @@ def test_scan_reads_require_explicit_late_result_recovery(tmp_path: Path) -> Non "deferred": [], }, } - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + write_checkpoint(scan_dir / "checkpoints", checkpoint) + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") stopped = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert stopped["resultsRecoveryNeeded"] is False late = copy.deepcopy(checkpoint) late["findings"][0]["locations"][0]["startLine"] = 91 late["findings"][0]["locations"][0]["endLine"] = 92 - late_path = write_checkpoint( - result_path.parent / "attempts" / "attempt-01" / "checkpoints", late - ) + late_path = write_checkpoint(scan_dir / "checkpoints", late) manifest_path = scan_dir / "scan-manifest.json" published_after_checkpoint = late_path.stat().st_mtime_ns + 1_000_000 os.utime(manifest_path, ns=(published_after_checkpoint, published_after_checkpoint)) @@ -191,12 +182,8 @@ def test_scan_reads_require_explicit_late_result_recovery(tmp_path: Path) -> Non def test_explicit_recovery_rejects_changed_frozen_source(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + finding = finding_fixture(relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -208,8 +195,8 @@ def test_explicit_recovery_rejects_changed_frozen_source(tmp_path: Path) -> None "deferred": [], }, } - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + write_checkpoint(scan_dir / "checkpoints", checkpoint) + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") manifest_path = scan_dir / "scan-manifest.json" original_manifest = manifest_path.read_bytes() original_findings = (scan_dir / "findings.json").read_bytes() @@ -225,7 +212,7 @@ def test_explicit_recovery_rejects_changed_frozen_source(tmp_path: Path) -> None late = copy.deepcopy(checkpoint) late["findings"][0]["locations"][0]["startLine"] = 91 late["findings"][0]["locations"][0]["endLine"] = 92 - write_checkpoint(result_path.parent / "attempts" / "attempt-01" / "checkpoints", late) + write_checkpoint(scan_dir / "checkpoints", late) assert ( run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"]["resultsRecoveryNeeded"] is True @@ -257,8 +244,7 @@ def test_explicit_recovery_rejects_changed_frozen_source(tmp_path: Path) -> None def test_explicit_recovery_preserves_unfrozen_parent_with_late_checkpoint( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") @@ -310,10 +296,10 @@ def test_explicit_recovery_preserves_unfrozen_parent_with_late_checkpoint( late_finding["occurrenceId"] = "occ_111111111111111111111111" late_finding["ruleId"] = "late.checkpoint" late_finding["title"] = "Late checkpoint finding" - late = json.loads(result_path.read_text()) + late = checkpoint_draft(scan_id) late["complete"] = False late["findings"] = [late_finding] - write_checkpoint(result_path.parent / "checkpoints", late) + write_checkpoint(scan_dir / "checkpoints", late) recovered = run_workbench( state_dir, @@ -333,9 +319,7 @@ def test_explicit_recovery_preserves_unfrozen_parent_with_late_checkpoint( def test_explicit_recovery_preserves_sealed_parent_with_empty_source_map( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - result_path.unlink() + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() scripts_dir = Path(__file__).resolve().parents[1] / "scripts" @@ -344,7 +328,7 @@ def test_explicit_recovery_preserves_sealed_parent_with_empty_source_map( scan_id, target, relative_path="app.py", - coverage_mode="deep_repository", + coverage_mode="repository", ) subprocess.run( [ @@ -367,7 +351,7 @@ def test_explicit_recovery_preserves_sealed_parent_with_empty_source_map( (f"sha256:{hashlib.sha256(sealed_manifest).hexdigest()}", scan_id), ) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") assert ( json.loads((scan_dir / "scan-manifest.json").read_text())["scan"]["preservedSources"] == {} ) @@ -388,7 +372,7 @@ def test_explicit_recovery_preserves_sealed_parent_with_empty_source_map( "deferred": [], }, } - write_checkpoint(result_path.parent / "checkpoints", late) + write_checkpoint(scan_dir / "checkpoints", late) manifest_before_failed_recovery = (scan_dir / "scan-manifest.json").read_bytes() findings_before_failed_recovery = (scan_dir / "findings.json").read_bytes() @@ -449,7 +433,7 @@ def test_explicit_recovery_preserves_sealed_parent_with_empty_source_map( later_finding["title"] = "Later checkpoint finding" later = copy.deepcopy(late) later["findings"] = [later_finding] - write_checkpoint(result_path.parent / "attempts" / "attempt-02" / "checkpoints", later) + write_checkpoint(scan_dir / "checkpoints", later) recovered_again = run_workbench( state_dir, @@ -476,8 +460,7 @@ def test_explicit_recovery_retries_frozen_parent_after_write_failure( tmp_path: Path, published_sources: dict[str, str] | None, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() scripts_dir = Path(__file__).resolve().parents[1] / "scripts" @@ -486,7 +469,7 @@ def test_explicit_recovery_retries_frozen_parent_after_write_failure( scan_id, target, relative_path="app.py", - coverage_mode="deep_repository", + coverage_mode="repository", ) subprocess.run( [ @@ -519,10 +502,10 @@ def test_explicit_recovery_retries_frozen_parent_after_write_failure( late_finding["identity"]["anchor"] = "late-checkpoint" late_finding["ruleId"] = "late.checkpoint" late_finding["title"] = "Late checkpoint finding" - late = json.loads(result_path.read_text()) + late = checkpoint_draft(scan_id) late["complete"] = False late["findings"] = [late_finding] - result_path.write_text(json.dumps(late)) + write_checkpoint(scan_dir / "checkpoints", late) with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: connection.execute( "UPDATE scans SET seal_manifest_digest = ? WHERE id = ?", @@ -613,7 +596,7 @@ def test_explicit_recovery_retries_frozen_parent_after_write_failure( def test_unsealed_manifest_without_saved_results_does_not_offer_recovery( tmp_path: Path, ) -> None: - state_dir, _, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) + state_dir, _, _, scan_dir, scan_id = scan_fixture(tmp_path) (scan_dir / "scan-manifest.json").write_text(json.dumps({"scan": {"status": "failed"}})) with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: connection.execute( @@ -640,12 +623,8 @@ def test_aggregate_queries_ignore_late_stopped_scan_checkpoints( collection: str, count_field: str | None, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + finding = finding_fixture(relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -657,11 +636,11 @@ def test_aggregate_queries_ignore_late_stopped_scan_checkpoints( "deferred": [], }, } - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + write_checkpoint(scan_dir / "checkpoints", checkpoint) + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") late = copy.deepcopy(checkpoint) late["findings"][0]["identity"]["anchor"] = "late-independent-finding" - write_checkpoint(result_path.parent / "attempts" / "attempt-01" / "checkpoints", late) + write_checkpoint(scan_dir / "checkpoints", late) rows = run_workbench(state_dir, command, environment={"CODEX_HOME": str(codex_home)})[ collection @@ -680,55 +659,62 @@ def test_aggregate_queries_ignore_late_stopped_scan_checkpoints( def test_unreadable_only_checkpoint_records_recovery_warning(tmp_path: Path) -> None: - state_dir, codex_home, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, _, scan_dir, scan_id = scan_fixture(tmp_path) + result_path = write_checkpoint(scan_dir / "checkpoints", checkpoint_draft(scan_id)) result_path.write_text("{not-json") - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert any("Preserved unreadable checkpoint" in warning for warning in failed["warnings"]) -def test_malformed_current_finding_does_not_override_worker_rejection(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) +def test_malformed_current_finding_retains_parent_rejection_history(tmp_path: Path) -> None: + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] finding["provenance"]["candidateId"] = "rejected-candidate" - checkpoint = json.loads(result_path.read_text()) + checkpoint = checkpoint_draft(scan_id) checkpoint["complete"] = False checkpoint["findings"] = [copy.deepcopy(finding)] - write_checkpoint(result_path.parent / "checkpoints", checkpoint) + write_checkpoint(scan_dir / "checkpoints", checkpoint) finding["summary"] = "" - current = json.loads(result_path.read_text()) - current["findings"] = [finding] - current["coverage"]["surfaces"] = [ + coverage = json.loads((contract_dir / "coverage.json").read_text()) + coverage["surfaces"] = [ { + "id": "rejected-candidate", "label": "Rejected candidate", "candidateId": "rejected-candidate", "disposition": "rejected", - "notes": "The completed worker rejected this checkpointed candidate.", + "receiptRefs": [], + "notes": "The parent rejected this checkpointed candidate.", } ] - result_path.write_text(json.dumps(current)) + (scan_dir / "findings.json").write_text(json.dumps({"findings": [finding]})) + (scan_dir / "coverage.json").write_text(json.dumps(coverage)) + (scan_dir / "scan-manifest.json").write_bytes( + (contract_dir / "scan-manifest.json").read_bytes() + ) - stop_legacy_scan( + stop_scan( state_dir, scan_id, "Stopped after rejecting a malformed current finding.", status="failed" ) failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] + assert failed["reportAvailable"] is True + assert failed["resultsRecoveryNeeded"] is False assert failed["findingCount"] == 0 coverage = json.loads((scan_dir / "coverage.json").read_text()) - assert coverage["surfaces"][0]["disposition"] == "rejected" - assert len(coverage["surfaces"][0]["previousFindings"]) == 1 + # Ordinary publication preserves rejection history but flags malformed current evidence. + assert coverage["surfaces"][0]["disposition"] == "needs_follow_up" + assert coverage["surfaces"][0]["previousFindings"] == checkpoint["findings"] + assert any(item["id"] == "discarded-finding-1" for item in coverage["deferred"]) def test_stopped_recovery_accepts_trailing_slash_scope(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract( @@ -740,16 +726,15 @@ def test_stopped_recovery_accepts_trailing_slash_scope(tmp_path: Path) -> None: coverage_mode="scoped_path", inventory_strategy="scoped_path", ) - checkpoint = json.loads(result_path.read_text()) + checkpoint = checkpoint_draft(scan_id) checkpoint["complete"] = False checkpoint["findings"] = json.loads((contract_dir / "findings.json").read_text())["findings"] checkpoint["coverage"] = json.loads((contract_dir / "coverage.json").read_text()) - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - result_path.write_text("{incomplete") + write_checkpoint(scan_dir / "checkpoints", checkpoint) with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: connection.execute("UPDATE scans SET scope = 'src/' WHERE id = ?", (scan_id,)) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert failed["findingCount"] == 1 @@ -759,12 +744,8 @@ def test_stopped_recovery_accepts_trailing_slash_scope(tmp_path: Path) -> None: def test_canceled_scan_retries_failed_publication_from_frozen_sources( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + finding = finding_fixture(relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -776,12 +757,7 @@ def test_canceled_scan_retries_failed_publication_from_frozen_sources( "deferred": [], }, } - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - result_path.write_text("{incomplete") - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_workers SET status = 'running' WHERE id = ?", (worker_id,) - ) + write_checkpoint(scan_dir / "checkpoints", checkpoint) scripts_dir = Path(__file__).resolve().parents[1] / "scripts" wrapper = tmp_path / "fail_canceled_publication.py" @@ -825,7 +801,7 @@ def test_canceled_scan_retries_failed_publication_from_frozen_sources( late = copy.deepcopy(checkpoint) late["findings"][0]["locations"][0]["startLine"] = 91 late["findings"][0]["locations"][0]["endLine"] = 92 - archived = result_path.parent / "attempts" / "attempt-01" / "checkpoints" + archived = scan_dir / "checkpoints" write_checkpoint(archived, late) preserved = run_workbench( @@ -853,9 +829,9 @@ def test_canceled_scan_retries_failed_publication_from_frozen_sources( def test_existing_non_canceled_output_recovers_structured_publication_failure( tmp_path: Path, deep_status: str ) -> None: - state_dir, codex_home, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - stop_legacy_scan(state_dir, scan_id, "Synthetic scan failure", status=deep_status) + state_dir, codex_home, _, scan_dir, scan_id = scan_fixture(tmp_path, legacy=True) + write_checkpoint(scan_dir / "checkpoints", checkpoint_draft(scan_id)) + stop_scan(state_dir, scan_id, "Synthetic scan failure", status=deep_status) with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: connection.execute( "UPDATE deep_scan_runs SET publication_error_message = ? WHERE scan_id = ?", @@ -875,8 +851,8 @@ def test_existing_non_canceled_output_recovers_structured_publication_failure( def test_canceled_scan_reports_noop_coordinator_publication(tmp_path: Path) -> None: - state_dir, codex_home, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, _, scan_dir, scan_id = scan_fixture(tmp_path) + result_path = write_checkpoint(scan_dir / "checkpoints", checkpoint_draft(scan_id)) result_path.write_text("{incomplete") scripts_dir = Path(__file__).resolve().parents[1] / "scripts" @@ -935,14 +911,13 @@ def test_canceled_scan_reports_noop_coordinator_publication(tmp_path: Path) -> N def test_canceled_scan_reseals_prepared_completion_with_frozen_sources( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - result = json.loads(result_path.read_text()) + result = checkpoint_draft(scan_id) result["findings"] = json.loads((contract_dir / "findings.json").read_text())["findings"] - result_path.write_text(json.dumps(result)) + result_path = write_checkpoint(scan_dir / "checkpoints", result) for filename in ("findings.json", "coverage.json", "scan-manifest.json"): (scan_dir / filename).write_bytes((contract_dir / filename).read_bytes()) manifest_path = scan_dir / "scan-manifest.json" @@ -960,13 +935,6 @@ def test_canceled_scan_reseals_prepared_completion_with_frozen_sources( ).hexdigest() } manifest_path.write_text(json.dumps(manifest)) - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET status = 'succeeded', phase = 'terminal', " - "terminal_reason = 'saturated', manifest_path = ?, completed_at = updated_at " - "WHERE scan_id = ?", - (str(manifest_path), scan_id), - ) run_workbench(state_dir, "prepare-scan-completion", "--scan-id", scan_id) assert json.loads(manifest_path.read_text())["scan"]["status"] == "completed" @@ -1020,21 +988,20 @@ def test_canceled_scan_reseals_prepared_completion_with_frozen_sources( ).fetchone()[0] -def test_stopped_deep_scan_recovers_when_parent_manifest_has_no_scan(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) +def test_stopped_scan_recovers_when_parent_manifest_has_no_scan(tmp_path: Path) -> None: + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - result = json.loads(result_path.read_text()) + result = checkpoint_draft(scan_id) result["findings"] = [finding] - result_path.write_text(json.dumps(result)) + write_checkpoint(scan_dir / "checkpoints", result) (scan_dir / "findings.json").write_bytes((contract_dir / "findings.json").read_bytes()) (scan_dir / "coverage.json").write_bytes((contract_dir / "coverage.json").read_bytes()) (scan_dir / "scan-manifest.json").write_text(json.dumps({"documentType": "broken-parent"})) - stop_legacy_scan(state_dir, scan_id, "Worker stopped.", status="failed") + stop_scan(state_dir, scan_id, "Worker stopped.", status="failed") stopped = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert stopped["findingCount"] == 1 @@ -1042,7 +1009,20 @@ def test_stopped_deep_scan_recovers_when_parent_manifest_has_no_scan(tmp_path: P assert json.loads((scan_dir / "scan-manifest.json").read_text())["scan"]["status"] == ("failed") -def deep_scan_fixture(tmp_path: Path, *, workers: int = 1, budget: bool = False): +def checkpoint_draft(scan_id: str) -> dict: + return { + "scanId": scan_id, + "findings": [], + "coverage": { + "completeness": "complete", + "surfaces": [], + "explicitExclusions": [], + "deferred": [], + }, + } + + +def scan_fixture(tmp_path: Path, *, legacy: bool = False): state_dir, codex_home, target = tmp_path / "state", tmp_path / "codex-home", tmp_path / "target" target.mkdir() (target / "app.py").write_text("# Synthetic legacy scan target\n") @@ -1059,10 +1039,9 @@ def deep_scan_fixture(tmp_path: Path, *, workers: int = 1, budget: bool = False) json.dumps( { "config": {}, - "mode": "deep", + "mode": "deep" if legacy else "standard", "repository": str(target), "target": {"kind": "repository", "paths": []}, - **({"maxCostUsd": 0.005} if budget else {}), } ), ) @@ -1070,36 +1049,19 @@ def deep_scan_fixture(tmp_path: Path, *, workers: int = 1, budget: bool = False) run_workbench( state_dir, "set-scan-thread", "--scan-id", scan_id, "--thread-id", "standard-worker-thread" ) - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "INSERT INTO deep_scan_runs (scan_id,schema_version,workflow_version,status,phase,workers," - "subagents,stop_after_no_new,max_discovery_runs,created_at,updated_at) " - "SELECT id,1,'deep-security-scan/v1','running','discovery',?,3,4,8,started_at,updated_at " - "FROM scans WHERE id = ?", - (workers, scan_id), - ) + if legacy: + with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: + connection.execute( + "INSERT INTO deep_scan_runs (scan_id,schema_version,workflow_version,status,phase,workers," + "subagents,stop_after_no_new,max_discovery_runs,created_at,updated_at) " + "SELECT id,1,'deep-security-scan/v1','running','discovery',?,3,4,8,started_at,updated_at " + "FROM scans WHERE id = ?", + (1, scan_id), + ) return state_dir, codex_home, target, scan_dir, scan_id -def seed_legacy_worker(state_dir, scan_id, worker_id, *, kind, status, prompt, directory): - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "INSERT INTO deep_scan_workers (id,scan_id,kind,status,prompt_path,artifact_dir,attempt," - "result_manifest_path,created_at,updated_at,completed_at) " - "SELECT ?,id,?,?,?,?,1,?,started_at,updated_at,updated_at FROM scans WHERE id = ?", - ( - worker_id, - kind, - status, - str(prompt), - str(directory), - str(Path(directory) / "result.json"), - scan_id, - ), - ) - - -def stop_legacy_scan(state_dir, scan_id, message, *, status="failed"): +def stop_scan(state_dir, scan_id, message, *, status="failed"): with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: connection.execute( "UPDATE deep_scan_runs SET status = ?, error_message = ? WHERE scan_id = ?", @@ -1108,166 +1070,19 @@ def stop_legacy_scan(state_dir, scan_id, message, *, status="failed"): return run_workbench(state_dir, "fail-scan", "--scan-id", scan_id, "--message", message) -def worker_paths(scan_dir: Path, name: str) -> tuple[Path, Path, Path]: - artifact_dir = scan_dir / "artifacts" / "deep_discovery" / name - artifact_dir.mkdir(parents=True) - prompt_path = artifact_dir / "prompt.md" - prompt_path.write_text(f"Prompt for {name}\n") - return prompt_path, artifact_dir, artifact_dir / "result.json" - - -def accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id, *, name="standard-worker"): - worker_id = str(uuid.uuid4()) - prompt_path, artifact_dir, result_path = worker_paths(scan_dir, name) - result_path.write_text( - json.dumps( - { - "scanId": scan_id, - "findings": [], - "coverage": { - "completeness": "complete", - "surfaces": [], - "explicitExclusions": [], - "deferred": [], - }, - "threatModel": {"summary": "The ordinary Standard worker threat model."}, - } - ) - ) - seed_legacy_worker( - state_dir, - scan_id, - worker_id, - kind="discovery", - status="succeeded", - prompt=prompt_path, - directory=artifact_dir, - ) - return worker_id, result_path - - -def committed_standard_reducer( - state_dir, - codex_home, - scan_dir, - scan_id, - discovery_worker_id, - discovery_result, - *, - additional_worker_ids=(), -): - reducer_id = str(uuid.uuid4()) - prompt_path, artifact_dir, result_path = worker_paths(scan_dir, "standard-reducer") - result_path.write_text(discovery_result.read_text()) - seed_legacy_worker( - state_dir, - scan_id, - reducer_id, - kind="dedup", - status="succeeded", - prompt=prompt_path, - directory=artifact_dir, - ) - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - for ordinal, worker_id in enumerate((discovery_worker_id, *additional_worker_ids)): - connection.execute( - "INSERT INTO deep_scan_dedup_inputs (scan_id,dedup_worker_id,discovery_worker_id,input_order) VALUES (?,?,?,?)", - (scan_id, reducer_id, worker_id, ordinal), - ) - connection.execute( - "UPDATE deep_scan_workers SET merge_state = 'merged' WHERE id = ?", (worker_id,) - ) - return reducer_id, result_path, {} - - -def test_failure_preserves_last_committed_reducer_without_parent_draft(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - draft = json.loads(result_path.read_text()) - draft["findings"] = json.loads((contract_dir / "findings.json").read_text())["findings"] - result_path.write_text(json.dumps(draft)) - _, reducer_path, _ = committed_standard_reducer( - state_dir, codex_home, scan_dir, scan_id, worker_id, result_path - ) - reduced = json.loads(reducer_path.read_text()) - reduced["findings"][0]["summary"] = ( - "The reducer retained additional independently reviewed evidence." - ) - reducer_path.write_text(json.dumps(reduced)) - stop_legacy_scan(state_dir, scan_id, "Later reducer failed.", status="failed") - failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] - assert failed["progress"]["status"] == "failed" - assert failed["findingCount"] == 1 - assert failed["findings"][0]["summary"] == reduced["findings"][0]["summary"] - - -def test_stopped_rejection_recovers_malformed_parent_surfaces(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - finding["extensions"] = {"candidateId": "rejected-candidate"} - finding["provenance"]["candidateId"] = "rejected-candidate" - finding["provenance"]["workerId"] = worker_id - (scan_dir / "findings.json").write_text(json.dumps({"scanId": scan_id, "findings": [finding]})) - malformed_coverage = json.loads((contract_dir / "coverage.json").read_text()) - malformed_coverage["surfaces"] = None - (scan_dir / "coverage.json").write_text(json.dumps(malformed_coverage)) - (scan_dir / "scan-manifest.json").write_bytes( - (contract_dir / "scan-manifest.json").read_bytes() - ) - current = json.loads(result_path.read_text()) - checkpoint = copy.deepcopy(current) - checkpoint["complete"] = False - checkpoint["findings"] = [copy.deepcopy(finding)] - write_checkpoint(result_path.parent / "checkpoints", checkpoint) - current["coverage"]["surfaces"] = [ - { - "label": "Rejected candidate", - "candidateId": "rejected-candidate", - "disposition": "rejected", - "notes": "The completed worker rejected this checkpointed candidate.", - } - ] - result_path.write_text(json.dumps(current)) - - stop_legacy_scan( - state_dir, scan_id, "Stopped after rejecting a checkpointed candidate.", status="failed" - ) - - failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] - assert failed["progress"]["status"] == "failed" - assert failed["findingCount"] == 1 - coverage = json.loads((scan_dir / "coverage.json").read_text()) - assert isinstance(coverage["surfaces"], list) - - def test_stopped_scan_rebinds_prepared_completion_seal(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - result = json.loads(result_path.read_text()) + result = checkpoint_draft(scan_id) result["findings"] = json.loads((contract_dir / "findings.json").read_text())["findings"] - result_path.write_text(json.dumps(result)) + write_checkpoint(scan_dir / "checkpoints", result) (scan_dir / "findings.json").write_bytes((contract_dir / "findings.json").read_bytes()) (scan_dir / "coverage.json").write_bytes((contract_dir / "coverage.json").read_bytes()) (scan_dir / "scan-manifest.json").write_bytes( (contract_dir / "scan-manifest.json").read_bytes() ) - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET status = 'succeeded', phase = 'terminal', " - "terminal_reason = 'saturated', manifest_path = ?, completed_at = updated_at " - "WHERE scan_id = ?", - (str(scan_dir / "scan-manifest.json"), scan_id), - ) run_workbench(state_dir, "prepare-scan-completion", "--scan-id", scan_id) prepared_manifest = json.loads((scan_dir / "scan-manifest.json").read_text()) assert prepared_manifest["scan"]["status"] == "completed" @@ -1306,15 +1121,11 @@ def test_stopped_scan_rebinds_prepared_completion_seal(tmp_path: Path) -> None: def test_stopped_findings_cannot_enter_remediation( tmp_path: Path, command: tuple[str, ...] ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - result = json.loads(result_path.read_text()) - result["findings"] = json.loads((contract_dir / "findings.json").read_text())["findings"] - result_path.write_text(json.dumps(result)) - stop_legacy_scan(state_dir, scan_id, "Stopped with a provisional finding.", status="failed") + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + result = checkpoint_draft(scan_id) + result["findings"] = [finding_fixture(relative_path="app.py")] + write_checkpoint(scan_dir / "checkpoints", result) + stop_scan(state_dir, scan_id, "Stopped with a provisional finding.", status="failed") failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] assert failed["remediationAvailable"] is False assert failed["remediationUnavailableReason"] == ( @@ -1338,9 +1149,9 @@ def test_stopped_findings_cannot_enter_remediation( assert "successfully completed scans" in blocked["stderr"] -def test_complete_worker_supersedes_obsolete_checkpoint_coverage(tmp_path: Path) -> None: - state_dir, codex_home, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - _, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) +def test_complete_parent_supersedes_obsolete_checkpoint_coverage(tmp_path: Path) -> None: + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + write_completed_contract(scan_dir, scan_id, target, relative_path="app.py") checkpoint = { "scanId": scan_id, "complete": False, @@ -1359,11 +1170,11 @@ def test_complete_worker_supersedes_obsolete_checkpoint_coverage(tmp_path: Path) "deferred": [{"id": "obsolete-work", "reason": "This was later completed."}], }, } - checkpoints = result_path.parent / "checkpoints" + checkpoints = scan_dir / "checkpoints" checkpoints.mkdir() (checkpoints / ("0" * 64 + ".json")).write_text(json.dumps(checkpoint)) - stop_legacy_scan(state_dir, scan_id, "Stopped after the worker completed.", status="failed") + stop_scan(state_dir, scan_id, "Stopped after the parent draft completed.", status="failed") coverage = json.loads((scan_dir / "coverage.json").read_text()) assert not any(item.get("id") == "obsolete-surface" for item in coverage["surfaces"]) @@ -1373,13 +1184,13 @@ def test_complete_worker_supersedes_obsolete_checkpoint_coverage(tmp_path: Path) def test_complete_partial_parent_supersedes_obsolete_checkpoint_questions( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) write_completed_contract( scan_dir, scan_id, target, relative_path="app.py", - coverage_mode="deep_repository", + coverage_mode="repository", ) coverage_path = scan_dir / "coverage.json" final_coverage = json.loads(coverage_path.read_text()) @@ -1402,118 +1213,15 @@ def test_complete_partial_parent_supersedes_obsolete_checkpoint_questions( }, ) - stop_legacy_scan( - state_dir, scan_id, "Stopped after the final partial parent draft.", status="failed" - ) + stop_scan(state_dir, scan_id, "Stopped after the final partial parent draft.", status="failed") recovered = json.loads(coverage_path.read_text()) assert recovered.get("openQuestions", []) == [] -def test_canceled_reducer_checkpoint_supersedes_discovery_result(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, worker_result = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - baseline = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - baseline["extensions"] = {"candidateId": "candidate-reducer"} - baseline["provenance"]["candidateId"] = "candidate-reducer" - discovery = json.loads(worker_result.read_text()) - discovery["findings"] = [baseline] - discovery["coverage"]["surfaces"] = [ - { - "id": "reducer-surface", - "label": "Reducer-reviewed route", - "disposition": "reported", - "notes": "Discovery evidence only.", - "receiptRefs": [], - } - ] - worker_result.write_text(json.dumps(discovery)) - - reducer_id = str(uuid.uuid4()) - prompt_path, artifact_dir, reducer_result = worker_paths(scan_dir, "canceled-reducer") - seed_legacy_worker( - state_dir, - scan_id, - reducer_id, - kind="dedup", - status="running", - prompt=str(prompt_path), - directory=str(artifact_dir), - ) - reduced = copy.deepcopy(discovery) - reduced["findings"][0]["summary"] = "The reducer retained stronger merged evidence." - reduced["coverage"]["surfaces"][0]["notes"] = "Reducer-validated merged evidence." - reducer_result.write_text(json.dumps(reduced)) - checkpoints = reducer_result.parent / "checkpoints" - checkpoints.mkdir() - (checkpoints / ("a" * 64 + ".json")).write_text(json.dumps(reduced)) - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_workers SET status = 'canceled', completed_at = ? WHERE id = ?", - (datetime.now(timezone.utc).isoformat(), reducer_id), - ) - - stop_legacy_scan(state_dir, scan_id, "Canceled after reducer validation.", status="failed") - - findings = json.loads((scan_dir / "findings.json").read_text())["findings"] - coverage = json.loads((scan_dir / "coverage.json").read_text()) - assert findings[0]["summary"] == reduced["findings"][0]["summary"] - assert coverage["surfaces"][0]["notes"] == "Reducer-validated merged evidence." - - -def test_archived_reducer_checkpoint_supersedes_discovery_result(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, worker_result = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - discovery = json.loads(worker_result.read_text()) - discovery["findings"] = [finding] - worker_result.write_text(json.dumps(discovery)) - - reducer_id = str(uuid.uuid4()) - prompt_path, artifact_dir, reducer_result = worker_paths(scan_dir, "archived-reducer") - seed_legacy_worker( - state_dir, - scan_id, - reducer_id, - kind="dedup", - status="running", - prompt=str(prompt_path), - directory=str(artifact_dir), - ) - reduced = copy.deepcopy(discovery) - reduced["findings"][0]["summary"] = "The archived reducer retained the newest evidence." - archived = artifact_dir / "attempts" / "attempt-01" - archived.mkdir(parents=True) - (archived / "result.json").write_text(json.dumps(reduced)) - write_checkpoint(archived / "checkpoints", reduced) - reducer_result.write_text("{incomplete current reducer") - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_workers SET status = 'canceled', completed_at = ? WHERE id = ?", - (datetime.now(timezone.utc).isoformat(), reducer_id), - ) - - stop_legacy_scan( - state_dir, scan_id, "Canceled after archiving a validated reducer attempt.", status="failed" - ) - - findings = json.loads((scan_dir / "findings.json").read_text())["findings"] - assert findings[0]["summary"] == reduced["findings"][0]["summary"] - - def test_recovery_selects_strongest_same_finding_checkpoint(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result_path = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - weak = json.loads((contract_dir / "findings.json").read_text())["findings"][0] + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) + weak = finding_fixture(relative_path="app.py") weak["severity"]["level"] = "low" weak["confidence"]["level"] = "low" weak["summary"] = "Earlier weak checkpoint evidence." @@ -1521,8 +1229,8 @@ def test_recovery_selects_strongest_same_finding_checkpoint(tmp_path: Path) -> N strong["severity"]["level"] = "high" strong["confidence"]["level"] = "high" strong["summary"] = "Later strong checkpoint evidence." - checkpoint_dir = result_path.parent / "checkpoints" - checkpoint_dir.mkdir() + checkpoint_dir = scan_dir / "checkpoints" + checkpoint_dir.mkdir(exist_ok=True) for name, finding in (("0" * 64, weak), ("f" * 64, strong)): (checkpoint_dir / f"{name}.json").write_text( json.dumps( @@ -1539,13 +1247,8 @@ def test_recovery_selects_strongest_same_finding_checkpoint(tmp_path: Path) -> N } ) ) - result_path.write_text("{incomplete") - with sqlite3.connect(state_dir / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_workers SET status = 'running' WHERE id = ?", (worker_id,) - ) - stop_legacy_scan(state_dir, scan_id, "Stopped between checkpoints.", status="failed") + stop_scan(state_dir, scan_id, "Stopped between checkpoints.", status="failed") retained = json.loads((scan_dir / "findings.json").read_text())["findings"][0] assert retained["severity"]["level"] == "high" @@ -1557,72 +1260,10 @@ def test_recovery_selects_strongest_same_finding_checkpoint(tmp_path: Path) -> N ) -def test_failed_reducer_preserves_later_successful_worker_findings(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path, workers=3) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - baseline = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - - first_worker_id, first_result = accepted_standard_worker( - state_dir, codex_home, scan_dir, scan_id, name="first-worker" - ) - first_document = json.loads(first_result.read_text()) - first_document["findings"] = [baseline] - first_result.write_text(json.dumps(first_document)) - empty_worker_id, _ = accepted_standard_worker( - state_dir, codex_home, scan_dir, scan_id, name="empty-worker" - ) - committed_standard_reducer( - state_dir, - codex_home, - scan_dir, - scan_id, - first_worker_id, - first_result, - additional_worker_ids=(empty_worker_id,), - ) - - second_worker_id, second_result = accepted_standard_worker( - state_dir, codex_home, scan_dir, scan_id, name="later-worker" - ) - later = copy.deepcopy(baseline) - later["identity"]["anchor"] = "later-successful-worker-finding" - later["title"] = "Later successful worker finding" - later["summary"] = "This finding completed after the last successful reduction." - second_document = json.loads(second_result.read_text()) - second_document["findings"] = [later] - second_result.write_text(json.dumps(second_document)) - - reducer_id = str(uuid.uuid4()) - prompt_path, artifact_dir, _ = worker_paths(scan_dir, "failed-reducer") - seed_legacy_worker( - state_dir, - scan_id, - reducer_id, - kind="dedup", - status="failed", - prompt=prompt_path, - directory=artifact_dir, - ) - - stop_legacy_scan(state_dir, scan_id, "Later reducers failed.", status="failed") - - failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] - assert failed["progress"]["status"] == "failed" - assert failed["findingCount"] == 2 - assert {finding["identity"]["anchor"] for finding in failed["findings"]} == { - baseline["identity"]["anchor"], - later["identity"]["anchor"], - } - assert json.loads((scan_dir / "coverage.json").read_text())["completeness"] == ("partial") - - def test_recovery_does_not_promote_already_retained_historical_finding( tmp_path: Path, ) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, worker_result = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) + state_dir, codex_home, target, scan_dir, scan_id = scan_fixture(tmp_path) contract_dir = tmp_path / "contract" contract_dir.mkdir() write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") @@ -1645,20 +1286,15 @@ def test_recovery_does_not_promote_already_retained_historical_finding( current["severity"]["level"] = "medium" current["confidence"]["level"] = "medium" current["provenance"]["previousFindings"] = [copy.deepcopy(historical)] - source_finding = copy.deepcopy(current) - source_finding["provenance"].pop("sourceFindings", None) - current["provenance"]["sourceFindings"] = [{"id": f"{worker_id}:0", "finding": source_finding}] - - worker_document = json.loads(worker_result.read_text()) - worker_document["findings"] = [current] - worker_result.write_text(json.dumps(worker_document)) - checkpoint = copy.deepcopy(worker_document) + (scan_dir / "findings.json").write_text(json.dumps({"findings": [current]})) + for filename in ("coverage.json", "scan-manifest.json"): + (scan_dir / filename).write_bytes((contract_dir / filename).read_bytes()) + checkpoint = checkpoint_draft(scan_id) checkpoint["complete"] = False checkpoint["findings"] = [checkpoint_historical] - write_checkpoint(worker_result.parent / "checkpoints", checkpoint) - committed_standard_reducer(state_dir, codex_home, scan_dir, scan_id, worker_id, worker_result) + write_checkpoint(scan_dir / "checkpoints", checkpoint) - stop_legacy_scan( + stop_scan( state_dir, scan_id, "Stopped after the canonical result was retained.", status="failed" ) @@ -1671,615 +1307,26 @@ def test_recovery_does_not_promote_already_retained_historical_finding( assert retained["provenance"]["previousFindings"] == [historical] -def test_recovery_retains_same_worker_checkpoint_version_as_history( - tmp_path: Path, -) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, worker_result = accepted_standard_worker(state_dir, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - checkpoint_finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - checkpoint_finding["extensions"] = {"candidateId": "candidate-refined-location"} - checkpoint_finding["provenance"]["candidateId"] = "candidate-refined-location" - checkpoint_finding["locations"][0]["startLine"] = 1 - checkpoint_finding["locations"][0]["endLine"] = 2 - checkpoint_finding.pop("identity") - checkpoint_finding["provenance"]["previousFindings"] = [ - None, - "malformed checkpoint history", - ] - - current = copy.deepcopy(checkpoint_finding) - current["locations"][0]["startLine"] = 2 - current["provenance"]["previousFindings"] = [17] - source_finding = copy.deepcopy(current) - source_finding["provenance"].pop("sourceFindings", None) - current["provenance"]["sourceFindings"] = [{"id": f"{worker_id}:0", "finding": source_finding}] - - worker_document = json.loads(worker_result.read_text()) - worker_document["findings"] = [current] - worker_result.write_text(json.dumps(worker_document)) - checkpoint = copy.deepcopy(worker_document) - checkpoint["complete"] = False - checkpoint["findings"] = [checkpoint_finding] - write_checkpoint(worker_result.parent / "checkpoints", checkpoint) - committed_standard_reducer(state_dir, codex_home, scan_dir, scan_id, worker_id, worker_result) - - stop_legacy_scan( - state_dir, scan_id, "Stopped after the canonical result was retained.", status="failed" - ) - - failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] - assert failed["findingCount"] == 1 - retained = json.loads((scan_dir / "findings.json").read_text())["findings"][0] - assert retained["locations"][0]["startLine"] == 2 - assert retained["identity"] == {"anchor": "candidate-refined-location"} - expected_checkpoint = copy.deepcopy(checkpoint_finding) - expected_checkpoint["provenance"].pop("previousFindings") - assert retained["provenance"]["previousFindings"] == [expected_checkpoint] - - -def test_independent_worker_candidate_ids_do_not_share_rejection(tmp_path: Path) -> None: - state_dir, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path, workers=2) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - finding["extensions"] = {"candidateId": "candidate-1"} - for ordinal, name in enumerate(("rejecting", "reporting"), 1): - prompt, output, result = worker_paths(scan_dir, name) - seed_legacy_worker( - state_dir, - scan_id, - f"00000000-0000-4000-8000-{ordinal:012}", - kind="discovery", - status="running", - prompt=str(prompt), - directory=str(output), - ) - result.write_text( - json.dumps( - { - "scanId": scan_id, - "findings": [finding] if name == "reporting" else [], - "coverage": { - "completeness": "complete", - "surfaces": [] - if name == "reporting" - else [ - { - "label": "Safe route", - "candidateId": "candidate-1", - "disposition": "rejected", - "notes": "This route enforces containment.", - } - ], - "explicitExclusions": [], - "deferred": [], - }, - } - ) - ) - stop_legacy_scan(state_dir, scan_id, "Stopped.", status="failed") - failed = run_workbench(state_dir, "get-scan", "--scan-id", scan_id)["scan"] - assert failed["findingCount"] == 1 - canonical_findings = json.loads((scan_dir / "findings.json").read_text())["findings"] - assert canonical_findings[0]["extensions"]["candidateId"] == "candidate-1" - coverage = json.loads((scan_dir / "coverage.json").read_text()) - rejected = next(item for item in coverage["surfaces"] if item["disposition"] == "rejected") - assert "previousFindings" not in rejected - - -@pytest.mark.parametrize( - ("completeness", "unreadable_checkpoint"), - [("complete", False), ("partial", False), ("complete", True)], - ids=["complete", "partial", "checkpoint-warning"], -) -def test_succeeded_legacy_resume_preserves_parent_coverage( - tmp_path: Path, completeness: str, unreadable_checkpoint: bool -) -> None: - state, _, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - write_completed_contract( - scan_dir, scan_id, target, relative_path="app.py", coverage_mode="deep_repository" - ) - original = json.loads((scan_dir / "findings.json").read_text())["findings"][0] - coverage_path = scan_dir / "coverage.json" - coverage = json.loads(coverage_path.read_text()) - coverage["completeness"] = completeness - if completeness == "partial": - coverage["deferred"] = [ - {"id": "pending-review", "reason": "A synthetic surface remains unreviewed."} - ] - coverage_path.write_text(json.dumps(coverage)) - if unreadable_checkpoint: - checkpoints = scan_dir / "checkpoints" - checkpoints.mkdir() - (checkpoints / ("0" * 64 + ".json")).write_text("{incomplete") - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET status = 'succeeded', phase = 'terminal', " - "terminal_reason = 'saturated', manifest_path = ?, completed_at = updated_at " - "WHERE scan_id = ?", - (str(scan_dir / "scan-manifest.json"), scan_id), - ) - - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - - checkpoint = json.loads((scan_dir / "artifacts/deep-scan/checkpoint.json").read_text()) - assert checkpoint["terminalReason"] == "saturated" - retained = checkpoint["aggregate"]["findings"] - assert len(retained) == 1 - assert retained[0]["identity"] == original["identity"] - assert retained[0]["locations"] == original["locations"] - migrated = checkpoint["aggregate"]["coverage"] - assert migrated == checkpoint["legacy"]["coverage"] - assert migrated["completeness"] == ("partial" if unreadable_checkpoint else completeness) - assert not any(item.get("id") == "scan-stopped" for item in migrated["deferred"]) - for item in coverage["deferred"]: - assert item in migrated["deferred"] - if unreadable_checkpoint: - assert any( - "Preserved unreadable checkpoint" in item["reason"] for item in migrated["deferred"] - ) - - -def test_legacy_resume_imports_accepted_progress_once(tmp_path: Path) -> None: - state, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result = accepted_standard_worker(state, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - original = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - document = json.loads(result.read_text()) - document["findings"] = [original] - result.write_text(json.dumps(document)) - _, reducer, _ = committed_standard_reducer( - state, codex_home, scan_dir, scan_id, worker_id, result - ) - reduced = json.loads(reducer.read_text()) - reduced["findings"][0]["summary"] = "Accepted reducer detail retained during migration." - reducer.write_text(json.dumps(reduced)) - pending_worker_id, pending = accepted_standard_worker( - state, codex_home, scan_dir, scan_id, name="unmerged" - ) - unmerged = json.loads(pending.read_text()) - unmerged["findings"] = [{**original, "identity": {"anchor": "unmerged-finding"}}] - pending.write_text(json.dumps(unmerged)) - snapshots = {path: path.read_bytes() for path in (result, reducer, pending)} - cost = { - "model": "synthetic-model", - "inputTokens": 10, - "cachedInputTokens": 0, - "cacheWriteInputTokens": 0, - "outputTokens": 5, - "estimatedUsd": 0.001, - } - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET discovery_runs_dispatched=3, " - "consecutive_no_new=2, consecutive_errors=1 WHERE scan_id=?", - (scan_id,), - ) - connection.execute("UPDATE scans SET cost_json=? WHERE id=?", (json.dumps(cost), scan_id)) - metadata = run_workbench(state, "get-cli-scan-resume", "--scan-id", scan_id) - assert metadata["threadId"] is None - assert not (scan_dir / "artifacts/deep-scan/checkpoint.json").exists() - with sqlite3.connect(state / "workbench.sqlite3") as connection: - assert ( - connection.execute( - "SELECT continuation_thread_id FROM scans WHERE id=?", (scan_id,) - ).fetchone()[0] - == "standard-worker-thread" - ) - resumed = run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - assert resumed["threadId"] is None - checkpoint_path = scan_dir / "artifacts/deep-scan/checkpoint.json" - checkpoint = json.loads(checkpoint_path.read_text()) - assert checkpoint["legacy"]["originThreadId"] == "standard-worker-thread" - assert checkpoint["legacy"]["discoveryRuns"] == 3 - assert checkpoint["legacy"]["cost"] == cost - assert checkpoint["noNewStreak"] == 2 - assert checkpoint["consecutiveErrors"] == 1 - assert checkpoint["passes"] == checkpoint["mergedScanIds"] == [] - findings = checkpoint["aggregate"]["findings"] - assert len(findings) == 2 - retained = next(finding for finding in findings if finding["identity"] == original["identity"]) - assert retained["identity"] == original["identity"] - assert retained["summary"] == reduced["findings"][0]["summary"] - assert retained["provenance"]["sourceFindings"][0]["finding"]["summary"] == retained["summary"] - unmerged_finding = next( - finding for finding in findings if finding["identity"] == {"anchor": "unmerged-finding"} - ) - assert unmerged_finding["summary"] == original["summary"] - assert unmerged_finding["provenance"]["workerId"] == pending_worker_id - assert not any( - "unmerged" in item["reason"] for item in checkpoint["legacy"]["coverage"]["deferred"] - ) - assert all(path.read_bytes() == contents for path, contents in snapshots.items()) - checkpoint["mergeFailures"] = 2 - run_workbench( - state, - "save-scan-artifact", - "--scan-id", - scan_id, - "--artifact-path", - "artifacts/deep-scan/checkpoint.json", - input_text=json.dumps(checkpoint), - ) - first_checkpoint = checkpoint_path.read_bytes() - run_workbench(state, "set-scan-thread", "--scan-id", scan_id, "--thread-id", "ordinary-merge") - assert ( - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id)["threadId"] - == "ordinary-merge" - ) - assert checkpoint_path.read_bytes() == first_checkpoint - with sqlite3.connect(state / "workbench.sqlite3") as connection: - assert ( - connection.execute( - "SELECT deep_scan_owner_thread_id FROM scans WHERE id=?", (scan_id,) - ).fetchone()[0] - == "standard-worker-thread" - ) - assert connection.execute("SELECT COUNT(*) FROM scans").fetchone()[0] == 1 - - failed = run_workbench( - state, "fail-scan", "--scan-id", scan_id, "--message", "New scan stopped." - )["scan"] - assert failed["progress"]["status"] == "failed" - assert failed["findingCount"] == 2 - assert {finding["identity"]["anchor"] for finding in failed["findings"]} == { - original["identity"]["anchor"], - "unmerged-finding", - } - assert all(path.read_bytes() == contents for path, contents in snapshots.items()) - - -@pytest.mark.parametrize( - ("generation", "status", "error_message", "dispatched", "expected"), - [ - (1, "queued", None, 1, 0), - (2, "running", None, 1, 0), - (1, "canceled", None, 1, 0), - (2, "canceled", "coordinator_shutdown: mcp_transport_closed", 1, 0), - (2, "canceled", None, 1, 1), - (2, "failed", "Worker failed.", 1, 1), - (2, "canceled", "Worker budget exceeded.", 1, 1), - ( - 2, - "canceled", - "coordinator_shutdown_recovered: replacement attempt required", - 0, - 0, - ), - ], - ids=[ - "queued", - "running", - "legacy-shutdown", - "transport-shutdown", - "unmarked-cancel", - "failed", - "budget-stop", - "already-refunded", - ], -) -def test_legacy_resume_refunds_abandoned_discovery_attempts( - tmp_path: Path, - generation: int, - status: str, - error_message: str | None, - dispatched: int, - expected: int, -) -> None: - state, _, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - prompt, directory, _ = worker_paths(scan_dir, "abandoned") - worker_id = str(uuid.uuid4()) - seed_legacy_worker( - state, - scan_id, - worker_id, - kind="discovery", - status=status, - prompt=prompt, - directory=directory, - ) - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET discovery_runs_dispatched = ?, max_discovery_runs = 1, " - "coordinator_generation = ?, updated_at = '2000-01-01T00:00:00Z' WHERE scan_id = ?", - (dispatched, generation, scan_id), - ) - connection.execute( - "UPDATE deep_scan_workers SET error_message = ? WHERE id = ?", - (error_message, worker_id), - ) - - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - - checkpoint_path = scan_dir / "artifacts/deep-scan/checkpoint.json" - checkpoint = json.loads(checkpoint_path.read_text()) - assert checkpoint["legacy"]["discoveryRuns"] == expected - assert checkpoint["aggregate"]["findings"] == [] - assert any( - item.get("id") == f"legacy-{worker_id}" - for item in checkpoint["legacy"]["coverage"]["deferred"] - ) - checkpoint_bytes = checkpoint_path.read_bytes() - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - assert checkpoint_path.read_bytes() == checkpoint_bytes - - -@pytest.mark.parametrize("merge_state", ["buffered", "merging"]) -def test_legacy_resume_at_discovery_cap_retains_unmerged_results( - tmp_path: Path, merge_state: str -) -> None: - state, codex_home, target, scan_dir, scan_id = deep_scan_fixture(tmp_path) - worker_id, result = accepted_standard_worker(state, codex_home, scan_dir, scan_id) - contract_dir = tmp_path / "contract" - contract_dir.mkdir() - write_completed_contract(contract_dir, scan_id, target, relative_path="app.py") - finding = json.loads((contract_dir / "findings.json").read_text())["findings"][0] - rejected = {**finding, "identity": {"anchor": "rejected-history"}} - rejected["extensions"] = {"candidateId": "rejected-candidate"} - document = json.loads(result.read_text()) - write_checkpoint( - result.parent / "checkpoints", - {**document, "complete": False, "findings": [rejected]}, - ) - document["findings"] = [finding] - document["coverage"]["surfaces"] = [ - { - "label": "Rejected candidate", - "candidateId": "rejected-candidate", - "disposition": "rejected", - "receiptRefs": [], - } - ] - result.write_text(json.dumps(document)) - saved_result = result.read_bytes() - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET discovery_runs_dispatched = 1, max_discovery_runs = 1, " - "completion_sequence = 1 WHERE scan_id = ?", - (scan_id,), - ) - connection.execute( - "UPDATE deep_scan_workers SET merge_state = ?, completion_sequence = 1 WHERE id = ?", - (merge_state, worker_id), - ) - - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - - checkpoint_path = scan_dir / "artifacts/deep-scan/checkpoint.json" - checkpoint = json.loads(checkpoint_path.read_text()) - assert checkpoint["legacy"]["discoveryRuns"] == 1 - assert checkpoint["passes"] == [] - aggregate = checkpoint["aggregate"] - assert len(aggregate["findings"]) == 1 - retained = aggregate["findings"][0] - assert retained["identity"] == finding["identity"] - assert retained["provenance"]["workerId"] == worker_id - assert ( - retained["provenance"]["sourceFindings"][0]["finding"]["validation"] - == finding["validation"] - ) - assert any(row["disposition"] == "rejected" for row in aggregate["coverage"]["surfaces"]) - - # With the discovery cap exhausted, the host publishes this migrated aggregate - # without another child scan or merge; exercise the ordinary completion path. - checkpoint["terminalReason"] = "capped" - run_workbench( - state, - "save-scan-artifact", - "--scan-id", - scan_id, - "--artifact-path", - "artifacts/deep-scan/checkpoint.json", - input_text=json.dumps(checkpoint), - ) - write_completed_contract( - scan_dir, scan_id, target, relative_path="app.py", coverage_mode="deep_repository" - ) - (scan_dir / "findings.json").write_text(json.dumps({"findings": aggregate["findings"]})) - coverage_path = scan_dir / "coverage.json" - coverage = {**json.loads(coverage_path.read_text()), **aggregate["coverage"]} - coverage_path.write_text(json.dumps(coverage)) - completed = run_workbench(state, "complete-scan", "--scan-id", scan_id)["scan"] - assert completed["progress"]["status"] == "complete" - assert completed["findingCount"] == 1 - assert completed["findings"][0]["identity"] == finding["identity"] - assert finding["title"] in (scan_dir / "report.md").read_text() - assert result.read_bytes() == saved_result - - -@pytest.mark.parametrize( - ("history", "expected"), - [ - ( - [ - ("dedup", "failed"), - ("discovery", "succeeded"), - ("dedup", "canceled"), - ("dedup", "failed"), - ("discovery", "failed"), - ("dedup", "failed"), - ], - 3, - ), - ( - [ - ("dedup", "failed"), - ("dedup", "queued"), - ("discovery", "canceled"), - ("dedup", "failed"), - ("dedup", "running"), - ], - 2, - ), - ( - [ - ("dedup", "failed"), - ("dedup", "failed"), - ("dedup", "succeeded"), - ("dedup", "failed"), - ("discovery", "succeeded"), - ("dedup", "canceled"), - ], - 1, - ), - ], - ids=["threshold", "under-threshold", "success-resets"], -) -def test_legacy_resume_preserves_merger_failure_streak( - tmp_path: Path, history: list[tuple[str, str]], expected: int -) -> None: - state, _, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - for index, (kind, status) in enumerate(history): - prompt, directory, result = worker_paths(scan_dir, f"history-{index}") - seed_legacy_worker( - state, - scan_id, - f"history-{index}", - kind=kind, - status=status, - prompt=prompt, - directory=directory, - ) - if status == "succeeded": - result.write_text( - json.dumps( - { - "scanId": scan_id, - "findings": [], - "coverage": { - "completeness": "complete", - "surfaces": [], - "explicitExclusions": [], - "deferred": [], - }, - } - ) - ) - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET updated_at='2000-01-01T00:00:00Z', " - "consecutive_no_new=2, consecutive_errors=1, stop_after_consecutive_errors=3 " - "WHERE scan_id=?", - (scan_id,), - ) - connection.execute("UPDATE deep_scan_workers SET attempt=4 WHERE scan_id=?", (scan_id,)) - workers_before = connection.execute( - "SELECT * FROM deep_scan_workers WHERE scan_id=? ORDER BY created_at,id", (scan_id,) - ).fetchall() - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - checkpoint = json.loads((scan_dir / "artifacts/deep-scan/checkpoint.json").read_text()) - assert checkpoint["mergeFailures"] == expected - assert checkpoint["consecutiveErrors"] == 1 - assert checkpoint["noNewStreak"] == 2 - with sqlite3.connect(state / "workbench.sqlite3") as connection: - assert ( - connection.execute( - "SELECT * FROM deep_scan_workers WHERE scan_id=? ORDER BY created_at,id", (scan_id,) - ).fetchall() - == workers_before - ) - assert connection.execute("SELECT status FROM scans WHERE id=?", (scan_id,)).fetchone() == ( - "running", - ) - - -def test_legacy_conversion_crash_does_not_resume_old_conversation(tmp_path: Path) -> None: - state, codex_home, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - scripts = Path(__file__).resolve().parents[1] / "scripts" - wrapper = tmp_path / "interrupt_conversion.py" - wrapper.write_text( - "import sys\n" - f"sys.path.insert(0, {str(scripts)!r})\n" - "import workbench_db, workbench_saved_results\n" - "original = workbench_saved_results.write_scan_local_bytes\n" - "def interrupt(directory, relative, payload):\n" - " if relative == 'artifacts/deep-scan/checkpoint.json':\n" - " raise OSError('injected checkpoint interruption')\n" - " return original(directory, relative, payload)\n" - "workbench_saved_results.write_scan_local_bytes = interrupt\n" - "raise SystemExit(workbench_db.main())\n" - ) - failed = subprocess.run( - [sys.executable, str(wrapper), "get-cli-scan-resume", "--migrate", "--scan-id", scan_id], - env={**os.environ, "CODEX_HOME": str(codex_home), "CODEX_SECURITY_STATE_DIR": str(state)}, - capture_output=True, - text=True, - check=False, - ) - assert failed.returncode != 0 - assert "injected checkpoint interruption" in failed.stderr - assert not (scan_dir / "artifacts/deep-scan/checkpoint.json").exists() +@pytest.mark.parametrize("converted", [False, True]) +def test_retired_runtime_resume_requires_a_fresh_scan(tmp_path: Path, converted: bool) -> None: + state, _, _, scan_dir, scan_id = scan_fixture(tmp_path, legacy=True) + evidence = scan_dir / "saved-evidence.json" + evidence.write_text('{"summary":"Saved legacy work"}') + if converted: + checkpoint = scan_dir / "artifacts/deep-scan/checkpoint.json" + checkpoint.parent.mkdir(parents=True) + checkpoint.write_text(json.dumps({"version": 2, "legacy": {"discoveryRuns": 1}})) + before = evidence.read_bytes() + rejected = run_workbench(state, "get-cli-scan-resume", "--scan-id", scan_id, check=False) + assert "retired runtime. Start a fresh scan" in rejected["stderr"] + assert evidence.read_bytes() == before with sqlite3.connect(state / "workbench.sqlite3") as connection: assert connection.execute( - "SELECT continuation_thread_id,deep_scan_owner_thread_id FROM scans WHERE id=?", - (scan_id,), - ).fetchone() == (None, "standard-worker-thread") - resumed = run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id) - assert resumed["threadId"] is None - assert ( - json.loads((scan_dir / "artifacts/deep-scan/checkpoint.json").read_text())["version"] == 2 - ) - - -@pytest.mark.parametrize("generation", [1, 2]) -def test_legacy_migration_waits_for_existing_owner_lease(tmp_path: Path, generation: int) -> None: - state, _, _, scan_dir, scan_id = deep_scan_fixture(tmp_path) - timestamp = datetime.now(timezone.utc) - expired = (timestamp - timedelta(minutes=5)).isoformat() - prompt, directory, _ = worker_paths(scan_dir, "active") - seed_legacy_worker( - state, - scan_id, - str(uuid.uuid4()), - kind="discovery", - status="running", - prompt=prompt, - directory=directory, - ) - with sqlite3.connect(state / "workbench.sqlite3") as connection: - connection.execute( - "UPDATE deep_scan_runs SET coordinator_generation=?, updated_at=? WHERE scan_id=?", - (generation, timestamp.isoformat() if generation == 1 else expired, scan_id), - ) - heartbeat = scan_dir / f"artifacts/deep_discovery/coordinator-heartbeat-{generation}.json" - if generation == 2: - heartbeat.write_text( - json.dumps({"coordinatorGeneration": generation, "updatedAt": timestamp.isoformat()}) - ) - rejected = run_workbench( - state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id, check=False - ) - assert "previous Deep Scan owner is still active" in rejected["stderr"] - assert not (scan_dir / "artifacts/deep-scan/checkpoint.json").exists() - with sqlite3.connect(state / "workbench.sqlite3") as connection: - assert ( - connection.execute( - "SELECT continuation_thread_id FROM scans WHERE id=?", (scan_id,) - ).fetchone()[0] - == "standard-worker-thread" - ) - connection.execute( - "UPDATE deep_scan_runs SET updated_at=? WHERE scan_id=?", (expired, scan_id) - ) - if generation == 2: - heartbeat.write_text( - json.dumps({"coordinatorGeneration": generation, "updatedAt": expired}) - ) - assert ( - run_workbench(state, "get-cli-scan-resume", "--migrate", "--scan-id", scan_id)["threadId"] - is None - ) - with sqlite3.connect(state / "workbench.sqlite3") as connection: + "SELECT status, continuation_thread_id FROM scans WHERE id = ?", (scan_id,) + ).fetchone() == ("running", "standard-worker-thread") assert connection.execute( - "SELECT status,cancel_requested,coordinator_generation FROM deep_scan_runs WHERE scan_id=?", - (scan_id,), - ).fetchone() == ("interrupted", 1, generation + 1) + "SELECT status FROM deep_scan_runs WHERE scan_id = ?", (scan_id,) + ).fetchone() == ("running",) @pytest.mark.parametrize( @@ -2371,7 +1418,7 @@ def test_merge_saved_results_deduplicates_open_questions( } result = workbench_saved_results.merge_saved_results( - scan_dir, scan_id, binding, [], [], stopped=False, reason="" + scan_dir, scan_id, binding, [], stopped=False, reason="" ) assert result is not None _, _, coverage = result diff --git a/plugins/codex-security/tests/workbench_test_support.py b/plugins/codex-security/tests/workbench_test_support.py index 757f2cbbe..1aeb62aa2 100644 --- a/plugins/codex-security/tests/workbench_test_support.py +++ b/plugins/codex-security/tests/workbench_test_support.py @@ -32,6 +32,65 @@ def write_checkpoint(checkpoint_dir: Path, payload: Any) -> Path: return checkpoint_path +def recipe(target: Path, mode: str = "standard") -> dict: + return { + "repository": str(target), + "target": {"kind": "repository", "paths": []}, + "mode": mode, + "config": {"model": "synthetic-model", "model_reasoning_effort": "high"}, + **({"deepScan": {"maxDiscoveryRuns": 8}} if mode == "deep" else {}), + } + + +def register( + state: Path, target: Path, directory: Path, *, mode="standard", parent=None, role=None, paths=() +) -> dict: + missing = [] + current = directory + while not current.exists(): + missing.append(current) + current = current.parent + for path in reversed(missing): + path.mkdir(mode=0o700) + saved_recipe = recipe(target, mode) + if paths: + saved_recipe["target"] = {"kind": "paths", "paths": list(paths)} + return run_workbench( + state, + "register-cli-scan", + "--repository", + str(target), + "--scan-dir", + str(directory), + "--registration-json-stdin", + *(("--parent-scan-id", parent) if parent else ()), + input_text=json.dumps({"recipe": saved_recipe, "parentScanRole": role}), + ) + + +def checkpoint(state: Path, scan: dict, *, passes=(), merged=(), terminal=None) -> dict: + value = { + "version": 2, + "startedAt": "2026-01-01T00:00:00Z", + "passes": list(passes), + "mergedScanIds": list(merged), + "aggregate": None, + "noNewStreak": 0, + "consecutiveErrors": 0, + **({"terminalReason": terminal} if terminal else {}), + } + run_workbench( + state, + "save-scan-artifact", + "--scan-id", + scan["scanId"], + "--artifact-path", + "artifacts/deep-scan/checkpoint.json", + input_text=json.dumps(value), + ) + return value + + def stable_target_id(target: Path) -> str: digest = hashlib.sha256(f"local-workspace\0{target.resolve()}".encode()).hexdigest() return f"target_sha256_{digest}" @@ -267,6 +326,68 @@ def begin_legacy_scan( return {"deepScan": scan} +def finding_fixture( + *, + relative_path: str = "src/extract.py", + identity_anchor: str = "archive-entry-write-without-containment", +) -> dict[str, Any]: + return { + "ruleId": "path-traversal.archive-extraction", + "identity": {"anchor": identity_anchor}, + "title": "Unsafe archive extraction can escape the output directory", + "summary": "An attacker-controlled path reaches a filesystem write.", + "severity": { + "level": "high", + "rationale": "The reachable write can escape the extraction root.", + }, + "confidence": {"level": "high", "rationale": "Direct source trace."}, + "taxonomy": {"category": "path-traversal", "cwe": ["CWE-22"]}, + "locations": [{"path": relative_path, "startLine": 41, "endLine": 44, "role": "sink"}], + "codeEvidence": [ + { + "id": "archive-write", + "label": "Unchecked archive write", + "path": relative_path, + "startLine": 41, + "endLine": 44, + "language": "python", + "code": "destination.write_bytes(entry.read())", + "explanation": "The destination is written before containment is checked.", + } + ], + "validation": { + "method": "archive extraction test", + "summary": "A crafted entry wrote outside the extraction root.", + "evidenceRefs": ["archive-write"], + "assertions": ["The archive entry controls the destination path."], + "limitations": ["The test used a temporary extraction directory."], + }, + "rootCause": { + "summary": "The archive destination is written before containment is enforced.", + "evidenceRefs": ["archive-write"], + }, + "evidenceExcerpt": "destination.write_bytes(entry.read())", + "attackPath": { + "dataFlow": "archive entry -> destination path -> filesystem write", + "reachability": "An archive uploader can supply the crafted entry.", + "evidenceRefs": ["archive-write"], + "impact": { + "level": "high", + "why": "The write can replace files outside the extraction root.", + }, + "likelihood": { + "level": "high", + "why": "No containment check blocks the crafted path.", + }, + "limitations": ["Writable targets depend on process permissions."], + }, + "preventiveControls": ["Use a containment-checking extraction helper."], + "remediation": "Reject archive entries that escape the extraction root.", + "remediationTests": ["Reject traversal entries during extraction."], + "provenance": {"source": "local_plugin"}, + } + + def write_completed_contract( scan_dir: Path, scan_id: str, @@ -310,65 +431,7 @@ def write_completed_contract( "documentType": "codex-security.findings", "schemaVersion": "1.0", "scanId": artifact_scan_id, - "findings": [ - { - "ruleId": "path-traversal.archive-extraction", - "identity": {"anchor": identity_anchor}, - "title": "Unsafe archive extraction can escape the output directory", - "summary": "An attacker-controlled path reaches a filesystem write.", - "severity": { - "level": "high", - "rationale": "The reachable write can escape the extraction root.", - }, - "confidence": {"level": "high", "rationale": "Direct source trace."}, - "taxonomy": {"category": "path-traversal", "cwe": ["CWE-22"]}, - "locations": [ - {"path": relative_path, "startLine": 41, "endLine": 44, "role": "sink"} - ], - "codeEvidence": [ - { - "id": "archive-write", - "label": "Unchecked archive write", - "path": relative_path, - "startLine": 41, - "endLine": 44, - "language": "python", - "code": "destination.write_bytes(entry.read())", - "explanation": "The destination is written before containment is checked.", - } - ], - "validation": { - "method": "archive extraction test", - "summary": "A crafted entry wrote outside the extraction root.", - "evidenceRefs": ["archive-write"], - "assertions": ["The archive entry controls the destination path."], - "limitations": ["The test used a temporary extraction directory."], - }, - "rootCause": { - "summary": "The archive destination is written before containment is enforced.", - "evidenceRefs": ["archive-write"], - }, - "evidenceExcerpt": "destination.write_bytes(entry.read())", - "attackPath": { - "dataFlow": "archive entry -> destination path -> filesystem write", - "reachability": "An archive uploader can supply the crafted entry.", - "evidenceRefs": ["archive-write"], - "impact": { - "level": "high", - "why": "The write can replace files outside the extraction root.", - }, - "likelihood": { - "level": "high", - "why": "No containment check blocks the crafted path.", - }, - "limitations": ["Writable targets depend on process permissions."], - }, - "preventiveControls": ["Use a containment-checking extraction helper."], - "remediation": "Reject archive entries that escape the extraction root.", - "remediationTests": ["Reject traversal entries during extraction."], - "provenance": {"source": "local_plugin"}, - } - ], + "findings": [finding_fixture(relative_path=relative_path, identity_anchor=identity_anchor)], } coverage = { "documentType": "codex-security.coverage", diff --git a/sdk/typescript/README.md b/sdk/typescript/README.md index f14140c50..352dc14b6 100644 --- a/sdk/typescript/README.md +++ b/sdk/typescript/README.md @@ -1582,7 +1582,16 @@ If discovery finished before the interruption, resume completes and seals the same scan. No archiving or new attempt directory is needed. A failed connection leaves the existing scan available for another resume attempt. -Compatible saved scans can resume after a plugin update. Already-sealed results +Compatible saved scans can resume after a plugin update. Unsealed scans from the +retired Deep Scan runtime cannot resume; start a fresh scan instead. Saved drafts +must match the current schema. Recover unfinished worker or reducer fragments +from that runtime with the prior release; the current runtime reads parent drafts +and ordinary scan checkpoints. Existing report files remain available, and +rejecting a failed or canceled checkpoint leaves its saved accounting unchanged. +Older unreleased builds identified child scans by their artifact paths. That +state is no longer migrated; start fresh scans for those database snapshots. +Existing report files remain available. +Already-sealed results keep their original producer version and contents when completion is recorded. Unsupported or invalid sealed artifacts are rejected before resuming, preserving the saved scan state and files. diff --git a/sdk/typescript/scripts/ci-test-durations.json b/sdk/typescript/scripts/ci-test-durations.json index e6cbcc1e5..3ee5e00ce 100644 --- a/sdk/typescript/scripts/ci-test-durations.json +++ b/sdk/typescript/scripts/ci-test-durations.json @@ -41,7 +41,6 @@ "cloud-publish.test.ts": 2.111, "codex-review.test.ts": 0.246, "compact-diff-scan.test.ts": 12.898, - "completed-scan-handoff.test.ts": 0.004, "component-scan.test.ts": 0.396, "config.test.ts": 0.403, "container-entrypoint.test.ts": 0.051, @@ -51,12 +50,6 @@ "custom-publish.test.ts": 0.463, "custom-validation.test.ts": 8.264, "deep-progress.test.ts": 0.001, - "deep-scan-parent-denials.test.ts": 0.013, - "deep-scan-reducer-recovery.test.ts": 0.504, - "deep-scan-timestamp-compat.test.ts": 1.176, - "deep-scan-windows-executable.test.ts": 0.008, - "deep-scan-workbench.test.ts": 15.302, - "deep-scan-worker-shutdown.test.ts": 0.009, "diff-rank-input.test.ts": 0.413, "errors.property.test.ts": 0.033, "errors.test.ts": 0.004, diff --git a/sdk/typescript/scripts/fixtures/mcp-smoke.d.mts b/sdk/typescript/scripts/fixtures/mcp-smoke.d.mts new file mode 100644 index 000000000..97d646c06 --- /dev/null +++ b/sdk/typescript/scripts/fixtures/mcp-smoke.d.mts @@ -0,0 +1,5 @@ +export const mcpSmokeInput: string; +export function mcpSmokeResponses(stdout: string): Array<{ + id?: number; + result: { tools: unknown[] }; +}>; diff --git a/sdk/typescript/scripts/fixtures/mcp-smoke.mjs b/sdk/typescript/scripts/fixtures/mcp-smoke.mjs new file mode 100644 index 000000000..af40ff503 --- /dev/null +++ b/sdk/typescript/scripts/fixtures/mcp-smoke.mjs @@ -0,0 +1,24 @@ +export const mcpSmokeInput = + [ + { + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2025-11-25", + capabilities: {}, + clientInfo: { name: "codex-security-package-smoke", version: "1" }, + }, + }, + { jsonrpc: "2.0", method: "notifications/initialized" }, + { jsonrpc: "2.0", id: 2, method: "tools/list", params: {} }, + ] + .map((request) => JSON.stringify(request)) + .join("\n") + "\n"; + +export function mcpSmokeResponses(stdout) { + return stdout + .trim() + .split("\n") + .map((line) => JSON.parse(line)); +} diff --git a/sdk/typescript/scripts/generate-models.cjs b/sdk/typescript/scripts/generate-models.cjs index f435c37da..5c660f5da 100644 --- a/sdk/typescript/scripts/generate-models.cjs +++ b/sdk/typescript/scripts/generate-models.cjs @@ -18,6 +18,16 @@ function withoutAllOf(value) { ); } +function compileModel(schema, name) { + // allOf with contains or if/then hides object fields from the compiler. + return compile({ ...withoutAllOf(schema), title: name }, name, { + bannerComment: "", + format: false, + ignoreMinAndMaxItems: true, + unknownAny: true, + }); +} + async function generate() { const documents = [ ["scan-manifest.schema.json", "ScanManifest"], @@ -27,15 +37,7 @@ async function generate() { const models = await Promise.all( documents.map(async ([filename, name]) => { const schema = JSON.parse(readFileSync(join(schemas, filename), "utf8")); - // json-schema-to-typescript drops object fields when allOf uses contains or if/then. - const input = withoutAllOf(schema); - input.title = name; - return compile(input, name, { - bannerComment: "", - format: false, - ignoreMinAndMaxItems: true, - unknownAny: true, - }); + return compileModel(schema, name); }), ); @@ -101,13 +103,7 @@ async function generateSemanticModels() { ), ); input.$defs.common = common; - input.title = "SemanticScan"; - const model = await compile(withoutAllOf(input), "SemanticScan", { - bannerComment: "", - format: false, - ignoreMinAndMaxItems: true, - unknownAny: true, - }); + const model = await compileModel(input, "SemanticScan"); return format( [ "/* Generated from the plugin semantic draft schema. Run `pnpm generate:models`. */", @@ -121,12 +117,11 @@ async function generateSemanticModels() { ); } -Promise.all([generate(), generateSemanticModels()]).then((documents) => { - for (const [index, filename] of [ - "models.ts", - "semantic-models.ts", - ].entries()) { - const models = documents[index]; +Promise.all([ + generate().then((document) => ["models.ts", document]), + generateSemanticModels().then((document) => ["semantic-models.ts", document]), +]).then((documents) => { + for (const [filename, models] of documents) { const output = join(packageRoot, "src", filename); if (process.argv.includes("--check")) { if (readFileSync(output, "utf8").replaceAll("\r\n", "\n") !== models) { diff --git a/sdk/typescript/scripts/merge-eval/fixtures.ts b/sdk/typescript/scripts/merge-eval/fixtures.ts index 79da66d65..66e75002c 100644 --- a/sdk/typescript/scripts/merge-eval/fixtures.ts +++ b/sdk/typescript/scripts/merge-eval/fixtures.ts @@ -1,5 +1,6 @@ import type { ScanAggregate, ScanMergeInput } from "../../src/scan-merge.js"; import type { SemanticFinding } from "../../src/semantic-models.js"; +import { semanticCoverage } from "../../tests-ts/helpers/semantic-scan.js"; export const parentId = "7fc17317-9594-49e0-b06a-d72fd7e14bba"; @@ -53,12 +54,10 @@ function input(scanId: string, findings: SemanticFinding[]): ScanMergeInput { sourceFindingIds: [`${scanId}:${index}`], }, })), - coverage: { + coverage: semanticCoverage({ completeness: "partial", - surfaces: [], - explicitExclusions: [], deferred: [{ reason: "Synthetic outstanding work." }], - }, + }), }, }; } diff --git a/sdk/typescript/scripts/smoke-package.mjs b/sdk/typescript/scripts/smoke-package.mjs index 662acfd22..35525e873 100644 --- a/sdk/typescript/scripts/smoke-package.mjs +++ b/sdk/typescript/scripts/smoke-package.mjs @@ -1,3 +1,4 @@ +import { mcpSmokeInput, mcpSmokeResponses } from "./fixtures/mcp-smoke.mjs"; import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; import { @@ -234,19 +235,7 @@ async function smokeSharedScanRuntime(installedRoot, consumer) { cwd: pluginRoot, encoding: "utf8", env: { ...scanEnvironment, CODEX_MCP_NODE_PATH: process.execPath }, - input: `${JSON.stringify({ - jsonrpc: "2.0", - id: 1, - method: "initialize", - params: { - protocolVersion: "2025-11-25", - capabilities: {}, - clientInfo: { - name: "codex-security-package-smoke", - version: "0.1.0", - }, - }, - })}\n${JSON.stringify({ jsonrpc: "2.0", method: "notifications/initialized" })}\n${JSON.stringify({ jsonrpc: "2.0", id: 2, method: "tools/list", params: {} })}\n`, + input: mcpSmokeInput, timeout: PACKAGE_SMOKE_TIMEOUT_MS, windowsHide: true, }, @@ -257,10 +246,7 @@ async function smokeSharedScanRuntime(installedRoot, consumer) { }); } assert.equal(initialized.status, 0, initialized.stderr); - const mcpResponses = initialized.stdout - .trim() - .split("\n") - .map((line) => JSON.parse(line)); + const mcpResponses = mcpSmokeResponses(initialized.stdout); assert.equal( mcpResponses.find((response) => response.id === 1)?.result.serverInfo.name, "codex-security", diff --git a/sdk/typescript/src/api.ts b/sdk/typescript/src/api.ts index b1289f867..f0eb4c1e3 100644 --- a/sdk/typescript/src/api.ts +++ b/sdk/typescript/src/api.ts @@ -1,7 +1,6 @@ /// import { prepareScanSkill, scanPrompt } from "./scan-preparation.js"; -import { findScanSession } from "./scan-logs.js"; import { scanAuthentication, @@ -66,7 +65,6 @@ import { randomUUID } from "node:crypto"; import { runDeepScans, ScanCostTrackingError, - TerminalDeepScanError, terminalDeepScanError, } from "./deep-scan.js"; import { @@ -74,17 +72,13 @@ import { ScanTransportClosedError, ScanPermissionError, } from "./scan-execution.js"; -import { - compositionCheckpointFromWorkbench, - type DeepScanCheckpointSummary, -} from "./deep-scan-checkpoint.js"; -import { - savedScanFromWorkbench, - savedScansFromWorkbench, -} from "./workbench-types.js"; +import { compositionCheckpointFromWorkbench } from "./deep-scan-checkpoint.js"; import { collectResult, publishScan, + readSealedScanTurn, + hasSealedScanArtifacts, + restorePriorScanCosts, preservePublishedArtifacts, writeSemanticScanDraft, type CompletedScanTurn, @@ -125,6 +119,7 @@ import { resolveCommandAuthConfig, scanApprovalPolicy, scanModelConfiguration, + scanModel, scanModelProvider, type CodexSecurityConfig, type JsonObject, @@ -132,6 +127,8 @@ import { } from "./config.js"; import { estimateScanCost, + addScanCosts, + scanCostUsage, ScanCostTracker, type ScanCost, type ScanSessionEvent, @@ -229,6 +226,7 @@ import { acquireCodexSecurityCredentialHomeLock, bootstrapPlugin, bundledPluginRoot, + canonicalizeModelSafePath, cleanupSdkDirectory, codexSecurityCredentialAllowsAmbientImport, codexSecurityCredentialHome, @@ -1209,6 +1207,7 @@ export class CodexSecurity { ): ScanCost | null => [...passCosts.values()].includes(null) ? null : combinedCost(current); let scanDir = ""; + let reportWorkspace: string | undefined; let archivedScanDir: string | null = null; let targetPathsFile: string | null = null; let knowledgeBase: PreparedKnowledgeBase | null = null; @@ -1220,7 +1219,6 @@ export class CodexSecurity { let artifactRestorationFailure: OutputDirectoryError | null = null; let customValidationComplete = false; let completionCost: ScanCost | null = null; - let terminalMergeCost: ScanCost | null | undefined; let budgetRecovery: { expectation: ScanExpectation; pluginRoot: string; @@ -1263,6 +1261,46 @@ export class CodexSecurity { input, ); }; + const reportTrackingError = (error: unknown): void => { + if (options.maxCostUsd !== undefined || options.requireCost) { + costAbortController.abort( + new ScanCostTrackingError( + `Scan interrupted because required cost tracking failed: ${errorMessage(error)}`, + scanDir, + { cause: error }, + ), + ); + return; + } + notifyObserver( + "onWarning", + options.onWarning, + options.onObserverError, + `Could not track scan activity: ${errorMessage(error)}`, + ); + }; + // Saved totals describe already completed work and cannot spend the execution budget. + const reportSavedCost = (cost: Readonly): void => + notifyObserver( + "onCost", + options.onCost, + options.onObserverError, + cost, + options.maxCostUsd, + ); + const reportWarnings = ( + warnings: Awaited>["warnings"], + ): void => { + for (const warning of warnings) { + notifyObserver( + "onWarning", + options.onWarning, + options.onObserverError, + warning.message, + warning.targetChanged ? { kind: "target_changed" } : undefined, + ); + } + }; try { const checkOpen = (): void => { this.#requireOpen(); @@ -1314,6 +1352,183 @@ export class CodexSecurity { } checkOpen(); + const resumeScanId = + options.resumeScanId ?? options.registeredScan?.scanId; + if ( + resumeScanId !== undefined && + !options.postScanPrompt?.trim() && + (await hasSealedScanArtifacts(requestedOutput!, signal)) + ) { + // Reading a sealed result needs the workbench and saved session logs, not Codex authentication. + let pluginRoot = + this.#runtime?.plugin.pluginRoot ?? + this.#dependencies.ambientExecution?.pluginRoot; + if (pluginRoot === undefined) { + reportWorkspace = await mkdtemp( + join(temporaryRoot!, "codex-security-report-"), + ); + pluginRoot = await resolvePluginPath( + this.config.pluginPath, + reportWorkspace, + signal, + ); + } + const environment = this.#dependencies.environment; + const codexHome = await canonicalizeModelSafePath( + this.#runtime?.codexHome ?? + (this.#dependencies.ambientExecution === undefined + ? codexSecurityCredentialHome(environment) + : configuredCodexHome( + this.#dependencies.ambientExecution.environment, + )), + ); + requireOutputOutsideRepositories( + inputs.protectedRoots, + codexHome, + "runtime", + ); + const python = await ( + this.#dependencies.resolvePluginPython ?? resolvePluginPython + )({ + configuredPath: this.config.pythonPath, + environment, + protectedRoot, + signal, + }); + let git: InspectedExecutable = { executable: null, environment }; + for (const root of [ + (await gitMarkerRoot(repo, signal, "outermost")) ?? repo, + ...(knowledgeBase?.snapshot.protectedRoots ?? []), + ]) { + git = await inspectTrustedExecutable("git", git.environment, root); + } + const readOptions: WorkbenchCommandOptions = { + python, + pluginRoot, + signal, + environment: { + ...withoutCodexHome(environmentWithGit(git.environment, git)), + CODEX_HOME: codexHome, + CODEX_SECURITY_STATE_DIR: stateDirectory, + }, + failureMessage: "Could not load the saved Codex Security scan", + }; + // Native registration must bind its owner and claim before using a saved recipe. + const savedRecipe = + options.registeredScan === undefined + ? {} + : ( + await workbench(readOptions, [ + "get-scan", + "--scan-id", + resumeScanId, + ]) + )["recipe"]; + if (isRecord(savedRecipe)) { + const expectation: ScanExpectation = { + repository: repo, + target: normalized, + mode, + repositoryRevision: await ( + this.#dependencies.repositoryRevision ?? repositoryRevision + )(repo, signal), + pluginVersion: (await pluginMetadata(pluginRoot)).version, + }; + if ( + options.expectedPluginVersion !== undefined && + options.expectedPluginVersion !== expectation.pluginVersion + ) + throw new CodexSecurityError( + `The original scan used plugin version ${options.expectedPluginVersion}, but the installed version is ${expectation.pluginVersion}.`, + ); + const recipe: JsonObject = { + ...savedRecipe, + repository: repo, + target: { ...normalized, paths: [...normalized.paths] }, + mode, + }; + delete recipe["knowledgeBaseSha256"]; + if (knowledgeBase !== null) + recipe["knowledgeBaseSha256"] = knowledgeBase.sha256; + scanDir = requestedOutput!; + releaseExecution = await ( + this.#dependencies.acquireScanExecution ?? acquireScanExecution + )(stateDirectory, scanDir, await bundledPluginRoot()); + const registered = await registerScan({ + scan: options, + recipe, + expectation, + scanDir: requestedOutput!, + archivedScanDir: null, + workbench: (args, input) => workbench(readOptions, args, input), + }); + if (registered.sealed) { + notifyObserver( + "onOutputDirReady", + options.onOutputDirReady, + options.onObserverError, + scanDir, + ); + const model = scanModel({ + ...DEFAULT_CODEX_CONFIG, + ...((registered.registration["recipe"] as JsonObject)[ + "config" + ] as JsonObject), + }); + if (typeof model !== "string" || model.trim().length === 0) + throw new ConfigurationError( + "The configured Codex model must be a nonempty string.", + ); + validateScanCostLimit(options.maxCostUsd, model); + const turn = await readSealedScanTurn({ + startedAt: registered.registration["startedAt"], + scanId: registered.scanId, + scanDir, + expectation, + model, + codexHome, + signal, + workbench: (args) => workbench(readOptions, args), + maxCostUsd: options.maxCostUsd, + onTrackingError: reportTrackingError, + onCost: reportSavedCost, + }); + await options.onRegisteredScan?.(registered.registration); + const { result, warnings } = await publishScan( + { + scanId: registered.scanId, + scanDir, + pluginRoot, + expectation, + signal, + workbench: (args) => workbench(readOptions, args), + }, + turn, + turn.cost, + true, + ); + reportWarnings(warnings); + if (!options.deepScanPass) + try { + result.repositoryFindings = (await listRepositoryFindings( + (args) => workbench(readOptions, args), + registered.targetId, + )) as RepositoryFinding[] | undefined; + } catch (error) { + if (error instanceof ScanPermissionError) throw error; + notifyObserver( + "onWarning", + options.onWarning, + options.onObserverError, + `Could not update repository findings: ${errorMessage(error)}`, + ); + } + + return result; + } + } + } + const session = await this.#prepareSession( { protectedRoot }, options, @@ -1358,8 +1573,7 @@ export class CodexSecurity { ); } scanDir = - options.resumeScanId !== undefined || - options.registeredScan !== undefined + resumeScanId !== undefined ? requestedOutput! : await (this.#dependencies.prepareOutputDir ?? prepareOutputDir)( requestedOutput ?? undefined, @@ -1379,7 +1593,7 @@ export class CodexSecurity { ); requireOutputOutsideRepository(protectedRoot, scanDir); requireModelSafeOutputDir(scanDir); - releaseExecution = await ( + releaseExecution ??= await ( this.#dependencies.acquireScanExecution ?? acquireScanExecution )(stateDirectory, scanDir, await bundledPluginRoot()); notifyObserver( @@ -1484,70 +1698,9 @@ export class CodexSecurity { ); } }; - const isLegacyScanSession = async ( - threadId: string, - ): Promise => { - const saved = await findScanSession(runtime.codexHome, threadId); - const startedAt = - typeof registration["startedAt"] === "string" - ? Date.parse(registration["startedAt"]) - : NaN; - // Native owners can include earlier conversation work, even from this directory. - return ( - saved?.workingDirectory === scanDir && - saved.startedAt !== null && - Number.isFinite(startedAt) && - saved.startedAt >= startedAt - ); - }; - if ( - !sealed && - mode === "deep" && - (options.resumeScanId !== undefined || - options.registeredScan !== undefined) - ) { - const readHistoricalCost = async ( - threadId: string, - scanDirectory: string, - ) => { - if ( - scanDirectory === scanDir && - !(await isLegacyScanSession(threadId)) - ) - return null; - const historical = new ScanCostTracker({ - codexHome: runtime.codexHome, - model, - repository: repo, - scanDirectory, - }); - historical.start(threadId); - return (await historical.stop()).cost; - }; - const terminal = await terminalDeepScanError({ - scanId, - scanDir, - repository: repo, - workbench: (args) => workbench(workbenchOptions, args), - historicalCost: (threadId) => readHistoricalCost(threadId, scanDir), - }); - if (terminal !== null) { - // A failed optional thread write does not prove the merge was free. - if (typeof resumeThreadId !== "string") terminalMergeCost = null; - const { constituents } = terminal.accounting; - if ( - constituents !== null && - typeof resumeThreadId === "string" && - resumeThreadId !== terminal.accounting.legacyThreadId - ) { - terminalMergeCost = await readHistoricalCost( - resumeThreadId, - join(scanDir, "artifacts/deep-scan/merge"), - ).catch(() => null); - } - activeScan = { id: scanId, options: workbenchOptions }; - throw terminal; - } + if (!sealed && mode === "deep" && resumeScanId !== undefined) { + const terminal = await terminalDeepScanError({ scanDir }); + if (terminal !== null) throw terminal; } await requireScanResumeSession({ registration: registered, @@ -1557,24 +1710,6 @@ export class CodexSecurity { }); const progress = new ScanProgressReporter(mode, options); const reportProgress = progress.report; - const reportTrackingError = (error: unknown): void => { - if (options.maxCostUsd !== undefined || options.requireCost) { - costAbortController.abort( - new ScanCostTrackingError( - `Scan interrupted because required cost tracking failed: ${errorMessage(error)}`, - scanDir, - { cause: error }, - ), - ); - return; - } - notifyObserver( - "onWarning", - options.onWarning, - options.onObserverError, - `Could not track scan activity: ${errorMessage(error)}`, - ); - }; const stopTracking = async ( activeTracker: ScanCostTracker, usage?: unknown, @@ -1592,14 +1727,6 @@ export class CodexSecurity { threadId: string, scanDirectory = scanDir, ) => { - if ( - scanDirectory === scanDir && - !(await isLegacyScanSession(threadId).catch((error: unknown) => { - reportTrackingError(error); - return false; - })) - ) - return null; const historical = new ScanCostTracker({ codexHome: runtime.codexHome, model, @@ -1665,129 +1792,39 @@ export class CodexSecurity { const reportScanProgress = (update: ScanProgress): void => progress.fromScan(update, tracker); progress.preflight(registered.scopeFileCount, tracker); - const restorePriorAccounting = ( - checkpoint: DeepScanCheckpointSummary | null, - ): void => { - if (checkpoint?.legacy) - passCosts.set("legacy", checkpoint.legacy.cost ?? null); - if ( - checkpoint?.costUnavailable || - (typeof resumeThreadId !== "string" && - checkpoint !== null && - (checkpoint.mergedScanIds.length > 0 || - checkpoint.passes.some((pass) => pass.completed))) - ) { - // Completed inputs precede merging. Missing optional session - // metadata does not establish zero prior cost. - passCosts.set("previous-work", null); - if (options.maxCostUsd !== undefined) - throw new ScanCostTrackingError( - "A prior scan session is unavailable; its cost limit cannot be verified.", - scanDir, - ); - } - }; - let sealedThreadId: string | null = null; - if (sealed) { - // Reconcile sealing before the ordinary completion transaction without rewriting artifacts. - const saved = await workbench(workbenchOptions, [ - "get-scan", - "--scan-id", - scanId, - ]); - const savedScan = savedScanFromWorkbench(saved); - const checkpoint = compositionCheckpointFromWorkbench(saved); - resumeThreadId = savedScan["continuationThreadId"]; - restorePriorAccounting(checkpoint); - // Legacy cost already includes this origin session; do not restart its tracker. - sealedThreadId = - typeof resumeThreadId === "string" - ? resumeThreadId - : (checkpoint?.legacy?.originThreadId ?? null); - scanThreadId = sealedThreadId ?? undefined; - const emptyComposition = - mode === "deep" && - checkpoint?.terminalReason === "capped" && - checkpoint.mergedScanIds.length === 0 && - Array.isArray(savedScan["findings"]) && - savedScan["findings"].length === 0; - if ( - sealedThreadId === null && - !emptyComposition && - !passCosts.has("previous-work") - ) - throw new CodexSecurityError( - "The sealed scan has no saved execution session.", - ); - if (checkpoint?.legacy) - passCosts.set( - "legacy", - checkpoint.legacy.cost ?? - (checkpoint.legacy.originThreadId - ? await historicalCost(checkpoint.legacy.originThreadId) - : null), - ); - if (checkpoint != null) { - const children = await workbench(workbenchOptions, [ - "list-scans", - "--scan-root", - join(scanDir, "artifacts/deep-scan/passes"), - ]); - for (const child of savedScansFromWorkbench(children)) { - if (child.parentScanId === scanId) - passCosts.set(child.scanId, child.cost ?? null); - } - } - if (mode === "deep" && checkpoint == null) { - completionCost = await historicalCost(sealedThreadId!); - completionCost ??= savedScan.cost ?? null; - passCosts.set("legacy", completionCost); - // This retired origin was measured above; it is not a composed merge session. - resumeThreadId = null; - } - if ( - !passCosts.has("previous-work") && - (savedScan.progress.status === "complete" || - ![...passCosts.values()].includes(null)) - ) - completionCost ??= savedScan.cost ?? null; - if ( - typeof resumeThreadId !== "string" && - (emptyComposition || checkpoint?.legacy) - ) - completionCost ??= completeCost(null); - if ( - completionCost === null && - options.maxCostUsd !== undefined && - [...passCosts.values()].includes(null) - ) - throw new ScanCostTrackingError( - "The saved child scan cost is unavailable; its cost limit cannot be verified.", + const sealedTurn = sealed + ? await readSealedScanTurn({ + startedAt: registration["startedAt"], + scanId, scanDir, - ); + expectation, + model, + codexHome: runtime.codexHome, + signal, + workbench: (args) => workbench(workbenchOptions, args), + maxCostUsd: options.maxCostUsd, + onTrackingError: reportTrackingError, + onCost: reportSavedCost, + }) + : null; + if (sealedTurn !== null) { + resumeThreadId = sealedTurn.resumeThreadId; + completionCost = sealedTurn.cost; + scanThreadId = sealedTurn.threadId ?? undefined; } else { - if ( - mode === "deep" && - (options.resumeScanId !== undefined || - options.registeredScan !== undefined) - ) { + if (mode === "deep" && resumeScanId !== undefined) { const saved = await workbench(workbenchOptions, [ "get-scan", "--scan-id", scanId, ]); - const checkpoint = compositionCheckpointFromWorkbench(saved); - restorePriorAccounting(checkpoint); - if ( - options.maxCostUsd !== undefined && - checkpoint?.legacy && - !checkpoint.legacy.cost && - (!checkpoint.legacy.originThreadId || - !(await historicalCost(checkpoint.legacy.originThreadId))) - ) - throw new CodexSecurityError( - "Restore the original Deep Scan session logs to verify its saved cost limit.", - ); + restorePriorScanCosts( + passCosts, + compositionCheckpointFromWorkbench(saved), + resumeThreadId, + scanDir, + options.maxCostUsd, + ); } activeScan = { id: scanId, options: workbenchOptions }; } @@ -1983,9 +2020,11 @@ export class CodexSecurity { ); } thread = codex.resumeThread(resumeThreadId, threadOptions); - tracker.start(resumeThreadId); - if (budgetRecovery !== null) budgetRecovery.threadId = resumeThreadId; - await tracker.refresh().catch(reportTrackingError); + if (!sealed) { + tracker.start(resumeThreadId); + if (budgetRecovery !== null) budgetRecovery.threadId = resumeThreadId; + await tracker.refresh().catch(reportTrackingError); + } checkOpen(); } else { thread = codex.startThread(threadOptions); @@ -2142,31 +2181,8 @@ export class CodexSecurity { : (await thread.runStreamed(prompt, { signal })).events; checkOpen(); - const completedTurn: CompletedScanTurn = sealed - ? await (async () => { - budgetAbortController.abort(); - const snapshot = await stopTracking(tracker); - const measuredCost = - snapshot.cost === null ? null : completeCost(snapshot.cost); - if ( - measuredCost && - (!completionCost || - measuredCost.estimatedUsd > completionCost.estimatedUsd) - ) - completionCost = measuredCost; - return { - threadId: sealedThreadId, - turnResult: { - status: "completed", - model, - usage: completionCost - ? scanCostUsage(completionCost) - : mode === "deep" - ? null - : snapshot.usage, - }, - }; - })() + const completedTurn: CompletedScanTurn = sealedTurn + ? sealedTurn : mode === "deep" ? await (async () => { const settings = deepScanConfiguration!.settings; @@ -2191,7 +2207,7 @@ export class CodexSecurity { options.onScanStarted, options.onObserverError, ); - const checkpoint = await runDeepScans({ + await runDeepScans({ scanId, scanDir, costUnavailable: passCosts.has("previous-work"), @@ -2338,17 +2354,12 @@ export class CodexSecurity { draft, ), }); - if (budgetRecovery !== null) - budgetRecovery.threadId ??= - checkpoint.legacy?.originThreadId ?? null; const usage = await finalize( undefined, thread.id === null ? completeCost(null) : null, ); - const resultThreadId = - thread.id ?? checkpoint.legacy?.originThreadId ?? null; return { - threadId: resultThreadId, + threadId: thread.id, turnResult: { status: "completed", model, usage }, }; })() @@ -2409,15 +2420,7 @@ export class CodexSecurity { sealed, ); activeScan = null; - for (const warning of warnings) { - notifyObserver( - "onWarning", - options.onWarning, - options.onObserverError, - warning.message, - warning.targetChanged ? { kind: "target_changed" } : undefined, - ); - } + reportWarnings(warnings); if (runPostScan !== null) { const followUp = runPostScan; runPostScan = null; @@ -2682,43 +2685,16 @@ export class CodexSecurity { this.#abortController.signal.aborted) && isCancellationDerivedFailure(failure, signal); - let terminalCost: Readonly | null = null; - if (error instanceof TerminalDeepScanError) { - const { constituents, savedTotal } = error.accounting; - terminalCost = savedTotal; - const costs = - constituents === null - ? null - : [ - ...constituents, - ...(terminalMergeCost === undefined ? [] : [terminalMergeCost]), - ]; - if (costs !== null && !costs.includes(null)) { - const recovered = costs.reduce( - (total, cost) => - cost === null ? total : addScanCosts(total, cost), - tracked?.cost ?? null, - ); - if ( - recovered && - (!terminalCost || - recovered.estimatedUsd > terminalCost.estimatedUsd) - ) - terminalCost = recovered; - } - } const preservedCost = - error instanceof TerminalDeepScanError - ? terminalCost - : options.mode === "deep" - ? (completionCost ?? - (tracked?.cost || - (scanThreadId === undefined && - options.resumeScanId === undefined && - options.registeredScan === undefined) - ? completeCost(tracked?.cost ?? null) - : null)) - : snapshot?.cost; + options.mode === "deep" + ? (completionCost ?? + (tracked?.cost || + (scanThreadId === undefined && + options.resumeScanId === undefined && + options.registeredScan === undefined) + ? completeCost(tracked?.cost ?? null) + : null)) + : snapshot?.cost; if ( activeScan !== null && (options.deepScanPass || transportClosed || canceled) @@ -2805,6 +2781,9 @@ export class CodexSecurity { try { for (const cleanup of await Promise.allSettled([ knowledgeBase?.cleanup(), + reportWorkspace === undefined + ? undefined + : cleanupSdkDirectory(reportWorkspace), removeTargetPathsFile(targetPathsFile), ])) { if (cleanup.status === "rejected") { @@ -4094,54 +4073,6 @@ function validateScanCostLimit( } } -function addScanCosts( - previous: Readonly | null, - current: Readonly, -): ScanCost { - if (previous === null) return { ...current }; - const { estimatedUsdRange: currentRange, ...currentCost } = current; - const previousRange = previous.estimatedUsdRange; - return { - ...currentCost, - inputTokens: previous.inputTokens + current.inputTokens, - cachedInputTokens: previous.cachedInputTokens + current.cachedInputTokens, - cacheWriteInputTokens: - previous.cacheWriteInputTokens + current.cacheWriteInputTokens, - outputTokens: previous.outputTokens + current.outputTokens, - estimatedUsd: previous.estimatedUsd + current.estimatedUsd, - ...(previous.cacheWriteInputTokensReported === false || - current.cacheWriteInputTokensReported === false - ? { cacheWriteInputTokensReported: false } - : {}), - ...(previousRange === undefined || currentRange === undefined - ? {} - : { - estimatedUsdRange: { - context: "unknown" as const, - min: previousRange.min + currentRange.min, - max: - previousRange.max === null || currentRange.max === null - ? null - : previousRange.max + currentRange.max, - }, - }), - }; -} - -function scanCostUsage( - cost: Readonly, -): Record { - return { - input_tokens: cost.inputTokens, - cached_input_tokens: cost.cachedInputTokens, - cache_write_input_tokens: cost.cacheWriteInputTokens, - output_tokens: cost.outputTokens, - ...(cost.cacheWriteInputTokensReported === false - ? { cache_write_input_tokens_reported: false } - : {}), - }; -} - /** Shell-neutral guidance so PowerShell users are not told to run POSIX `unset`. */ export function formatEnvironmentVariableRemovalGuidance( names: readonly string[], @@ -4289,13 +4220,12 @@ function sharedCredentialCodexConfig( features: { plugins: true }, }; for (const key of CODEX_AUTH_CONFIG_KEYS) { - if (Object.hasOwn(config, key)) shared[key] = structuredClone(config[key]!); + if (Object.hasOwn(config, key)) shared[key] = config[key]!; } const modelProvider = scanModelProvider(config); if (hasCommandAuth(config)) { for (const key of ["profile", "profiles"]) { - if (Object.hasOwn(config, key)) - shared[key] = structuredClone(config[key]!); + if (Object.hasOwn(config, key)) shared[key] = config[key]!; } } if (typeof modelProvider === "string" && modelProvider.length > 0) { @@ -4303,7 +4233,7 @@ function sharedCredentialCodexConfig( const providers = config["model_providers"]; if (isRecord(providers) && Object.hasOwn(providers, modelProvider)) { shared["model_providers"] = { - [modelProvider]: structuredClone(providers[modelProvider]!), + [modelProvider]: providers[modelProvider]!, }; } } diff --git a/sdk/typescript/src/config.ts b/sdk/typescript/src/config.ts index 42b6a3125..c77ae38bc 100644 --- a/sdk/typescript/src/config.ts +++ b/sdk/typescript/src/config.ts @@ -205,6 +205,20 @@ export function resolveCodexProfile(config: JsonObject): JsonObject { return resolved; } +/** Apply a worker budget to a config owned by this scan. */ +export function setScanSubagentBudget( + config: JsonObject, + subagents: number, +): void { + const features = isObject(config["features"]) ? config["features"] : {}; + features["multi_agent_v2"] = { + ...(isObject(features["multi_agent_v2"]) ? features["multi_agent_v2"] : {}), + enabled: true, + max_concurrent_threads_per_session: subagents + 1, + }; + config["features"] = features; +} + /** Carry a selected scan into another ordinary client without copying managed plugin registration. */ export function scanCompositionOverrides( config: JsonObject, @@ -213,14 +227,8 @@ export function scanCompositionOverrides( const result = resolveCodexProfile(config); delete result["plugins"]; delete result["marketplaces"]; - const features = isObject(result["features"]) ? result["features"] : {}; - delete features["plugins"]; - features["multi_agent_v2"] = { - ...(isObject(features["multi_agent_v2"]) ? features["multi_agent_v2"] : {}), - enabled: true, - max_concurrent_threads_per_session: subagents + 1, - }; - result["features"] = features; + setScanSubagentBudget(result, subagents); + delete (result["features"] as JsonObject)["plugins"]; if (isObject(result["agents"])) delete result["agents"]["max_threads"]; return result; } diff --git a/sdk/typescript/src/cost.ts b/sdk/typescript/src/cost.ts index 5dea9c6d9..8aa72ab87 100644 --- a/sdk/typescript/src/cost.ts +++ b/sdk/typescript/src/cost.ts @@ -943,3 +943,51 @@ function isRecord(value: unknown): value is Record { function isMissingFile(error: unknown): boolean { return isRecord(error) && error["code"] === "ENOENT"; } + +export function addScanCosts( + previous: Readonly | null, + current: Readonly, +): ScanCost { + if (previous === null) return { ...current }; + const { estimatedUsdRange: currentRange, ...currentCost } = current; + const previousRange = previous.estimatedUsdRange; + return { + ...currentCost, + inputTokens: previous.inputTokens + current.inputTokens, + cachedInputTokens: previous.cachedInputTokens + current.cachedInputTokens, + cacheWriteInputTokens: + previous.cacheWriteInputTokens + current.cacheWriteInputTokens, + outputTokens: previous.outputTokens + current.outputTokens, + estimatedUsd: previous.estimatedUsd + current.estimatedUsd, + ...(previous.cacheWriteInputTokensReported === false || + current.cacheWriteInputTokensReported === false + ? { cacheWriteInputTokensReported: false } + : {}), + ...(previousRange === undefined || currentRange === undefined + ? {} + : { + estimatedUsdRange: { + context: "unknown" as const, + min: previousRange.min + currentRange.min, + max: + previousRange.max === null || currentRange.max === null + ? null + : previousRange.max + currentRange.max, + }, + }), + }; +} + +export function scanCostUsage( + cost: Readonly, +): Record { + return { + input_tokens: cost.inputTokens, + cached_input_tokens: cost.cachedInputTokens, + cache_write_input_tokens: cost.cacheWriteInputTokens, + output_tokens: cost.outputTokens, + ...(cost.cacheWriteInputTokensReported === false + ? { cache_write_input_tokens_reported: false } + : {}), + }; +} diff --git a/sdk/typescript/src/deep-scan.ts b/sdk/typescript/src/deep-scan.ts index 110a10a8d..03da48e0f 100644 --- a/sdk/typescript/src/deep-scan.ts +++ b/sdk/typescript/src/deep-scan.ts @@ -39,11 +39,7 @@ import { reservePass, stopDiscovery, } from "./deep-scan-lifecycle.js"; -import { - savedScanFromWorkbench, - savedScansFromWorkbench, - type SavedScanRecord, -} from "./workbench-types.js"; +import { type SavedScanRecord } from "./workbench-types.js"; export { DEEP_SCAN_CHECKPOINT, type DeepScanCheckpoint, @@ -52,20 +48,8 @@ export { /** Required usage tracking must stop the entire composition before another pass. */ export class ScanCostTrackingError extends ScanInterruptedError {} -/** A terminal checkpoint rejects execution while retaining read-only accounting. */ -export class TerminalDeepScanError extends ScanInterruptedError { - constructor( - message: string, - scanDir: string, - readonly accounting: { - constituents: ReadonlyArray | null> | null; - savedTotal: Readonly | null; - legacyThreadId?: string; - }, - ) { - super(message, scanDir); - } -} +/** A terminal checkpoint rejects execution without changing saved results. */ +export class TerminalDeepScanError extends ScanInterruptedError {} export interface DeepScanComposition { scanId: string; @@ -130,10 +114,7 @@ function missingRunningSession(record: SavedScanRecord): boolean { /** Reject stopped discovery without writes, worker startup or cost notifications. */ export async function terminalDeepScanError( - input: Pick< - DeepScanComposition, - "scanId" | "scanDir" | "repository" | "workbench" | "historicalCost" - >, + input: Pick, checkpoint?: DeepScanCheckpoint, ): Promise { const state = checkpoint ?? (await loadDeepScanCheckpoint(input.scanDir)); @@ -142,56 +123,9 @@ export async function terminalDeepScanError( (state.terminalReason !== "failed" && state.terminalReason !== "canceled") ) return null; - let savedTotal: ScanCost | null = null; - let constituents: Array | null> | null = null; - try { - const saved = await input.workbench([ - "get-scan", - "--scan-id", - input.scanId, - ]); - savedTotal = savedScanFromWorkbench(saved).cost ?? null; - validatePassDirectories(state); - const listed = await input.workbench([ - "list-scans", - "--scan-root", - join(input.scanDir, "artifacts/deep-scan/passes"), - ]); - // A reserved slot cannot incur cost until its registration succeeds. - const costs: Array | null | undefined> = - state.passes.map((pass) => - pass.scanId === undefined ? undefined : null, - ); - for (const record of savedScansFromWorkbench(listed)) { - const index = savedPassIndex(input, state, record); - // Missing optional thread/cost persistence does not establish zero usage. - if (index >= 0) - costs[index] = missingRunningSession(record) - ? null - : (record.cost ?? null); - } - if (state.legacy) { - costs.push( - state.legacy.cost ?? - (state.legacy.originThreadId - ? await input.historicalCost?.(state.legacy.originThreadId) - : null) ?? - null, - ); - } - if (state.costUnavailable) costs.push(null); - constituents = costs.filter((cost) => cost !== undefined); - } catch { - // Optional accounting must not replace the terminal rejection or a known total. - } return new TerminalDeepScanError( `The saved Deep Scan is ${state.terminalReason}; its retained results remain available.`, input.scanDir, - { - constituents, - savedTotal, - legacyThreadId: state.legacy?.originThreadId ?? undefined, - }, ); } @@ -203,6 +137,10 @@ export async function runDeepScans( const state = (await loadDeepScanCheckpoint(scanDir)) ?? newDeepScanCheckpoint(input.startedAt); + if (state.legacy) + throw new Error( + "Saved legacy Deep Scans cannot be resumed; their reports remain available.", + ); if (input.costUnavailable) state.costUnavailable = true; const terminal = await terminalDeepScanError(input, state); if (terminal !== null) throw terminal; @@ -242,21 +180,6 @@ export async function runDeepScans( return queued.pending; }; await save(); - const previousRuns = state.legacy?.discoveryRuns ?? 0; - if (state.legacy && !state.legacy.cost) { - const cost = state.legacy.originThreadId - ? await input.historicalCost?.(state.legacy.originThreadId) - : null; - if (cost) { - state.legacy.cost = cost; - await save(); - } - if (!cost && input.scanOptions.requireCost) - throw new Error( - "Restore the original Deep Scan session logs to verify its saved cost limit.", - ); - } - if (state.legacy) input.onCost("legacy", state.legacy.cost ?? null); const validateMerge = await createScanMergeValidator(input.pluginRoot); const completed = new Map(); const saved = new Map(); @@ -272,7 +195,7 @@ export async function runDeepScans( .map((pass) => pass.directory), state.mergedScanIds.some((id) => !completed.has(id)) ? state.aggregate.coverage - : state.legacy?.coverage, + : undefined, ), }; }; @@ -290,7 +213,7 @@ export async function runDeepScans( "--scan-root", join(scanDir, "artifacts/deep-scan/passes"), ]); - const records = savedScansFromWorkbench(listed); + const records = listed["scans"] as SavedScanRecord[]; if (recoverOutcomes) records.sort((a, b) => (a.completedAt ?? "").localeCompare(b.completedAt ?? ""), @@ -380,7 +303,7 @@ export async function runDeepScans( polling = true; void workbench(["get-scan", "--scan-id", scanId]) .then((result) => { - const { progress } = savedScanFromWorkbench(result); + const { progress } = result["scan"] as SavedScanRecord; if ( progress["status"] === "canceled" || progress["status"] === "failed" @@ -408,7 +331,7 @@ export async function runDeepScans( if (!pending.length) { state.aggregate = { ...validateMerge({ scanId, findings: [] }, [], null).aggregate, - coverage: combineScanCoverage([], [], state.legacy?.coverage), + coverage: combineScanCoverage([]), }; await save(); return; @@ -457,7 +380,7 @@ export async function runDeepScans( state, merged, pending.map((result) => result.scanId), - combineScanCoverage([...completed.values()], [], state.legacy?.coverage), + combineScanCoverage([...completed.values()]), ); await save(); await input.publish(state.aggregate!); @@ -606,7 +529,7 @@ export async function runDeepScans( const batch = unfinished.slice(0, settings.workers); while ( batch.length < settings.workers && - previousRuns + state.passes.length < settings.maxDiscoveryRuns + state.passes.length < settings.maxDiscoveryRuns ) { batch.push(reservePass(state)); } diff --git a/sdk/typescript/src/execution-preparation.ts b/sdk/typescript/src/execution-preparation.ts index 4d4061689..01839d153 100644 --- a/sdk/typescript/src/execution-preparation.ts +++ b/sdk/typescript/src/execution-preparation.ts @@ -18,6 +18,7 @@ import { modelProviderConfigOverride, scanModelProvider, scanCompositionOverrides, + setScanSubagentBudget, writeCodexConfig, type JsonObject, } from "./config.js"; @@ -434,11 +435,9 @@ export async function prepareAmbientExecution( ); } } - const selectedEnvironment = { - ...(configuredProvider - ? environment - : selectedScanEnvironment(environment, auth, modelProvider)), - }; + const selectedEnvironment = configuredProvider + ? environment + : selectedScanEnvironment(environment, auth, modelProvider); if (!configuredProvider && selectedEnvironment["CODEX_API_KEY"]?.trim()) delete selectedEnvironment["OPENAI_API_KEY"]; @@ -559,17 +558,7 @@ export function prepareMergeExecution( subagents: number, ): PreparedExecution { const config = deepWorkerConfig(session.sessionConfig); - const features = isRecord(config["features"]) ? config["features"] : {}; - config["features"] = { - ...features, - multi_agent_v2: { - ...(isRecord(features["multi_agent_v2"]) - ? features["multi_agent_v2"] - : {}), - enabled: true, - max_concurrent_threads_per_session: subagents + 1, - }, - }; + setScanSubagentBudget(config, subagents); return { ...session, policy: "merge", sessionConfig: config }; } diff --git a/sdk/typescript/src/scan-merge.ts b/sdk/typescript/src/scan-merge.ts index 940e66ed7..270bcbad3 100644 --- a/sdk/typescript/src/scan-merge.ts +++ b/sdk/typescript/src/scan-merge.ts @@ -30,7 +30,6 @@ export interface ScanMergeInput { export interface ScanMergeResult { aggregate: ScanAggregate; - newFindings: number; /** Each novel issue belongs to the earliest input that discovered it. */ newFindingScanIds: string[]; } @@ -164,35 +163,14 @@ function reconcileScanMerge( } return identity; }; - let sourcesByIdentity: Map> | undefined; const retainSources = () => { const claimed = new Set(); for (const finding of aggregate.findings) { const provenance = finding.provenance; - let refs = provenance.sourceFindingIds; - if (refs === undefined) { - if (sourcesByIdentity === undefined) { - sourcesByIdentity = new Map(); - for (const entry of sources) { - const identity = identityOf(entry[1]); - const group = sourcesByIdentity.get(identity) ?? []; - group.push(entry); - sourcesByIdentity.set(identity, group); - } - } - const matches = sourcesByIdentity.get(identityOf(finding)) ?? []; - if ( - new Set(matches.map(([, source]) => JSON.stringify(source))).size > 1 - ) { - throw new Error( - "Scan merge has ambiguous source findings; preserve each sourceFindingIds reference explicitly.", - ); - } - refs = matches.map(([id]) => id); - } - if (refs.length === 0) + const refs = provenance.sourceFindingIds; + if (!refs?.length) throw new Error( - "Scan merge contains a finding with no assigned source finding.", + "Scan merge requires explicit sourceFindingIds for every finding.", ); for (const id of refs) { if (!sources.has(id)) @@ -323,7 +301,6 @@ function reconcileScanMerge( } return { aggregate: structuredClone(aggregate), - newFindings: newFindings.length, newFindingScanIds: inputs .filter((_, index) => novelInputs.has(index)) .map((input) => input.scanId), diff --git a/sdk/typescript/src/scan-publication.ts b/sdk/typescript/src/scan-publication.ts index 53143485b..42655c03d 100644 --- a/sdk/typescript/src/scan-publication.ts +++ b/sdk/typescript/src/scan-publication.ts @@ -14,11 +14,28 @@ import { requireScanFile, type ScanExpectation, } from "./contract.js"; -import { IncompleteScanError, OutputDirectoryError } from "./errors.js"; +import { + CodexSecurityError, + IncompleteScanError, + OutputDirectoryError, +} from "./errors.js"; import { ScanPermissionError } from "./scan-execution.js"; import { ScanResult, type TurnResultMetadata } from "./result.js"; -import type { ScanCost } from "./cost.js"; +import { + addScanCosts, + scanCostUsage, + ScanCostTracker, + type ScanCost, +} from "./cost.js"; +import { ScanCostTrackingError } from "./deep-scan.js"; +import { + compositionCheckpointFromWorkbench, + type DeepScanCheckpointSummary, +} from "./deep-scan-checkpoint.js"; +import type { SavedScanRecord } from "./workbench-types.js"; +import { throwIfAborted } from "./scan-events.js"; import type { JsonObject } from "./config.js"; +import { findScanSession } from "./scan-logs.js"; export interface CompletedScanTurn { threadId: string | null; @@ -34,6 +51,225 @@ export interface ScanPublicationContext { workbench: (args: readonly string[]) => Promise; } +/** This is only a read-path hint; the workbench still validates the complete seal and binding. */ +export async function hasSealedScanArtifacts( + scanDir: string, + signal: AbortSignal, +): Promise { + let manifest: unknown; + try { + manifest = JSON.parse( + ( + await readScanFile( + scanDir, + "scan-manifest.json", + "scan-manifest.json", + signal, + ) + ).toString("utf8"), + ); + } catch (error) { + if ( + error instanceof Error && + isRecord(error.cause) && + error.cause["code"] === "ENOENT" + ) + return false; + throw error; + } + const scan = isRecord(manifest) ? manifest["scan"] : undefined; + return ( + isRecord(scan) && + (scan["sealedAt"] != null || + (Array.isArray(scan["artifacts"]) && scan["artifacts"].length > 0)) + ); +} + +/** Missing continuation metadata does not establish zero prior work. */ +export function restorePriorScanCosts( + costs: Map | null>, + checkpoint: DeepScanCheckpointSummary | null, + resumeThreadId: unknown, + scanDir: string, + maxCostUsd?: number, +): void { + if (checkpoint?.legacy) costs.set("legacy", checkpoint.legacy.cost ?? null); + if ( + checkpoint?.costUnavailable || + (typeof resumeThreadId !== "string" && + checkpoint !== null && + (checkpoint.mergedScanIds.length > 0 || + checkpoint.passes.some((pass) => pass.completed))) + ) { + costs.set("previous-work", null); + if (maxCostUsd !== undefined) + throw new ScanCostTrackingError( + "A prior scan session is unavailable; its cost limit cannot be verified.", + scanDir, + ); + } +} + +/** Recover publication accounting from saved records and logs without creating a model session. */ +export async function readSealedScanTurn( + context: Omit & { + codexHome: string; + model: string; + startedAt: unknown; + maxCostUsd?: number; + onTrackingError(error: unknown): void; + onCost(cost: Readonly): void; + }, +): Promise< + CompletedScanTurn & { cost: ScanCost | null; resumeThreadId: string | null } +> { + const { scanId, scanDir, expectation, model, codexHome, workbench, signal } = + context; + const mode = expectation.mode; + const costs = new Map | null>(); + const completeCost = (current: Readonly | null): ScanCost | null => + [...costs.values()].includes(null) + ? null + : [...costs.values()].reduce( + (total, cost) => (cost === null ? total : addScanCosts(total, cost)), + current === null ? null : { ...current }, + ); + const measure = async (threadId: string | null, directory: string) => { + const tracker = new ScanCostTracker({ + codexHome, + model, + repository: expectation.repository, + scanDirectory: directory, + }); + if (threadId !== null) tracker.start(threadId); + const snapshot = await tracker.stop().catch((error: unknown) => { + context.onTrackingError(error); + return { cost: null, usage: null }; + }); + throwIfAborted(signal, scanDir); + return snapshot; + }; + const saved = await workbench(["get-scan", "--scan-id", scanId]); + const savedScan = saved["scan"] as SavedScanRecord; + const checkpoint = compositionCheckpointFromWorkbench(saved); + let resumeThreadId = savedScan.continuationThreadId; + const historicalCost = async (threadId: string) => { + const session = await findScanSession(codexHome, threadId).catch( + (error: unknown) => { + context.onTrackingError(error); + return null; + }, + ); + const startedAt = + typeof context.startedAt === "string" + ? Date.parse(context.startedAt) + : NaN; + // Native owners can include earlier conversation work, even from this directory. + if ( + session?.workingDirectory !== scanDir || + session.startedAt === null || + !Number.isFinite(startedAt) || + session.startedAt < startedAt + ) + return null; + return (await measure(threadId, scanDir)).cost; + }; + restorePriorScanCosts( + costs, + checkpoint, + resumeThreadId, + scanDir, + context.maxCostUsd, + ); + // Legacy cost already includes its origin session; do not count that session again. + const threadId = + typeof resumeThreadId === "string" + ? resumeThreadId + : (checkpoint?.legacy?.originThreadId ?? null); + const emptyComposition = + mode === "deep" && + checkpoint?.terminalReason === "capped" && + checkpoint.mergedScanIds.length === 0 && + Array.isArray(savedScan["findings"]) && + savedScan["findings"].length === 0; + if (threadId === null && !emptyComposition && !costs.has("previous-work")) + throw new CodexSecurityError( + "The sealed scan has no saved execution session.", + ); + if (checkpoint?.legacy) + costs.set( + "legacy", + checkpoint.legacy.cost ?? + (checkpoint.legacy.originThreadId + ? await historicalCost(checkpoint.legacy.originThreadId) + : null), + ); + if (checkpoint !== null) { + const children = await workbench([ + "list-scans", + "--scan-root", + join(scanDir, "artifacts/deep-scan/passes"), + ]); + for (const child of children["scans"] as SavedScanRecord[]) { + if (child.parentScanId === scanId) + costs.set(child.scanId, child.cost ?? null); + } + } + let cost: ScanCost | null = null; + if (mode === "deep" && checkpoint === null) { + cost = (await historicalCost(threadId!)) ?? savedScan.cost ?? null; + costs.set("legacy", cost); + // This retired origin was measured above; it is not a composed merge session. + resumeThreadId = null; + } + if ( + !costs.has("previous-work") && + (savedScan.progress.status === "complete" || + ![...costs.values()].includes(null)) + ) + cost ??= savedScan.cost ?? null; + if ( + typeof resumeThreadId !== "string" && + (emptyComposition || checkpoint?.legacy) + ) + cost ??= completeCost(null); + if ( + cost === null && + context.maxCostUsd !== undefined && + [...costs.values()].includes(null) + ) + throw new ScanCostTrackingError( + "The saved child scan cost is unavailable; its cost limit cannot be verified.", + scanDir, + ); + const snapshot = await measure( + resumeThreadId ?? null, + mode === "deep" + ? join(scanDir, "artifacts", "deep-scan", "merge") + : scanDir, + ); + const measuredCost = + snapshot.cost === null ? null : completeCost(snapshot.cost); + if (measuredCost && (!cost || measuredCost.estimatedUsd > cost.estimatedUsd)) + cost = measuredCost; + if (cost !== null) context.onCost(cost); + throwIfAborted(signal, scanDir); + return { + cost, + threadId, + resumeThreadId: resumeThreadId ?? null, + turnResult: { + status: "completed", + model, + usage: cost + ? scanCostUsage(cost) + : mode === "deep" + ? null + : snapshot.usage, + }, + }; +} + /** Seal and load the same contract for ordinary, composed and already-sealed scans. */ export async function publishScan( context: ScanPublicationContext, diff --git a/sdk/typescript/src/scan-registration.ts b/sdk/typescript/src/scan-registration.ts index e711d6650..c74206790 100644 --- a/sdk/typescript/src/scan-registration.ts +++ b/sdk/typescript/src/scan-registration.ts @@ -39,7 +39,6 @@ export async function registerScan(options: { "get-cli-scan-resume", "--scan-id", scanOptions.resumeScanId, - "--migrate", ]) : await workbench( [ diff --git a/sdk/typescript/src/workbench-types.ts b/sdk/typescript/src/workbench-types.ts index d90e86849..b21ca9530 100644 --- a/sdk/typescript/src/workbench-types.ts +++ b/sdk/typescript/src/workbench-types.ts @@ -14,16 +14,3 @@ export interface SavedScanRecord { cost?: ScanCost | null; [extension: string]: unknown; } - -/** Decode responses from the selected local workbench without dropping extensions. */ -export function savedScansFromWorkbench( - response: Record, -): SavedScanRecord[] { - return response["scans"] as SavedScanRecord[]; -} - -export function savedScanFromWorkbench( - response: Record, -): SavedScanRecord { - return response["scan"] as SavedScanRecord; -} diff --git a/sdk/typescript/tests-ts/api-cancellation-order.test.ts b/sdk/typescript/tests-ts/api-cancellation-order.test.ts index 527450bf4..31fa4b816 100644 --- a/sdk/typescript/tests-ts/api-cancellation-order.test.ts +++ b/sdk/typescript/tests-ts/api-cancellation-order.test.ts @@ -1,14 +1,13 @@ -import { mkdir } from "node:fs/promises"; -import { join } from "node:path"; import { afterEach, expect, test } from "bun:test"; import { ScanCostTrackingError } from "../src/deep-scan.js"; import { ScanPermissionError } from "../src/scan-execution.js"; import type { WorkbenchCommandOptions } from "../src/runtime.js"; -import { mockWorkbench, TestClient } from "./support/api-client.js"; import { - createApiTestFixtures, - preparedRuntime, -} from "./support/api-events.js"; + cancellationSetup, + mockWorkbench, + TestClient, +} from "./support/api-client.js"; +import { createApiTestFixtures } from "./support/api-events.js"; const fixtures = createApiTestFixtures(); afterEach(fixtures.cleanup); @@ -16,29 +15,16 @@ afterEach(fixtures.cleanup); test.each(["permission", "cost tracking"] as const)( "preserves an earlier %s failure when the caller cancels during cleanup", async (kind) => { - const root = await fixtures.temporaryDirectory(); - const repository = join(root, "repository"); - const codexHome = join(root, "codex-home"); - const scanDir = join(root, "scan"); - await Promise.all( - [repository, codexHome, scanDir].map((path) => - mkdir(path, { mode: 0o700 }), - ), - ); + const { repository, scanDir, commands, controller, dependencies } = + await cancellationSetup(await fixtures.temporaryDirectory()); const failure = kind === "permission" ? new ScanPermissionError("Selected permissions could not be verified") : new ScanCostTrackingError("Child usage is unavailable", scanDir); - const controller = new AbortController(); - const commands: Array = []; const client = new TestClient( {}, { - environment: {}, - prepareRuntime: async () => preparedRuntime(codexHome), - resolvePluginPython: async () => "/managed/python", - prepareOutputDir: async () => scanDir, - repositoryRevision: async () => "deadbeef", + ...dependencies, runWorkbench: async ( options: WorkbenchCommandOptions, args: readonly string[], diff --git a/sdk/typescript/tests-ts/api-deep-composition.test.ts b/sdk/typescript/tests-ts/api-deep-composition.test.ts index acd551b6a..1121dda3a 100644 --- a/sdk/typescript/tests-ts/api-deep-composition.test.ts +++ b/sdk/typescript/tests-ts/api-deep-composition.test.ts @@ -1,7 +1,6 @@ import type { SemanticScan, SemanticFinding } from "../src/semantic-models.js"; import { randomUUID } from "node:crypto"; import * as childProcess from "node:child_process"; -import { execFileSync } from "node:child_process"; import { existsSync } from "node:fs"; import { appendFile, @@ -9,7 +8,6 @@ import { mkdtemp, readFile, realpath, - rename, rm, writeFile, } from "node:fs/promises"; @@ -23,24 +21,13 @@ import { afterEach, expect, spyOn, test } from "bun:test"; import { build } from "esbuild"; import { CodexSecurity, type ScanOptions } from "../src/api.js"; import type { JsonObject } from "../src/config.js"; -import { estimateScanCost, type ScanSessionEvent } from "../src/cost.js"; -import type { ScanCost } from "../src/cost-model.js"; +import type { ScanSessionEvent } from "../src/cost.js"; import type { ScanActivity } from "../src/scan-activity.js"; -import { - prepareScanArtifactRestorer, - runWorkbench, - type WorkbenchCommandOptions, -} from "../src/runtime.js"; -import { prepareSemanticScanDraft } from "../src/scan-semantics.js"; -import { - DEEP_SCAN_CHECKPOINT, - ScanCostTrackingError, -} from "../src/deep-scan.js"; +import { runWorkbench, type WorkbenchCommandOptions } from "../src/runtime.js"; +import { publishDraft } from "./support/scan-publication.js"; +import { DEEP_SCAN_CHECKPOINT } from "../src/deep-scan.js"; import { ScanTransportClosedError } from "../src/scan-execution.js"; -import { - ScanCostLimitExceededError, - ScanInterruptedError, -} from "../src/errors.js"; +import { ScanCostLimitExceededError } from "../src/errors.js"; import type { ScanProgress } from "../src/worker-progress.js"; import { readSavedScanLogs, type ScanLogSource } from "../src/scan-logs.js"; import { PLUGIN_ROOT } from "./plugin-root.js"; @@ -107,119 +94,20 @@ afterEach(async () => { }); test.each([ - { workers: 1, budget: false, emptyDeadline: true }, - { workers: 1, budget: false, emptyDeadline: true, measuredChild: true }, - { - workers: 1, - budget: false, - emptyDeadline: true, - measuredChild: true, - lostCompletion: true, - }, - { - workers: 1, - budget: false, - emptyDeadline: true, - measuredChild: true, - lostCompletion: true, - usage: "missing-child", - }, - { workers: 1, budget: false, emptyDeadline: true, lostCompletion: true }, - ...[ - "changed", - "renamed-file", - "removed-file", - "renamed-directory", - "removed-directory", - ].map((knowledge) => ({ workers: 1, budget: false, knowledge })), - { workers: 1, budget: false, provider: undefined }, - { workers: 2, budget: false, provider: undefined }, - { workers: 1, budget: true, provider: undefined }, - { workers: 1, budget: true, firstChildBudget: true }, - { workers: 1, budget: false, provider: { env_key: "OPENAI_API_KEY" } }, - { - workers: 1, - budget: false, - provider: { auth: { type: "command", command: "synthetic-auth-provider" } }, - }, - { workers: 1, budget: false, native: "feedback" }, - { - workers: 1, - budget: false, - native: "discovery", - provider: { env_key: "PROVIDER_KEY" }, - }, - { workers: 1, budget: false, native: "discovery", userCancel: true }, - { - workers: 1, - budget: false, - native: "sealed", - provider: { env_key: "PROVIDER_KEY" }, - }, - { workers: 1, budget: false, trackingFailure: true }, - { workers: 1, budget: false, artifactFailure: "directory" }, - { workers: 1, budget: false, artifactFailure: "draft" }, - { workers: 1, budget: false, artifactFailure: "checkpoint" }, - { workers: 1, budget: false, cleanupFailure: true }, - { - workers: 1, - budget: false, - artifactFailure: "publication", - cleanupFailure: true, - }, - { workers: 1, budget: false, logFailure: true }, - { workers: 1, budget: false, sessionFailure: true }, - { workers: 1, budget: false, usage: "missing-merge" }, - { workers: 1, budget: false, native: "sealed", usage: "missing-merge" }, - { workers: 1, budget: false, usage: "unreported-cache" }, - { workers: 1, budget: false, usage: "missing-child" }, - { workers: 1, budget: false, usage: "missing-child", requiredCost: true }, - { workers: 1, budget: false, native: "discovery", usage: "missing-child" }, - { workers: 1, budget: false, native: "sealed", usage: "missing-child" }, + { budget: false }, + { budget: true }, + { budget: true, firstChildBudget: true }, + { budget: false, native: "discovery" }, + { budget: false, native: "discovery", provider: { env_key: "PROVIDER_KEY" } }, + { budget: false, native: "sealed", provider: { env_key: "PROVIDER_KEY" } }, ] as { - workers: number; budget: boolean; firstChildBudget?: boolean; - trackingFailure?: boolean; - artifactFailure?: "directory" | "draft" | "checkpoint" | "publication"; - cleanupFailure?: boolean; provider?: JsonObject; - native?: "feedback" | "discovery" | "sealed"; - usage?: "missing-merge" | "missing-child" | "unreported-cache"; - requiredCost?: boolean; - logFailure?: boolean; - sessionFailure?: boolean; - emptyDeadline?: boolean; - lostCompletion?: boolean; - knowledge?: - | "changed" - | "renamed-file" - | "removed-file" - | "renamed-directory" - | "removed-directory"; - measuredChild?: boolean; - userCancel?: boolean; + native?: "discovery" | "sealed"; }[])( "Deep composes sealed ordinary scans and preserves a budgeted parent: %j", - async ({ - workers, - budget, - firstChildBudget, - provider, - native, - trackingFailure, - artifactFailure, - cleanupFailure, - usage, - requiredCost, - logFailure, - sessionFailure, - emptyDeadline, - lostCompletion, - knowledge, - measuredChild, - userCancel, - }) => { + async ({ budget, firstChildBudget, provider, native }) => { const python = Bun.which("python3") ?? Bun.which("python"); if (python === null) throw new Error("Python is required for this test."); const root = await realpath( @@ -229,16 +117,6 @@ test.each([ const repo = join(root, "repo"); const codexHome = join(root, "codex"); let scanDir = join(root, "scan"); - const knowledgeDirectory = knowledge?.endsWith("directory"); - const knowledgePath = join( - root, - knowledgeDirectory ? "knowledge" : "policy.md", - ); - const knowledgeDocument = knowledgeDirectory - ? join(knowledgePath, "policy.md") - : knowledgePath; - if (knowledgeDirectory) await mkdir(knowledgePath); - if (knowledge) await writeFile(knowledgeDocument, "Original policy."); await Promise.all([mkdir(repo), mkdir(codexHome)]); await Promise.all( ["app.py", "routes.py", "models.py", "helpers.py"].map((name) => @@ -343,27 +221,7 @@ process.exit(0); } const commandOptions = { python, pluginRoot, environment }; let registeredScan: ScanOptions["registeredScan"]; - let feedbackBefore: Buffer | undefined; if (native) { - if (native === "feedback") { - execFileSync(python, [ - "-c", - `import sys -from pathlib import Path -sys.path.insert(0, sys.argv[1]) -from workbench_test_support import create_saved_workspace, start_delivered_scan, write_completed_contract, run_workbench -state, repository, scans = map(Path, sys.argv[2:]) -workspace = create_saved_workspace(state, repository) -scan = start_delivered_scan(state, '--workspace-id', workspace['id'], '--scan-root', str(scans))['results'] -write_completed_contract(Path(scan['scanDir']), scan['scanId'], repository, relative_path='app.py') -completed = run_workbench(state, 'complete-scan', '--scan-id', scan['scanId'])['scan'] -run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['findings'][0]['occurrenceId'], '--status', 'closed', '--close-reason', 'false_positive', '--note', 'The synthetic route verifies the session.')`, - join(pluginRoot, "tests"), - environment.CODEX_SECURITY_STATE_DIR, - repo, - join(root, "prior-scans"), - ]); - } const started = await runWorkbench(commandOptions, [ "begin-deep-scan", "--thread-id", @@ -383,24 +241,15 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding threadId: "native-owner", handoffClaimToken: scan["handoffClaimToken"] as string, }; - if (native === "feedback") - feedbackBefore = await readFile( - join(scanDir, "artifacts/01_context/false_positive_feedback.json"), - ); } let controller = new AbortController(); let interrupted = false; let savedExecutionThread: string | undefined; let sealedArtifacts: Map> | undefined; const registrations = new Map(); - const costs: ScanCost[] = []; const followUpThreads: string[] = []; const previousFollowUp = native === "sealed" ? randomUUID() : undefined; - const finishedFollowUps: string[] = []; - const warnings: string[] = []; let childTurns = 0; - let mergeAttempts = 0; - let sessionWrites = 0; let threadCount = 0; const progressRuns: ScanProgress[][] = []; let progress: ScanProgress[]; @@ -454,12 +303,10 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding userContext: "Inspect the synthetic source.", }, recipe: nativeRecipe ?? { - postScanPrompt: emptyDeadline - ? undefined - : "Post-scan instructions once.", + postScanPrompt: "Post-scan instructions once.", }, savedDeepScanSettings: { - workers, + workers: 1, subagents: 3, stopAfterNoNew: 2, maxDiscoveryRuns: 4, @@ -515,60 +362,7 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding }), ...prepared?.client.dependencies, resolvePluginPython: async () => python, - prepareScanArtifactRestorer: async (...args) => { - const writer = await prepareScanArtifactRestorer(...args); - return { - ...writer, - async prepareDirectory(path) { - if (artifactFailure === "directory") - throw new Error("Synthetic directory write failure."); - await writer.prepareDirectory(path); - }, - async restore(path, contents) { - if (artifactFailure === "draft" && path.startsWith("drafts/")) - throw new Error("Synthetic draft write failure."); - if (logFailure && path.endsWith("execution-threads.json")) - throw new Error("Synthetic session index write failure."); - await writer.restore(path, contents); - }, - async remove(path) { - if (cleanupFailure && path.endsWith(".checkpoint.json")) - throw new Error("Synthetic staging cleanup failure."); - await writer.remove(path); - }, - }; - }, runWorkbench: async (options, args, input) => { - if ( - sessionFailure && - args[0] === "set-scan-thread" && - registrations.get(args[args.indexOf("--scan-id") + 1]!)?.[ - "mode" - ] === "deep" && - ++sessionWrites === 1 - ) - throw new Error("Synthetic session metadata write failure."); - // Model a process exit after sealing: its catch block cannot persist parent cost. - if ( - measuredChild && - lostCompletion && - interrupted && - args[0] === "preserve-scan-results" && - registrations.get(args[args.indexOf("--scan-id") + 1]!)?.[ - "mode" - ] === "deep" - ) - return {}; - if ( - artifactFailure === "checkpoint" && - args[0] === "save-scan-artifact" - ) - throw new Error("Synthetic checkpoint write failure."); - if ( - artifactFailure === "publication" && - args[0] === "write-scan-draft" - ) - throw new Error("Synthetic publication write failure."); const result = await runWorkbench(options, args, input); const id = args.includes("--scan-id") ? args[args.indexOf("--scan-id") + 1] @@ -585,97 +379,9 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding recipe: JSON.parse(input!).recipe, }); workbenches.set(scanId, options); - if (knowledge && JSON.parse(input!).recipe.mode === "deep") { - if (knowledge.startsWith("renamed")) - await rename(knowledgePath, `${knowledgePath}.moved`); - else if (knowledge.startsWith("removed")) - await rm(knowledgePath, { recursive: true }); - else await writeFile(knowledgeDocument, "Changed policy."); - } - if (emptyDeadline && JSON.parse(input!).recipe.mode === "deep") - result["startedAt"] = "2020-01-01T00:00:00Z"; - if (measuredChild && JSON.parse(input!).recipe.mode === "deep") { - const directory = "artifacts/deep-scan/passes/pass-1"; - const childDirectory = join(scanDir, directory); - await mkdir(childDirectory, { recursive: true, mode: 0o700 }); - const recipe = JSON.parse(input!).recipe; - recipe.mode = "standard"; - delete recipe.deepScan; - const child = await runWorkbench( - options, - [ - "register-cli-scan", - "--repository", - repo, - "--scan-dir", - childDirectory, - "--parent-scan-id", - scanId, - "--registration-json-stdin", - ], - JSON.stringify({ recipe }), - ); - const childId = child["scanId"] as string; - registrations.set(childId, { ...child, mode: "standard" }); - await runWorkbench(options, [ - "set-scan-thread", - "--scan-id", - childId, - "--thread-id", - "synthetic-interrupted-child", - ]); - await runWorkbench(options, [ - "fail-scan", - "--scan-id", - childId, - "--message", - "Discovery deadline reached.", - ...(usage === "missing-child" - ? [] - : [ - "--cost-json", - JSON.stringify( - estimateScanCost("gpt-6-astra", { - input_tokens: 100, - output_tokens: 20, - }), - ), - ]), - ]); - await runWorkbench( - options, - [ - "save-scan-artifact", - "--scan-id", - scanId, - "--artifact-path", - DEEP_SCAN_CHECKPOINT, - ], - JSON.stringify({ - version: 2, - startedAt: result["startedAt"], - passes: [{ directory, scanId: childId, failed: true }], - mergedScanIds: [], - aggregate: null, - noNewStreak: 0, - consecutiveErrors: 0, - }), - ); - } } if (args[0] === "get-cli-scan-resume") workbenches.set(result["scanId"] as string, options); - if ( - (knowledge || lostCompletion) && - !interrupted && - args[0] === "prepare-scan-completion" && - registrations.get(id!)?.["mode"] === "deep" - ) { - interrupted = true; - controller.abort( - new ScanTransportClosedError("synthetic_lost_completion"), - ); - } if ( native === "sealed" && !interrupted && @@ -729,17 +435,13 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding threadCount += 1; const thread = { id: savedThreadId, - async runStreamed( - prompt: string, - turnOptions?: { signal?: AbortSignal }, - ) { + async runStreamed(prompt: string) { const record = registrations.get(id)!; const mode = record["mode"] as string; if ( mode === "deep" && prompt !== "Post-scan instructions once." ) { - mergeAttempts += 1; const mergePath = join( record["scanDir"] as string, "artifacts/deep-scan/merge-inputs.json", @@ -758,17 +460,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding expect(scan).toMatchObject({ scanId: id, findings: [] }); } } - if (knowledge) { - expect( - await readFile( - join( - env["CODEX_SECURITY_KNOWLEDGE_BASE"]!, - "0-policy.md.txt", - ), - "utf8", - ), - ).toBe("Original policy."); - } turns.push({ id, mode, @@ -824,22 +515,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding async function* events(): AsyncGenerator { thread.id ??= randomUUID(); const sessionHome = env["CODEX_HOME"]!; - if (trackingFailure) { - expect(mode).toBe("standard"); - await writeFile( - join(sessionHome, "sessions"), - "not a directory", - ); - yield { type: "thread.started", thread_id: thread.id }; - const signal = turnOptions!.signal!; - if (!signal.aborted) - await new Promise((resolve) => - signal.addEventListener("abort", () => resolve(), { - once: true, - }), - ); - throw signal.reason; - } await mkdir(join(sessionHome, "sessions"), { recursive: true, }); @@ -917,11 +592,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding ); } yield { type: "thread.started", thread_id: thread.id }; - if ( - artifactFailure === "checkpoint" && - prompt === "Post-scan instructions once." - ) - throw new Error("Synthetic follow-up failure."); if (mode === "standard") { for (const type of [ "item.started", @@ -968,11 +638,9 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding }, }) + "\n", ); - const error = userCancel - ? new Error("Synthetic user cancellation") - : new ScanTransportClosedError( - "mcp_transport_closed", - ); + const error = new ScanTransportClosedError( + "mcp_transport_closed", + ); controller.abort(error); throw error; } @@ -1023,39 +691,12 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding deferred: [], }, }; - const documents = prepareSemanticScanDraft( - { - targetContract: record["contract"] as JsonObject, - mode: "standard", - targetRevision: record["targetRevision"] as string, - }, + await publishDraft( + (args) => runWorkbench(workbenches.get(id)!, [...args]), + record, + "standard", draft, ); - const draftPath = join( - directory, - "drafts", - randomUUID() + ".json", - ); - const checkpointPath = join( - directory, - "drafts", - randomUUID() + ".checkpoint.json", - ); - await mkdir(join(directory, "drafts"), { - recursive: true, - mode: 0o700, - }); - await writeFile(draftPath, JSON.stringify(documents)); - await writeFile(checkpointPath, JSON.stringify(draft)); - await runWorkbench(workbenches.get(id)!, [ - "write-scan-draft", - "--scan-id", - id, - "--draft-path", - draftPath, - "--checkpoint-path", - checkpointPath, - ]); } yield { type: "item.completed", @@ -1068,31 +709,20 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding : "Complete", }, }; - if (prompt === "Post-scan instructions once.") - finishedFollowUps.push(thread.id); yield { type: "turn.completed", - usage: - (usage === "missing-merge" && mode === "deep") || - (usage === "missing-child" && + usage: { + input_tokens: + budget && mode === "standard" && - childTurns === 1) - ? null - : { - input_tokens: - budget && - mode === "standard" && - childTurns === (firstChildBudget ? 1 : 2) - ? 100000 - : 10, - cached_input_tokens: 0, - output_tokens: 3, - cache_write_input_tokens: 0, - ...(usage === "unreported-cache" - ? { cache_write_input_tokens_reported: false } - : {}), - reasoning_output_tokens: 0, - }, + childTurns === (firstChildBudget ? 1 : 2) + ? 100000 + : 10, + cached_input_tokens: 0, + output_tokens: 3, + cache_write_input_tokens: 0, + reasoning_output_tokens: 0, + }, } as ThreadEvent; } return { events: events() }; @@ -1131,12 +761,10 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding try { const scanOptions: ScanOptions = { mode: "deep", - ...(knowledge ? { knowledgeBasePaths: [knowledgePath] } : {}), preserveProviderEnvironment: provider !== undefined, - workers, + workers: 1, subagents: 3, stopAfterNoNew: 2, - ...(sessionFailure ? { stopAfterConsecutiveErrors: 1 } : {}), maxDiscoveryRuns: 4, maxTimeHours: 1, outputDir: scanDir, @@ -1145,26 +773,11 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding ? { safetyIdentifier: "saved-native-identifier" } : {}), scanPrompt: "Inspect the synthetic source.", - ...(budget - ? { maxCostUsd: 0.001 } - : trackingFailure || requiredCost - ? { maxCostUsd: 1 } - : {}), - postScanPrompt: emptyDeadline - ? undefined - : "Post-scan instructions once.", + ...(budget ? { maxCostUsd: 0.001 } : {}), + postScanPrompt: "Post-scan instructions once.", onProgress: (update) => progress.push(update), onActivity: (activity) => workerRun.activities.push(activity), onSessionEvent: (event) => workerRun.sessions.push(event), - onCost: (cost) => costs.push(cost), - onWarning: (message) => { - warnings.push(message); - if (trackingFailure && message.startsWith("Deep Scan pass ")) - controller.abort( - new Error("Unexpected retry after required metering failure."), - ); - else console.error(message); - }, }; const run = () => { progress = []; @@ -1198,87 +811,8 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding expect(scan.continuationThreadId).not.toBe(id); } }; - if (emptyDeadline) { - if (lostCompletion) { - await expect(run()).rejects.toBeInstanceOf(ScanTransportClosedError); - controller = new AbortController(); - scanOptions.resumeScanId = [...registrations.keys()][0]!; - if (measuredChild) { - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - scanOptions.resumeScanId, - ]); - expect((saved["scan"] as JsonObject)["cost"]).toBeUndefined(); - } - } - const result = await run(); - expect(result.threadId).toBeNull(); - expect(result.findings.findings).toEqual([]); - expect(result.coverage.completeness).toBe("partial"); - expect(turns).toEqual([]); - expect(mergeAttempts).toBe(0); - expect(registrations.size).toBe(measuredChild ? 2 : 1); - if (measuredChild && usage !== "missing-child") { - const expectedCost = estimateScanCost("gpt-6-astra", { - input_tokens: 100, - output_tokens: 20, - }); - expect(result.cost).toEqual(expectedCost); - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - result.manifest.scan.id, - ]); - expect( - (saved["scan"] as JsonObject)["cost"] as unknown as ScanCost, - ).toEqual(expectedCost!); - expect(result.turnResult).toMatchObject({ - usage: { input_tokens: 100, output_tokens: 20 }, - }); - } else if (measuredChild) { - expect(result.cost).toBeNull(); - expect(result.turnResult.usage).toBeNull(); - } - const manifest = await readFile(join(scanDir, "scan-manifest.json")); - scanOptions.resumeScanId = result.manifest.scan.id; - await expect(run()).rejects.toThrow("Resume requires a running scan"); - expect(await readFile(join(scanDir, "scan-manifest.json"))).toEqual( - manifest, - ); - expect(turns).toEqual([]); - return; - } - if (knowledge) { - await expect(run()).rejects.toBeInstanceOf(ScanTransportClosedError); - const count = turns.length; - expect(count).toBeGreaterThan(0); - const scanId = [...registrations].find( - ([, value]) => value["mode"] === "deep", - )![0]; - const manifest = await readFile(join(scanDir, "scan-manifest.json")); - controller = new AbortController(); - scanOptions.resumeScanId = scanId; - if (knowledgeDirectory) await mkdir(knowledgePath); - await writeFile(knowledgeDocument, "Changed policy."); - await expect(run()).rejects.toThrow("knowledge base changed"); - expect(turns).toHaveLength(count); - expect(await readFile(join(scanDir, "scan-manifest.json"))).toEqual( - manifest, - ); - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - scanId, - ]); - expect(saved["scan"]).toMatchObject({ - progress: { status: "running" }, - }); - return; - } if (firstChildBudget) { await expect(run()).rejects.toBeInstanceOf(ScanCostLimitExceededError); - expect(mergeAttempts).toBe(0); expect(turns.map((turn) => turn.mode)).toEqual(["standard"]); expect(registrations.size).toBe(2); const parent = [...registrations.values()].find( @@ -1320,103 +854,16 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding ).toMatchObject({ completeness: "partial" }); return; } - if (artifactFailure) { - await expect(run()).rejects.toThrow( - `Synthetic ${artifactFailure} write failure.`, - ); - if (cleanupFailure) - expect(warnings).toContain( - "Could not clean up after the Codex Security scan: Synthetic staging cleanup failure.", - ); - if (artifactFailure === "directory") - expect([threadCount, turns.length]).toEqual([0, 0]); - expect( - commands.filter(({ command }) => command === "write-scan-draft"), - ).toEqual([]); - const parent = [...registrations.values()].find( - ({ mode }) => mode === "deep", - )!; - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - parent["scanId"] as string, - ]); - expect(saved["scan"]).toMatchObject({ progress: { status: "failed" } }); - if (artifactFailure !== "directory") - await assertFollowUpLogs(saved["scan"] as ScanLogSource); - if (artifactFailure === "checkpoint") { - expect(saved["compositionCheckpoint"]).toBeNull(); - expect( - (saved["scan"] as ScanLogSource).continuationThreadId, - ).toBeNull(); - } - return; - } - if (trackingFailure) { - await expect(run()).rejects.toBeInstanceOf(ScanCostTrackingError); - expect(turns).toHaveLength(1); - expect( - [...registrations.values()].map((record) => record["mode"]).sort(), - ).toEqual(["deep", "standard"]); - expect( - JSON.parse( - await readFile(join(scanDir, DEEP_SCAN_CHECKPOINT), "utf8"), - ), - ).toMatchObject({ - terminalReason: "failed", - mergedScanIds: [], - consecutiveErrors: 0, - }); - for (const scanId of registrations.keys()) { - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - scanId, - ]); - expect(saved["scan"]).toMatchObject({ - progress: { status: "failed" }, - }); - } - return; - } - if (requiredCost) { - await expect(run()).rejects.toBeInstanceOf(ScanCostTrackingError); - expect(turns).toHaveLength(1); - const parent = [...registrations.values()].find( - ({ mode }) => mode === "deep", - )!; - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - parent["scanId"] as string, - ]); - expect(saved["scan"]).toMatchObject({ progress: { status: "failed" } }); - expect(saved["compositionCheckpoint"]).toMatchObject({ - terminalReason: "failed", - consecutiveErrors: 0, - noNewStreak: 0, - }); - return; - } if (native === "discovery" || native === "sealed") { - if (userCancel) - await expect(run()).rejects.toBeInstanceOf(ScanInterruptedError); - else - await expect(run()).rejects.toBeInstanceOf(ScanTransportClosedError); + await expect(run()).rejects.toBeInstanceOf(ScanTransportClosedError); const saved = await runWorkbench(commandOptions, [ "get-scan", "--scan-id", registeredScan!.scanId, ]); expect(saved["scan"]).toMatchObject({ - progress: { status: userCancel ? "canceled" : "running" }, + progress: { status: "running" }, }); - if (userCancel) { - expect(saved["compositionCheckpoint"]).toMatchObject({ - terminalReason: "canceled", - }); - return; - } savedExecutionThread = (saved["scan"] as JsonObject)[ "continuationThreadId" ] as string; @@ -1515,29 +962,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding } } const result = await run(); - if (cleanupFailure) - expect(warnings).toContain( - "Could not clean up after the Codex Security scan: Synthetic staging cleanup failure.", - ); - if (usage) { - const saved = await runWorkbench(commandOptions, [ - "get-scan", - "--scan-id", - result.manifest.scan.id, - ]); - const scan = saved["scan"] as JsonObject; - expect(scan["progress"]).toMatchObject({ status: "complete" }); - expect(costs.at(-1)!.estimatedUsd).toBeGreaterThan(0); - if (usage === "missing-merge" || usage === "missing-child") { - expect(result.cost).toBeNull(); - expect(scan["cost"]).toBeUndefined(); - expect(scan["usage"]).toMatchObject({ coverage: "unavailable" }); - } else { - expect(result.cost).toEqual(costs.at(-1)!); - expect(result.cost!.cacheWriteInputTokensReported).toBe(false); - expect(result.cost).toEqual(scan["cost"] as unknown as ScanCost); - } - } for (const observed of workerRuns) { const labels = new Map(); for (const event of observed.sessions) { @@ -1580,20 +1004,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding expect(await readFile(join(scanDir, path))).toEqual(bytes); expect(result.manifest.scan.producer.version).toBe(version); } - if (feedbackBefore) { - expect( - await readFile( - join(scanDir, "artifacts/01_context/false_positive_feedback.json"), - ), - ).toEqual(feedbackBefore); - expect( - turns - .filter((turn) => turn.mode === "standard") - .every((turn) => - turn.prompt.includes("false_positive_feedback.json"), - ), - ).toBe(true); - } expect(result.findings.findings).toEqual([]); expect(result.coverage.completeness).toBe( budget ? "partial" : "complete", @@ -1721,7 +1131,7 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding ) .map(({ account }) => account), ).toEqual(["A", "B"]); - if (!usage) expect(result.cost!.estimatedUsd).toBeGreaterThan(0); + expect(result.cost!.estimatedUsd).toBeGreaterThan(0); expect(existsSync(join(codexHome, "sessions"))).toBe(true); expect(existsSync(join(managedHome, "sessions"))).toBe(false); expect(await readFile(join(managedHome, "auth.json"), "utf8")).toBe( @@ -1774,13 +1184,7 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding expect( turns.filter((turn) => turn.prompt === "Post-scan instructions once."), ).toHaveLength(budget ? 0 : 1); - if (logFailure) { - expect(finishedFollowUps).toEqual(followUpThreads); - expect(finishedFollowUps).toHaveLength(1); - expect(warnings).toContain( - "Could not save post-scan session: Synthetic session index write failure.", - ); - } else if (!budget) { + if (!budget) { const saved = await runWorkbench(commandOptions, [ "get-scan", "--scan-id", @@ -1790,17 +1194,6 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding expect(scan.continuationThreadId).toBe(result.threadId!); await assertFollowUpLogs(scan); } - if (sessionFailure) { - expect(sessionWrites).toBe(2); - expect(mergeAttempts).toBe(2); - expect( - warnings.filter((warning) => - warning.startsWith("Could not save scan session:"), - ), - ).toEqual([ - "Could not save scan session: Synthetic session metadata write failure.", - ]); - } if (budget) { expect(result.cost!.estimatedUsd).toBeGreaterThan(0.001); expect(result.coverage.deferred.length).toBeGreaterThan(0); @@ -1818,7 +1211,7 @@ run_workbench(state, 'set-finding-triage', '--occurrence-id', completed['finding (scan) => scan["scanId"], ); expect(listedIds).toContain(result.manifest.scan.id); - expect(listedIds).toHaveLength(native === "feedback" ? 2 : 1); + expect(listedIds).toHaveLength(1); } finally { try { await client.close(); diff --git a/sdk/typescript/tests-ts/api-permission-profile.test.ts b/sdk/typescript/tests-ts/api-permission-profile.test.ts index e4e222f48..647b8467e 100644 --- a/sdk/typescript/tests-ts/api-permission-profile.test.ts +++ b/sdk/typescript/tests-ts/api-permission-profile.test.ts @@ -7,6 +7,7 @@ import { CodexSecurity, type ScanOptions } from "../src/api.js"; import type { JsonObject } from "../src/config.js"; import { DEEP_SCAN_CHECKPOINT } from "../src/deep-scan.js"; import { ScanInterruptedError } from "../src/errors.js"; +import { createPermissionCheckedCodex } from "../src/permission-profile.js"; import { executablePathForSpawn } from "../src/runtime.js"; import { ScanPermissionError } from "../src/scan-execution.js"; import { PLUGIN_ROOT } from "./plugin-root.js"; @@ -20,7 +21,12 @@ const { temporaryDirectory, cleanup } = createApiTestFixtures(); afterEach(cleanup); type Role = "discovery" | "merge" | "standard" | "custom" | "comparison"; -type Scenario = "rejected" | "fallback" | "active"; +type Scenario = + | "rejected" + | "fallback" + | "active" + | "substituted-default" + | "substituted-profile"; async function fixture( role: Role, @@ -72,10 +78,22 @@ async function fixture( 'record({ kind: args.includes("mcp") ? "mcp" : args.includes("app-server") ? "preflight" : "exec", args, cwd: process.cwd(), surface: process.env.CODEX_SECURITY_SURFACE, profile: config.default_permissions, permissions: config.permissions, mcpServers: config.mcp_servers, context: process.env.SYNTHETIC_EXECUTION_CONTEXT, apiKey: process.env.CODEX_API_KEY });', 'if (args.includes("mcp")) { console.log("[]"); process.exit(0); }', 'if (args.includes("app-server")) {', + " const selected = config.default_permissions;", + " const profile = config.permissions[selected];", + ' profile.description = "Synthetic description"; if (profile.network) profile.network.fixtureNull = null;', + ...(scenario === "substituted-default" + ? [' config.default_permissions = ":read-only";'] + : []), + ...(scenario === "substituted-profile" + ? [ + ' for (const [path, value] of Object.entries(profile.filesystem)) if (value === "deny") delete profile.filesystem[path];', + ] + : []), ' require("node:readline").createInterface({ input: process.stdin }).on("line", (line) => {', " const request = JSON.parse(line);", + ' record({ kind: "request", method: request.method, params: request.params });', " if (request.id === undefined) return;", - ' const result = request.method === "initialize" ? {} : request.method === "config/read" ? { config } : request.method === "permissionProfile/list" ? { data: [{ id: config.default_permissions, allowed: ' + + ' const result = request.method === "initialize" ? {} : request.method === "config/read" ? { config } : request.method === "permissionProfile/list" ? request.params?.cursor !== "selected-page" ? { data: [{ id: "other-profile", allowed: true }], nextCursor: "selected-page" } : { data: [{ id: selected, allowed: ' + (role === "comparison" ? 'config.default_permissions !== "codex_security_comparison" || ' : "") + @@ -366,6 +384,27 @@ async function fixture( signal: controller.signal, }; return { + async runProtocol() { + const sdk = createPermissionCheckedCodex({ + codexPathOverride: executablePathForSpawn(executable), + env: environment, + config: { + default_permissions: "codex_security_scan", + permissions: { codex_security_scan: inheritedPermissions }, + }, + }); + const threadOptions = { workingDirectory: cwd, skipGitRepoCheck: true }; + const thread = resumed + ? sdk.resumeThread(threadId, threadOptions) + : sdk.startThread(threadOptions); + const { events } = await thread.runStreamed( + "Synthetic permission check.", + { signal: controller.signal }, + ); + for await (const _event of events) { + /* Consume the real child protocol. */ + } + }, run: (onScanStarted?: () => void) => client.run(repository, { ...options, @@ -383,23 +422,26 @@ async function fixture( warnings, completedArtifacts, customCalls: () => customCalls, - observations: async () => + observations: async (requests = false) => (await readFile(capture, "utf8")) .trim() .split("\n") .filter(Boolean) - .map((line) => JSON.parse(line)), + .map((line) => JSON.parse(line)) + .filter(({ kind }) => + requests ? kind === "request" : kind !== "request", + ), async close() { clearTimeout(timeout); try { await client.close(); - // The Codex SDK removes child listeners during cleanup; exit state is retained. - for (const child of children) - while (child.exitCode === null && child.signalCode === null) - await new Promise((resolve) => setImmediate(resolve)); } finally { spawn.mockRestore(); } + // The SDK removes child listeners during cleanup; exit state is retained. + for (const child of children) + while (child.exitCode === null && child.signalCode === null) + await new Promise((resolve) => setImmediate(resolve)); }, }; } @@ -413,6 +455,18 @@ test.each(["sdk", "cli"] as const)( const h = await fixture(role, resumed, scenario, surface); try { await expect(h.run()).rejects.toBeInstanceOf(ScanPermissionError); + const requests = await h.observations(true); + expect(requests.map(({ method }) => method)).toEqual([ + "initialize", + "initialized", + "config/read", + "permissionProfile/list", + "permissionProfile/list", + ]); + expect(requests.at(-1).params).toMatchObject({ + cursor: "selected-page", + cwd: h.cwd, + }); const observations = await h.observations(); expect( observations.filter(({ kind }) => kind === "preflight"), @@ -445,6 +499,34 @@ test.each(["sdk", "cli"] as const)( }, ); +test.each(["rejected", "substituted-default", "substituted-profile"] as const)( + "checks the paginated profile and rejects %s before executing the child", + async (scenario) => { + const h = await fixture("discovery", false, scenario, "sdk"); + try { + await expect(h.runProtocol()).rejects.toBeInstanceOf(ScanPermissionError); + const requests = await h.observations(true); + expect(requests.map(({ method }) => method)).toEqual([ + "initialize", + "initialized", + "config/read", + "permissionProfile/list", + "permissionProfile/list", + ]); + expect(requests.at(-1).params).toMatchObject({ + cursor: "selected-page", + cwd: h.cwd, + }); + expect( + (await h.observations()).filter(({ kind }) => kind === "exec"), + ).toEqual([]); + expect(h.commands).not.toContain("complete-scan"); + } finally { + await h.close(); + } + }, +); + test("forwards the caller's exact cancellation reason to an active guarded child", async () => { const h = await fixture("discovery", false, "active", "sdk"); const reason = new Error("synthetic caller cancellation"); diff --git a/sdk/typescript/tests-ts/api-workbench-environment.test.ts b/sdk/typescript/tests-ts/api-workbench-environment.test.ts index 8d74292d7..d3e7d3795 100644 --- a/sdk/typescript/tests-ts/api-workbench-environment.test.ts +++ b/sdk/typescript/tests-ts/api-workbench-environment.test.ts @@ -37,6 +37,7 @@ test.each(["runtime", "sqlite override", "database override"] as const)( )!; expect(python).not.toBeNull(); let helperCalls = 0; + let observedEnvironment: WorkbenchCommandOptions["environment"] | undefined; const client = new TestClient( { codexOverrides: { model: "unpriced-model" } }, { @@ -56,20 +57,14 @@ test.each(["runtime", "sqlite override", "database override"] as const)( input?: string, ) => { helperCalls += 1; - const database = execFileSync( - python, - [ - "-c", - "import sys; sys.path.insert(0, sys.argv[1]); from workbench_scan_usage import _codex_state_database; print(_codex_state_database())", - join(PLUGIN_ROOT, "scripts"), - ], - { - env: { ...process.env, ...options.environment }, - encoding: "utf8", - }, - ).trim(); - expect(database).toBe(expectedDatabase); expect(options.environment["CODEX_HOME"]).toBe(runtimeHome); + expect(options.environment["CODEX_SQLITE_HOME"]).toBe( + source === "sqlite override" ? overrideHome : "", + ); + expect(options.environment["CODEX_STATE_DB"]).toBe( + source === "database override" ? expectedDatabase : "", + ); + observedEnvironment = options.environment; return mockWorkbench(args, input); }, createCodex: () => { @@ -82,6 +77,19 @@ test.each(["runtime", "sqlite override", "database override"] as const)( "environment observed", ); expect(helperCalls).toBeGreaterThan(0); + const database = execFileSync( + python, + [ + "-c", + "import sys; sys.path.insert(0, sys.argv[1]); from workbench_scan_usage import _codex_state_database; print(_codex_state_database())", + join(PLUGIN_ROOT, "scripts"), + ], + { + env: { ...process.env, ...observedEnvironment }, + encoding: "utf8", + }, + ).trim(); + expect(database).toBe(expectedDatabase); } finally { await client.close(); } diff --git a/sdk/typescript/tests-ts/api.test.ts b/sdk/typescript/tests-ts/api.test.ts index 427172184..460d4f127 100644 --- a/sdk/typescript/tests-ts/api.test.ts +++ b/sdk/typescript/tests-ts/api.test.ts @@ -63,6 +63,7 @@ import { normalizeTarget } from "../src/targets.js"; import { SYNTHETIC_CREDENTIALS } from "./cli-fixtures.js"; import { INTEGRATION_TARGET, PLUGIN_ROOT } from "./plugin-root.js"; import { + cancellationSetup, mockScanRegistration, mockWorkbench, shellEnvironmentReference, @@ -668,16 +669,9 @@ describe("CodexSecurity finding validation", () => { describe("CodexSecurity orchestration", () => { test("records deeply nested caller cancellation as canceled instead of failed", async () => { - const root = await temporaryDirectory(); - const repository = join(root, "repository"); - const codexHome = join(root, "codex-home"); - const scanDir = join(root, "scan"); - await mkdir(repository); - await mkdir(codexHome); - await mkdir(scanDir, { mode: 0o700 }); - const commands: Array = []; + const { repository, scanDir, commands, controller, dependencies } = + await cancellationSetup(await temporaryDirectory()); const started = Promise.withResolvers(); - const controller = new AbortController(); let cancellationReason: unknown = new DOMException("aborted", "AbortError"); for (let depth = 0; depth < 10; depth += 1) { cancellationReason = new ScanInterruptedError( @@ -690,29 +684,7 @@ describe("CodexSecurity orchestration", () => { const client = new TestClient( {}, { - environment: {}, - prepareRuntime: async () => preparedRuntime(codexHome), - resolvePluginPython: async () => "/managed/python", - prepareOutputDir: async () => scanDir, - repositoryRevision: async () => "deadbeef", - runWorkbench: async ( - _options: unknown, - args: readonly string[], - input?: string, - ): Promise => { - commands.push(args); - if (args[0] === "register-cli-scan") { - return mockScanRegistration(args, input); - } - if (args[0] === "get-scan-feedback") { - return { - scanId: "scan_example_001", - targetId: "target_sha256_example", - falsePositives: [], - }; - } - return {}; - }, + ...dependencies, createCodex: () => ({ startThread: () => ({ id: null, @@ -759,34 +731,20 @@ describe("CodexSecurity orchestration", () => { }); test("records a workbench AbortError as canceled instead of failed", async () => { - const root = await temporaryDirectory(); - const repository = join(root, "repository"); - const codexHome = join(root, "codex-home"); - const scanDir = join(root, "scan"); - await mkdir(repository); - await mkdir(codexHome); - await mkdir(scanDir, { mode: 0o700 }); - const commands: Array = []; + const { repository, commands, controller, dependencies } = + await cancellationSetup(await temporaryDirectory()); const feedbackStarted = Promise.withResolvers(); - const controller = new AbortController(); const client = new TestClient( {}, { - environment: {}, - prepareRuntime: async () => preparedRuntime(codexHome), - resolvePluginPython: async () => "/managed/python", - prepareOutputDir: async () => scanDir, - repositoryRevision: async () => "deadbeef", + ...dependencies, runWorkbench: async ( options: unknown, args: readonly string[], input?: string, ): Promise => { commands.push(args); - if (args[0] === "register-cli-scan") { - return mockScanRegistration(args, input); - } if (args[0] === "get-scan-feedback") { feedbackStarted.resolve(); const signal = (options as { signal: AbortSignal }).signal; @@ -801,7 +759,7 @@ describe("CodexSecurity orchestration", () => { ); }); } - return {}; + return mockWorkbench(args, input); }, createCodex: () => ({ startThread: () => ({ @@ -833,40 +791,26 @@ describe("CodexSecurity orchestration", () => { }); test("records an ordinary failure as failed when cancellation follows it", async () => { - const root = await temporaryDirectory(); - const repository = join(root, "repository"); - const codexHome = join(root, "codex-home"); - const scanDir = join(root, "scan"); - await mkdir(repository); - await mkdir(codexHome); - await mkdir(scanDir, { mode: 0o700 }); - const commands: Array = []; - const controller = new AbortController(); + const { repository, commands, controller, dependencies } = + await cancellationSetup(await temporaryDirectory()); const client = new TestClient( {}, { - environment: {}, - prepareRuntime: async () => preparedRuntime(codexHome), - resolvePluginPython: async () => "/managed/python", - prepareOutputDir: async () => scanDir, - repositoryRevision: async () => "deadbeef", + ...dependencies, runWorkbench: async ( _options: unknown, args: readonly string[], input?: string, ): Promise => { commands.push(args); - if (args[0] === "register-cli-scan") { - return mockScanRegistration(args, input); - } if (args[0] === "get-scan-feedback") { const failure = new Error("underlying scan failure"); return await Promise.reject(failure).finally(() => { controller.abort("caller canceled"); }); } - return {}; + return mockWorkbench(args, input); }, createCodex: () => { throw new Error("Codex must not start after feedback failure"); @@ -893,45 +837,26 @@ describe("CodexSecurity orchestration", () => { }); test("records a client-close cancellation as canceled instead of failed", async () => { - const root = await temporaryDirectory(); - const repository = join(root, "repository"); - const codexHome = join(root, "codex-home"); - const scanDir = join(root, "scan"); - await mkdir(repository); - await mkdir(codexHome); - await mkdir(scanDir, { mode: 0o700 }); - const commands: Array = []; + const { repository, commands, dependencies } = await cancellationSetup( + await temporaryDirectory(), + ); const feedbackStarted = Promise.withResolvers(); const releaseFeedback = Promise.withResolvers(); - let client: TestClient; - - client = new TestClient( + const client = new TestClient( {}, { - environment: {}, - prepareRuntime: async () => preparedRuntime(codexHome), - resolvePluginPython: async () => "/managed/python", - prepareOutputDir: async () => scanDir, - repositoryRevision: async () => "deadbeef", + ...dependencies, runWorkbench: async ( _options: unknown, args: readonly string[], input?: string, ): Promise => { commands.push(args); - if (args[0] === "register-cli-scan") { - return mockScanRegistration(args, input); - } if (args[0] === "get-scan-feedback") { feedbackStarted.resolve(); await releaseFeedback.promise; - return { - scanId: "scan_example_001", - targetId: "target_sha256_example", - falsePositives: [], - }; } - return {}; + return mockWorkbench(args, input); }, createCodex: () => { throw new Error("Codex must not start after client close"); diff --git a/sdk/typescript/tests-ts/build-plugin.test.ts b/sdk/typescript/tests-ts/build-plugin.test.ts index bdc35a8b0..7288c9699 100644 --- a/sdk/typescript/tests-ts/build-plugin.test.ts +++ b/sdk/typescript/tests-ts/build-plugin.test.ts @@ -1,3 +1,7 @@ +import { + mcpSmokeInput, + mcpSmokeResponses, +} from "../scripts/fixtures/mcp-smoke.mjs"; import { execFile } from "node:child_process"; import { chmod, @@ -133,33 +137,9 @@ describe("bundled plugin build", () => { timeout: 10_000, }, ); - execution.child.stdin?.end( - [ - JSON.stringify({ - jsonrpc: "2.0", - id: 1, - method: "initialize", - params: { - protocolVersion: "2024-11-05", - capabilities: {}, - clientInfo: { name: "standalone-package-test", version: "1" }, - }, - }), - JSON.stringify({ jsonrpc: "2.0", method: "notifications/initialized" }), - JSON.stringify({ - jsonrpc: "2.0", - id: 2, - method: "tools/list", - params: {}, - }), - "", - ].join("\n"), - ); + execution.child.stdin?.end(mcpSmokeInput); const standalone = await execution; - const responses = standalone.stdout - .trim() - .split("\n") - .map((line) => JSON.parse(line)); + const responses = mcpSmokeResponses(standalone.stdout); expect( responses.find((response) => response.id === 2)?.result.tools, ).toEqual( diff --git a/sdk/typescript/tests-ts/deep-scan-composition.test.ts b/sdk/typescript/tests-ts/deep-scan-composition.test.ts index 552a69eb5..1f78ea07b 100644 --- a/sdk/typescript/tests-ts/deep-scan-composition.test.ts +++ b/sdk/typescript/tests-ts/deep-scan-composition.test.ts @@ -328,57 +328,6 @@ async function harness( } describe("ordinary scan composition", () => { - test("serializes concurrent child checkpoint writes", async () => { - const h = await harness({ workers: 2, stopAfterNoNew: 2 }); - const registrationsReady = Promise.withResolvers(); - const releaseWrite = Promise.withResolvers(); - let registrations = 0; - const createClient = h.input.createClient; - h.input.createClient = () => { - const client = createClient(); - return { - ...client, - run(repository, options = {}) { - return client.run(repository, { - ...options, - async onRegisteredScan(registration) { - const pending = options.onRegisteredScan?.(registration); - if (++registrations === 2) registrationsReady.resolve(); - return pending; - }, - }); - }, - }; - }; - const workbench = h.input.workbench; - let writing = 0; - let maximumWriting = 0; - h.input.workbench = async (args, contents) => { - if (args[0] !== "save-scan-artifact") return workbench(args, contents); - writing += 1; - maximumWriting = Math.max(maximumWriting, writing); - try { - const state = JSON.parse(contents!) as DeepScanCheckpoint; - if (state.passes.some((pass) => pass.scanId)) - await releaseWrite.promise; - return await workbench(args, contents); - } finally { - writing -= 1; - } - }; - const execution = runDeepScans(h.input); - try { - await Promise.race([registrationsReady.promise, execution]); - } finally { - releaseWrite.resolve(); - } - await execution; - expect(maximumWriting).toBe(1); - const state = await h.checkpoint(); - expect(state.mergedScanIds).toHaveLength(2); - expect(state.terminalReason).toBe("saturated"); - }); - test("preserves a checkpoint write failure and still saves terminal state", async () => { const h = await harness({ stopAfterNoNew: 1 }); const workbench = h.input.workbench; @@ -445,20 +394,26 @@ describe("ordinary scan composition", () => { }, ); - test.each(["failed", "canceled"] as const)( - "does not complete a %s checkpoint when its database transition was interrupted", - async (terminalReason) => { + test.each( + (["failed", "canceled"] as const).flatMap((terminalReason) => + [false, true].map((populated) => ({ terminalReason, populated })), + ), + )( + "does not execute a saved $terminalReason checkpoint (aggregate: $populated)", + async ({ terminalReason, populated }) => { const h = await harness(); const checkpoint: DeepScanCheckpoint = { version: 2, startedAt: h.input.startedAt, passes: [], mergedScanIds: [], - aggregate: { - scanId: h.input.scanId, - findings: [], - coverage: semanticCoverage(), - }, + aggregate: populated + ? { + scanId: h.input.scanId, + findings: [], + coverage: semanticCoverage(), + } + : null, noNewStreak: 0, consecutiveErrors: 0, terminalReason, @@ -1190,104 +1145,26 @@ describe("ordinary scan composition", () => { expect((await h.checkpoint()).terminalReason).toBe("capped"); }); - test("continues saved legacy counters and coverage using only new ordinary scans", async () => { - const h = await harness({ maxDiscoveryRuns: 3, stopAfterNoNew: 4 }); - const coverage = semanticCoverage({ - completeness: "partial", - surfaces: [], - deferred: [ - { id: "legacy-unresolved", reason: "Saved unresolved validation." }, - ], - }); - await h.seed({ - version: 2, - startedAt: h.input.startedAt, - passes: [], - mergedScanIds: [], - aggregate: { scanId: h.input.scanId, findings: [], coverage }, - legacy: { discoveryRuns: 2, coverage }, - noNewStreak: 3, - consecutiveErrors: 0, - }); - const costs = new Map | null>(); - h.input.onCost = (key, cost) => costs.set(key, cost); - await runDeepScans(h.input); - expect(costs.has("legacy")).toBe(true); - expect(costs.get("legacy")).toBeNull(); - costs.clear(); - await runDeepScans(h.input); - expect(costs.has("legacy")).toBe(true); - expect(costs.get("legacy")).toBeNull(); - const state = await h.checkpoint(); - expect(h.calls).toHaveLength(1); - expect(state.passes).toHaveLength(1); - expect(state.noNewStreak).toBe(4); - expect(state.terminalReason).toBe("saturated"); - expect(state.aggregate!.coverage["deferred"]).toEqual(coverage.deferred); - expect(state.aggregate!.coverage["completeness"]).toBe("partial"); - - const ready = await harness(); - await ready.seed({ - ...state, - passes: [], - mergedScanIds: [], - aggregate: { - ...state.aggregate!, - scanId: ready.input.scanId, - }, - }); - await runDeepScans(ready.input); - expect(ready.calls).toEqual([]); - expect(ready.mergeInputs).toEqual([]); - expect(ready.published.at(-1)!.coverage["deferred"]).toEqual( - coverage.deferred, - ); - }); - - test("recovers legacy paid usage once and requires it before spending under a saved limit", async () => { - const h = await harness({ maxDiscoveryRuns: 1 }); - const coverage = semanticCoverage({ completeness: "partial" }); - const state: DeepScanCheckpoint = { + test("rejects legacy active checkpoints without changing saved evidence", async () => { + const h = await harness(); + const checkpoint: DeepScanCheckpoint = { version: 2, startedAt: h.input.startedAt, passes: [], mergedScanIds: [], - aggregate: { scanId: h.input.scanId, findings: [], coverage }, - legacy: { - discoveryRuns: 1, - coverage, - originThreadId: "original-session", - }, + aggregate: null, noNewStreak: 0, consecutiveErrors: 0, + legacy: { discoveryRuns: 1, coverage: semanticCoverage() }, }; - await h.seed(state); - h.input.scanOptions.requireCost = true; - h.input.historicalCost = async () => null; + await h.seed(checkpoint); await expect(runDeepScans(h.input)).rejects.toThrow( - "original Deep Scan session logs", + "Saved legacy Deep Scans cannot be resumed; their reports remain available.", ); expect(h.calls).toEqual([]); - const cost = estimateScanCost("gpt-6-astra", { - input_tokens: 10000, - output_tokens: 2000, - })!; - let recoveries = 0; - h.input.historicalCost = async (threadId) => { - expect(threadId).toBe("original-session"); - recoveries++; - return cost; - }; - const costs = new Map(); - h.input.onCost = (id, receipt) => { - costs.set(id, receipt); - }; - await runDeepScans(h.input); - await runDeepScans(h.input); - expect(recoveries).toBe(1); - expect(costs.get("legacy")).toEqual(cost); - expect((await h.checkpoint()).legacy!.cost).toEqual(cost); - expect(h.calls).toEqual([]); + expect(h.mergeInputs).toEqual([]); + expect(h.published).toEqual([]); + expect(await h.checkpoint()).toEqual(checkpoint); }); test.each([ @@ -2238,7 +2115,6 @@ describe("ordinary scan composition", () => { passes: [{ directory, scanId }], mergedScanIds: [], aggregate: { scanId: h.input.scanId, findings: [], coverage }, - legacy: { discoveryRuns: 2, coverage }, noNewStreak: 2, consecutiveErrors: 1, mergeFailures: 1, @@ -2314,33 +2190,6 @@ describe("ordinary scan composition", () => { ); }); -test.each(["failed", "canceled"] as const)( - "does not execute a saved %s checkpoint", - async (terminalReason) => { - const h = await harness(); - const checkpoint: DeepScanCheckpoint = { - version: 2, - startedAt: h.input.startedAt, - passes: [], - mergedScanIds: [], - aggregate: null, - noNewStreak: 0, - consecutiveErrors: 0, - terminalReason, - }; - const path = join(h.input.scanDir, DEEP_SCAN_CHECKPOINT); - await mkdir(dirname(path), { recursive: true }); - await writeFile(path, JSON.stringify(checkpoint)); - await expect(runDeepScans(h.input)).rejects.toThrow( - `saved Deep Scan is ${terminalReason}`, - ); - expect(h.calls).toEqual([]); - expect(h.mergeInputs).toEqual([]); - expect(h.published).toEqual([]); - expect(JSON.parse(await readFile(path, "utf8"))).toEqual(checkpoint); - }, -); - test("caps an expired empty composition without requesting a model merge", async () => { const h = await harness(); h.input.startedAt = "2000-01-01T00:00:00.000Z"; diff --git a/sdk/typescript/tests-ts/knowledge-base.test.ts b/sdk/typescript/tests-ts/knowledge-base.test.ts index 34dab6ae4..d71988306 100644 --- a/sdk/typescript/tests-ts/knowledge-base.test.ts +++ b/sdk/typescript/tests-ts/knowledge-base.test.ts @@ -11,7 +11,6 @@ import { writeFile, } from "node:fs/promises"; import * as filesystem from "node:fs/promises"; -import { createHash } from "node:crypto"; import * as os from "node:os"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -31,23 +30,12 @@ test("ordinary passes share immutable extracted inputs while resume detects docu const source = join(root, "policy.md"); await writeFile(source, "Original policy."); const snapshot = await readKnowledgeBaseSnapshot([source]); - expect(Object.isFrozen(snapshot.documents)).toBe(true); await writeFile(source, "Updated policy."); const first = await prepareKnowledgeBase(snapshot); const second = await prepareKnowledgeBase(snapshot); const changed = await prepareKnowledgeBase([source]); try { expect(first.sha256).toBe(second.sha256); - expect(first.sha256).toBe( - createHash("sha256") - .update( - JSON.stringify({ - sources: [source], - documents: { "0-policy.md.txt": "Original policy." }, - }), - ) - .digest("hex"), - ); expect(changed.sha256).not.toBe(first.sha256); for (const prepared of [first, second]) { const [document] = await readdir(prepared.path); diff --git a/sdk/typescript/tests-ts/prompt-only-start-timeout.test.ts b/sdk/typescript/tests-ts/prompt-only-start-timeout.test.ts index 943063b71..e889ac3ef 100644 --- a/sdk/typescript/tests-ts/prompt-only-start-timeout.test.ts +++ b/sdk/typescript/tests-ts/prompt-only-start-timeout.test.ts @@ -1,40 +1,8 @@ import { expect, test } from "bun:test"; -import { loadBundledRuntime, PLUGIN_ROOT } from "./plugin-root.js"; +import { workbenchCommandTimeout } from "../../../plugins/codex-security/mcp-app/src/python_command.js"; -test("gives prompt-only scan startup the five-minute scan timeout", async () => { - const runtime = await loadBundledRuntime(); - const source = - /async function executeWorkbench\([^\n]*\) \{[\s\S]*?\n\}/u.exec( - runtime, - )?.[0]; - expect(source).toBeDefined(); - const execFileHelper = /\b(execFileAsync\d*)\(/u.exec(source ?? "")?.[1]; - expect(execFileHelper).toBeDefined(); - - const executeWorkbench = new Function( - execFileHelper!, - "workbenchScriptPath", - "PLUGIN_ROOT", - /\bisJsonObject\d*\b/u.exec(source!)![0], - `${source}\nreturn executeWorkbench;`, - )( - async ( - _command: string, - _args: string[], - options: { timeout: number }, - ) => ({ stdout: JSON.stringify({ timeout: options.timeout }) }), - () => "workbench.py", - PLUGIN_ROOT, - () => true, - ) as (command: string, args: string[]) => Promise<{ timeout: number }>; - - expect(await executeWorkbench("python", ["start-prompt-only-scan"])).toEqual({ - timeout: 300_000, - }); - expect(await executeWorkbench("python", ["start-scan"])).toEqual({ - timeout: 300_000, - }); - expect(await executeWorkbench("python", ["other-operation"])).toEqual({ - timeout: 30_000, - }); +test("gives prompt-only startup the same timeout as other scan operations", () => { + expect(workbenchCommandTimeout("start-prompt-only-scan")).toBe(300_000); + expect(workbenchCommandTimeout("start-scan")).toBe(300_000); + expect(workbenchCommandTimeout("other-operation")).toBe(30_000); }); diff --git a/sdk/typescript/tests-ts/scan-merge-reconciliation.test.ts b/sdk/typescript/tests-ts/scan-merge-reconciliation.test.ts index 3a117c9c2..13952211c 100644 --- a/sdk/typescript/tests-ts/scan-merge-reconciliation.test.ts +++ b/sdk/typescript/tests-ts/scan-merge-reconciliation.test.ts @@ -4,6 +4,7 @@ import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import { join } from "node:path"; import { fileURLToPath } from "node:url"; import { afterEach, expect, test } from "bun:test"; +import { semanticCoverage } from "./helpers/semantic-scan.js"; import { createScanMergeValidator, type ScanMergeInput, @@ -129,12 +130,7 @@ function input(scanId: string, findings = [finding()]): ScanMergeInput { sourceFindingIds: [`${scanId}:${index}`], }, })), - coverage: { - completeness: "complete", - surfaces: [], - explicitExclusions: [], - deferred: [], - }, + coverage: semanticCoverage(), }, sourceFindings: findings.map((entry, index) => ({ ...structuredClone(entry), @@ -169,8 +165,8 @@ test("returned aggregates detach inherited history, candidates, originals and co raw.findings[0]!["summary"] = "Current synthesis."; provenance(raw.findings[0]!)["sourceFindingIds"] = ["source:0"]; const before = structuredClone({ source, previous, raw }); - const { aggregate, newFindings } = validate(raw, [], previous); - expect(newFindings).toBe(0); + const { aggregate, newFindingScanIds } = validate(raw, [], previous); + expect(newFindingScanIds).toEqual([]); const saved = provenance(aggregate.findings[0]!); expect(saved["sourceFindings"]).toEqual([ { id: "source:0", finding: source.sourceFindings[0] }, @@ -214,26 +210,12 @@ test("a retained finding cannot split across outputs in either order", async () expect({ previous, retained, split }).toEqual(before); }); -test("implicit source grouping keeps exact-source ambiguity checks and insertion order", async () => { +test("requires explicit provenance even when source identities match", async () => { const validate = await createScanMergeValidator(pluginRoot); - const source = input("implicit"); - source.sourceFindings.push(structuredClone(source.sourceFindings[0]!)); + const source = input("explicit"); const raw = submission([finding()]); - const result = validate(raw, [source], null).aggregate; - expect(provenance(result.findings[0]!)["sourceFindingIds"]).toEqual([ - "implicit:0", - "implicit:1", - ]); - expect(provenance(result.findings[0]!)["sourceFindings"]).toEqual( - source.sourceFindings.map((entry, index) => ({ - id: `implicit:${index}`, - finding: entry, - })), - ); - source.sourceFindings[1]!["summary"] = - "Distinct evidence with the same identity."; expect(() => validate(raw, [source], null)).toThrow( - "ambiguous source findings", + "explicit sourceFindingIds", ); }); diff --git a/sdk/typescript/tests-ts/scan-merge.test.ts b/sdk/typescript/tests-ts/scan-merge.test.ts index 94d9999dc..865a90429 100644 --- a/sdk/typescript/tests-ts/scan-merge.test.ts +++ b/sdk/typescript/tests-ts/scan-merge.test.ts @@ -6,6 +6,7 @@ import { join } from "node:path"; import { execFileSync } from "node:child_process"; import { fileURLToPath } from "node:url"; import { beforeAll, describe, expect, test } from "bun:test"; +import { semanticFinding, semanticCoverage } from "./helpers/semantic-scan.js"; import { build } from "esbuild"; import { combineScanCoverage, @@ -34,22 +35,7 @@ beforeAll(async () => { }); function finding(id = "shared", extra: JsonObject = {}): SemanticFinding { - return { - ruleId: "cross-site-scripting.request-output", - identity: { anchor: id }, - title: "Unsafe request output", - summary: "A request-controlled value reaches an HTML response.", - severity: { level: "high" }, - confidence: { - level: "high", - rationale: "The source establishes reachability.", - }, - taxonomy: { category: "cross-site-scripting", cwe: ["CWE-79"] }, - locations: [{ path: "src/render.js", startLine: 1, endLine: 2 }], - remediation: "Encode request-controlled values before emitting HTML.", - provenance: { source: "local_plugin" }, - ...extra, - }; + return semanticFinding({ identity: { anchor: id }, ...extra }); } function child( @@ -76,13 +62,7 @@ function child( sourceFindingIds: [`${scanId}:${index}`], }, })), - coverage: { - completeness: "complete", - surfaces: [], - explicitExclusions: [], - deferred: [], - ...coverage, - }, + coverage: semanticCoverage(coverage), }, }; } @@ -117,7 +97,7 @@ describe("local scan merging", () => { { id: "first:0", finding: { summary: "Model-authored replacement." } }, ]; const result = merge(submission(submitted), [input], null); - expect(result.newFindings).toBe(1); + expect(result.newFindingScanIds).toEqual(["first"]); expect(sources(result.aggregate.findings[0]!)).toEqual([ { id: "first:0", finding: input.sourceFindings[0]! }, ]); @@ -149,7 +129,7 @@ describe("local scan merging", () => { }, }); const result = merge(submission([combined]), [second], initial); - expect(result.newFindings).toBe(0); + expect(result.newFindingScanIds).toEqual([]); expect(sources(result.aggregate.findings[0]!)).toEqual([ { id: "first:0", finding: first.sourceFindings[0]! }, { id: "second:0", finding: second.sourceFindings[0]! }, @@ -163,13 +143,13 @@ describe("local scan merging", () => { ); }); - test("rejects omitted, invented, reused, and ambiguous sources", () => { + test("rejects omitted, invented, reused, and implicit sources", () => { const input = child("first", [finding(), finding("distinct")]); expect(() => merge(submission([input.draft.findings[0]!]), [input], null), ).toThrow("unaccounted source"); expect(() => merge(submission([finding("new")]), [input], null)).toThrow( - "no assigned source", + "explicit sourceFindingIds", ); expect(() => merge( @@ -197,7 +177,7 @@ describe("local scan merging", () => { finding("shared", { summary: "Independent vulnerable path." }), ]); expect(() => merge(submission([finding()]), [collision], null)).toThrow( - "ambiguous source", + "explicit sourceFindingIds", ); }); @@ -219,11 +199,11 @@ describe("local scan merging", () => { [next], previous, ); - expect(accepted.newFindings).toBe(1); + expect(accepted.newFindingScanIds).toEqual([next.scanId]); expect( merge(submission(accepted.aggregate.findings), [], accepted.aggregate) - .newFindings, - ).toBe(0); + .newFindingScanIds, + ).toEqual([]); }); test("validates findings before accepting a merge and excludes model-authored coverage", () => { @@ -276,7 +256,7 @@ describe("local scan merging", () => { [next], previous, ); - expect(result.newFindings).toBe(1); + expect(result.newFindingScanIds).toEqual([next.scanId]); const identities = result.aggregate.findings.map(scanFindingIdentity); expect(new Set(identities).size).toBe(2); expect(identities[0]).toBe(scanFindingIdentity(previous.findings[0]!)); @@ -308,7 +288,6 @@ describe("local scan merging", () => { expect(reordered.aggregate.findings.map(scanFindingIdentity)).toEqual( [...identities].reverse(), ); - expect(reordered.newFindings).toBe(1); expect(reordered.newFindingScanIds).toEqual([next.scanId]); }); @@ -323,7 +302,6 @@ describe("local scan merging", () => { }, }); const duplicateOnly = merge(submission([shared]), [first, second], null); - expect(duplicateOnly.newFindings).toBe(1); expect(duplicateOnly.newFindingScanIds).toEqual(["first"]); const result = merge( @@ -331,7 +309,6 @@ describe("local scan merging", () => { [first, second, third], null, ); - expect(result.newFindings).toBe(2); expect(result.newFindingScanIds).toEqual(["first", "third"]); const rediscovered = child("fourth"); const retained = structuredClone(result.aggregate.findings); @@ -341,7 +318,6 @@ describe("local scan merging", () => { [rediscovered], result.aggregate, ); - expect(repeated.newFindings).toBe(0); expect(repeated.newFindingScanIds).toEqual([]); }); @@ -358,7 +334,6 @@ describe("local scan merging", () => { merge(submission([], { threatModel }), [first, second], null), ).toEqual({ aggregate: submission([], { threatModel }), - newFindings: 0, newFindingScanIds: [], }); }); @@ -631,7 +606,7 @@ test("consolidates accepted aliases without counting their retained lineage as n const combined = structuredClone(previous.findings[0]!); provenance(combined)["sourceFindingIds"] = ["alias-a:0", "alias-b:0"]; const result = merge(submission([combined]), [], previous); - expect(result.newFindings).toBe(0); + expect(result.newFindingScanIds).toEqual([]); expect(sources(result.aggregate.findings[0]!)).toHaveLength(2); expect(provenance(result.aggregate.findings[0]!)["previousFindings"]).toEqual( expect.arrayContaining([ diff --git a/sdk/typescript/tests-ts/scan-resume.test.ts b/sdk/typescript/tests-ts/scan-resume.test.ts index f310a7ae6..3a68e0e70 100644 --- a/sdk/typescript/tests-ts/scan-resume.test.ts +++ b/sdk/typescript/tests-ts/scan-resume.test.ts @@ -1,3 +1,4 @@ +import { publishDraft } from "./support/scan-publication.js"; import { semanticCoverage, semanticFinding } from "./helpers/semantic-scan.js"; import { randomUUID } from "node:crypto"; import { execFileSync } from "node:child_process"; @@ -25,10 +26,7 @@ import { TerminalDeepScanError, type DeepScanCheckpoint, } from "../src/deep-scan.js"; -import { - prepareSemanticScanDraft, - type SemanticScan, -} from "../src/scan-semantics.js"; +import type { SemanticScan } from "../src/scan-semantics.js"; import { prepareScanArtifactRestorer, runWorkbench } from "../src/runtime.js"; import { ScanTransportClosedError } from "../src/scan-execution.js"; import { capture, dependencies } from "./cli-fixtures.js"; @@ -255,7 +253,7 @@ async function interruptedScan( deferred: [{ id: "time-cap", reason: "Synthetic time cap" }], }, }; - await writeDraft(command, child, "standard", { + await publishDraft(command, child, "standard", { ...aggregate, scanId: childId, }); @@ -312,41 +310,6 @@ async function interruptedScan( }; } -async function writeDraft( - command: (args: readonly string[], input?: string) => Promise, - registration: JsonObject, - mode: "deep" | "standard", - draft: SemanticScan, -) { - const directory = registration["scanDir"] as string; - const documents = prepareSemanticScanDraft( - { - targetContract: registration["contract"] as JsonObject, - mode, - targetRevision: registration["targetRevision"] as string, - }, - draft, - ); - const draftPath = join(directory, "drafts", randomUUID() + ".json"); - const checkpointPath = join( - directory, - "drafts", - randomUUID() + ".checkpoint.json", - ); - await mkdir(join(directory, "drafts"), { recursive: true, mode: 0o700 }); - await writeFile(draftPath, JSON.stringify(documents)); - await writeFile(checkpointPath, JSON.stringify(draft)); - await command([ - "write-scan-draft", - "--scan-id", - registration["scanId"] as string, - "--draft-path", - draftPath, - "--checkpoint-path", - checkpointPath, - ]); -} - test("resume preserves its identity, launch recipe, accepted child and checkpoint", async () => { const f = await interruptedScan(); const before = await readFile(join(f.scanDir, DEEP_SCAN_CHECKPOINT)); @@ -528,274 +491,22 @@ async function finishDiscovery(f: Awaited>) { ], JSON.stringify(checkpoint), ); - await writeDraft(f.command, f.registration, "deep", checkpoint.aggregate!); + await publishDraft(f.command, f.registration, "deep", checkpoint.aggregate!); } -// Accounting is shared by both terminal states; canceled cases retain coverage -// for recovered totals and missing optional child or merge persistence. -test.each([ - ...( - [ - ["absent", "complete"], - ["stale", "complete"], - ["exact", "complete"], - ["larger", "complete"], - ["absent", "unknown-child"], - ["larger", "unknown-child"], - ["stale", "running-child"], - ["absent", "running-threadless"], - ["larger", "running-threadless"], - ["absent", "failed-threadless"], - ["stale", "failed-threadless"], - ["larger", "unknown-merge"], - ["absent", "unregistered-merge"], - ["stale", "unregistered-merge"], - ["absent", "marked-merge"], - ["larger", "marked-merge"], - ["larger", "unavailable"], - ["larger", "parent-unavailable"], - ["larger", "unknown-legacy"], - ["larger", "mismatched-child"], - ["larger", "missing-child"], - ["absent", "legacy-saved"], - ["absent", "legacy-current"], - ["absent", "legacy-logs"], - ["absent", "shared-legacy"], - ["larger", "shared-legacy"], - ] as const - ).map(([saved, accounting]) => ["failed", saved, accounting] as const), - ["canceled", "absent", "complete"], - ["canceled", "larger", "unknown-child"], - ["canceled", "stale", "failed-threadless"], - ["canceled", "absent", "unregistered-merge"], -] as const)( - "rejecting a %s checkpoint reconciles %s saved cost with %s accounting", - async (terminalReason, saved, accounting) => { - const cost = (input_tokens: number, output_tokens: number) => - estimateScanCost("gpt-5.6-sol", { input_tokens, output_tokens })!; - const childCost = - accounting === "unknown-child" ? undefined : cost(100_000, 10_000); - const legacy = - accounting === "legacy-saved" || - accounting === "legacy-logs" || - accounting === "legacy-current" || - accounting === "shared-legacy" || - accounting === "unknown-legacy"; - const separateLegacy = legacy && accounting !== "legacy-current"; - const recoveredCost = cost( - (separateLegacy ? 103_000 : 101_000) + - (accounting === "legacy-logs" ? 425 : 0), - (separateLegacy ? 10_300 : 10_100) + - (accounting === "legacy-logs" ? 42 : 0), - ); - const savedCost = - saved === "absent" - ? undefined - : saved === "stale" - ? cost(1_000, 100) - : saved === "larger" - ? cost(200_000, 20_000) - : recoveredCost; - const complete = - accounting === "complete" || - (legacy && - accounting !== "unknown-legacy" && - accounting !== "shared-legacy"); - const expectedCost = - complete && saved !== "larger" ? recoveredCost : savedCost; - const f = await interruptedScan( - "deep", - false, - {}, - false, - accounting !== "unregistered-merge", - { cost: childCost }, - ); - const sessionStartedAt = ( - await f.command(["get-cli-scan-resume", "--scan-id", f.scanId]) - )["startedAt"] as string; - // Give independent child and merge sessions overlapping lifetimes. Merge - // accounting must not include the child a second time through its directory. - const writeUsage = async ( - path: string, - id: string, - cwd: string, - inputTokens: number, - outputTokens: number, - parentThreadId?: string, - ) => { - await writeFile( - path, - [ - { - type: "session_meta", - payload: { - id, - cwd, - timestamp: sessionStartedAt, - ...(parentThreadId === undefined - ? {} - : { parent_thread_id: parentThreadId }), - }, - }, - { - type: "event_msg", - payload: { - type: "token_count", - info: { - total_token_usage: { - input_tokens: inputTokens, - output_tokens: outputTokens, - }, - }, - }, - }, - ] - .map((event) => JSON.stringify(event)) - .join("\n") + "\n", - ); - }; - await writeUsage( - f.sessionPath, - f.threadId, - join(f.scanDir, "artifacts/deep-scan/merge"), - 1_000, - 100, - ); - const childThreadId = randomUUID(); - await writeUsage( - join(f.codexHome, "sessions", `rollout-${childThreadId}.jsonl`), - childThreadId, - f.childDir!, - 100_000, - 10_000, - ); - if (accounting === "unknown-merge") await rm(f.sessionPath); +test.each(["failed", "canceled"] as const)( + "rejecting a %s checkpoint preserves saved accounting and artifacts", + async (terminalReason) => { + const cost = estimateScanCost("gpt-5.6-sol", { + input_tokens: 1000, + output_tokens: 100, + })!; + const f = await interruptedScan("deep", false, {}, false, true, { cost }); const checkpointPath = join(f.scanDir, DEEP_SCAN_CHECKPOINT); const checkpoint = JSON.parse( await readFile(checkpointPath, "utf8"), ) as DeepScanCheckpoint; checkpoint.terminalReason = terminalReason; - if (accounting === "marked-merge") checkpoint.costUnavailable = true; - if ( - (saved === "absent" && accounting === "complete") || - accounting === "running-child" || - accounting === "running-threadless" || - accounting === "failed-threadless" - ) { - const directory = "artifacts/deep-scan/passes/pass-2"; - await mkdir(join(f.scanDir, directory), { recursive: true, mode: 0o700 }); - const registration = await f.command( - [ - "register-cli-scan", - "--repository", - f.repository, - "--scan-dir", - join(f.scanDir, directory), - "--parent-scan-id", - f.scanId, - "--registration-json-stdin", - ], - JSON.stringify({ recipe: { ...f.recipe, mode: "standard" } }), - ); - const scanId = registration["scanId"] as string; - if (accounting === "failed-threadless") { - const unrecordedThread = randomUUID(); - await writeUsage( - join(f.codexHome, "sessions", `rollout-${unrecordedThread}.jsonl`), - unrecordedThread, - join(f.scanDir, directory), - 50_000, - 5_000, - ); - } - if (accounting === "running-threadless") - await f.command([ - "preserve-scan-results", - "--scan-id", - scanId, - "--cost-json", - JSON.stringify(cost(1000, 100)), - ]); - else if (accounting !== "running-child") - await f.command([ - "fail-scan", - "--scan-id", - scanId, - "--defer-publication", - "--message", - accounting === "failed-threadless" - ? "Synthetic failure after optional session persistence failed." - : "Synthetic failure before session startup.", - ...(accounting === "failed-threadless" - ? [] - : ["--cost-json", JSON.stringify(cost(0, 0))]), - ]); - checkpoint.passes.push( - { directory, scanId }, - { directory: "artifacts/deep-scan/passes/pass-3" }, - ); - } - if (legacy) { - const legacyThreadId = - accounting === "legacy-current" ? f.threadId : randomUUID(); - checkpoint.legacy = { - discoveryRuns: 1, - coverage: semanticCoverage(), - originThreadId: legacyThreadId, - ...(accounting === "legacy-saved" - ? { cost: cost(2_000, 200) } - : accounting === "legacy-current" - ? { cost: cost(1_000, 100) } - : {}), - }; - if (accounting === "legacy-logs" || accounting === "shared-legacy") { - await writeUsage( - join(f.codexHome, "sessions", `rollout-${legacyThreadId}.jsonl`), - legacyThreadId, - f.scanDir, - 2_000, - 200, - ); - if (accounting === "shared-legacy") { - const path = join( - f.codexHome, - "sessions", - `rollout-${legacyThreadId}.jsonl`, - ); - await writeFile( - path, - (await readFile(path, "utf8")).replace( - sessionStartedAt, - "2000-01-01T00:00:00Z", - ), - ); - } - const workerId = randomUUID(); - // Legacy workers and reducers started independently. Their output - // directories identify them without admitting the newer pass or merge. - for (const [id, cwd, input, output, parent] of [ - [ - workerId, - join(f.scanDir, "artifacts/deep_discovery/workers/worker/output"), - 250, - 25, - undefined, - ], - [randomUUID(), join(f.scanDir, "artifacts"), 125, 12, undefined], - [randomUUID(), f.repository, 50, 5, workerId], - ] as const) { - await writeUsage( - join(f.codexHome, "sessions", `rollout-${id}.jsonl`), - id, - cwd, - input, - output, - parent, - ); - } - } - } await f.command( [ "save-scan-artifact", @@ -806,15 +517,13 @@ test.each([ ], JSON.stringify(checkpoint), ); - // The terminal checkpoint can reach disk before the aggregate cost DB write. - if (savedCost) - await f.command([ - "preserve-scan-results", - "--scan-id", - f.scanId, - "--cost-json", - JSON.stringify(savedCost), - ]); + await f.command([ + "preserve-scan-results", + "--scan-id", + f.scanId, + "--cost-json", + JSON.stringify(cost), + ]); const paths = [ checkpointPath, f.checkpoint, @@ -826,32 +535,19 @@ test.each([ ].map((name) => join(f.childDir!, name)), ]; const artifacts = await Promise.all(paths.map((path) => readFile(path))); - const calls: string[][] = []; - const clients: unknown[] = []; + const saved = await f.command(["get-scan", "--scan-id", f.scanId]); + // Terminal rejection must work without recovering session accounting. + await rm(f.sessionPath); + const calls: string[] = []; const notifications: string[] = []; const client = resumeClient( f, - (options) => { - clients.push(options); - throw new Error( - "A terminal scan must not create a worker or run another model turn.", - ); + () => { + throw new Error("Terminal resume must not create a worker."); }, async (options, args, input) => { - calls.push([...args]); - if ( - (args[0] === "list-scans" && accounting === "unavailable") || - (args[0] === "get-scan" && accounting === "parent-unavailable") - ) - throw new Error("Optional accounting is unavailable."); - const result = await runWorkbench(options, args, input); - if (args[0] === "list-scans" && accounting === "mismatched-child") { - const scans = result["scans"] as JsonObject[]; - scans[0]!["parentScanId"] = randomUUID(); - } - if (args[0] === "list-scans" && accounting === "missing-child") - result["scans"] = []; - return result; + calls.push(args[0]!); + return runWorkbench(options, args, input); }, )({ codexOverrides: f.recipe.config }); try { @@ -875,43 +571,21 @@ test.each([ }, }), ).rejects.toBeInstanceOf(TerminalDeepScanError); - expect(clients).toEqual([]); expect(notifications).toEqual([]); - const failures = calls.filter(([command]) => command === "fail-scan"); - expect(failures).toHaveLength(1); - expect(failures[0]!.includes("--cost-json")).toBe( - expectedCost !== undefined && accounting !== "parent-unavailable", - ); - expect(calls.map(([command]) => command)).not.toContain( + for (const command of [ + "fail-scan", + "preserve-scan-results", "save-scan-artifact", - ); - expect(calls.map(([command]) => command)).not.toContain( "prepare-scan-completion", - ); + "list-scans", + ]) + expect(calls).not.toContain(command); expect(await Promise.all(paths.map((path) => readFile(path)))).toEqual( artifacts, ); - const parent = (await f.command(["get-scan", "--scan-id", f.scanId]))[ - "scan" - ] as JsonObject; - expect(parent).toMatchObject({ progress: { status: "failed" } }); - if (expectedCost) { - expect(parent["cost"]).toMatchObject({ - inputTokens: expectedCost.inputTokens, - outputTokens: expectedCost.outputTokens, - }); - expect((parent["cost"] as JsonObject)["estimatedUsd"]).toBeCloseTo( - expectedCost.estimatedUsd, - 12, - ); - if (saved === "larger") - expect(parent["cost"] as unknown).toEqual(savedCost!); - } else expect(parent["cost"]).toBeUndefined(); - const child = (await f.command(["get-scan", "--scan-id", f.childId!]))[ - "scan" - ] as JsonObject; - expect(child).toMatchObject({ progress: { status: "complete" } }); - expect(child["cost"] as unknown).toEqual(childCost); + expect(await f.command(["get-scan", "--scan-id", f.scanId])).toEqual( + saved, + ); } finally { await client.close(); } @@ -1169,7 +843,12 @@ test.each([true, false])( ], JSON.stringify(checkpoint), ); - await writeDraft(f.command, f.registration, "deep", checkpoint.aggregate!); + await publishDraft( + f.command, + f.registration, + "deep", + checkpoint.aggregate!, + ); let turns = 0; const client = resumeClient( f, @@ -1362,7 +1041,12 @@ test.each([ coverage: semanticCoverage({ completeness: "partial" }), }; checkpoint.terminalReason = "saturated"; - await writeDraft(f.command, f.registration, "deep", checkpoint.aggregate); + await publishDraft( + f.command, + f.registration, + "deep", + checkpoint.aggregate, + ); } await f.command( [ @@ -1544,43 +1228,148 @@ test.each([ }, ); +test("sealed legacy results retain their saved accounting", async () => { + const f = await interruptedScan( + "deep", + false, + { maxCostUsd: 0.001 }, + true, + false, + null, + ); + const cost = estimateScanCost("gpt-5.6-sol", { + input_tokens: 10000, + output_tokens: 2000, + })!; + const expectedCost = { + inputTokens: 10000, + outputTokens: 2000, + estimatedUsd: cost.estimatedUsd, + }; + await appendFile( + f.sessionPath, + JSON.stringify({ + type: "event_msg", + payload: { + type: "token_count", + info: { + total_token_usage: { input_tokens: 10000, output_tokens: 2000 }, + }, + }, + }) + "\n", + ); + const coverage = semanticCoverage({ + completeness: "partial", + surfaces: [], + explicitExclusions: [], + deferred: [{ reason: "Retained legacy discovery coverage." }], + }); + const checkpoint: DeepScanCheckpoint = { + version: 2, + startedAt: "2000-01-01T00:00:00Z", + passes: [], + mergedScanIds: [], + aggregate: { scanId: f.scanId, findings: [], coverage }, + noNewStreak: 0, + consecutiveErrors: 0, + terminalReason: "capped", + mergeFailures: 1, + legacy: { + discoveryRuns: 1, + coverage, + originThreadId: f.threadId, + cost, + }, + }; + await f.command( + [ + "save-scan-artifact", + "--scan-id", + f.scanId, + "--artifact-path", + DEEP_SCAN_CHECKPOINT, + ], + JSON.stringify(checkpoint), + ); + const artifactNames = [ + "scan-manifest.json", + "findings.json", + "coverage.json", + "report.md", + DEEP_SCAN_CHECKPOINT, + ]; + await publishDraft(f.command, f.registration, "deep", checkpoint.aggregate!); + await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); + const artifacts = await Promise.all( + artifactNames.map((name) => readFile(join(f.scanDir, name))), + ); + let turns = 0; + const client = resumeClient(f, () => ({ + startThread() { + return { + id: null, + async runStreamed() { + turns++; + throw new Error("Completed legacy discovery needs no model turn."); + }, + }; + }, + resumeThread() { + throw new Error("The retired coordinator must not resume."); + }, + }))({ codexOverrides: f.recipe.config }); + const options: ScanOptions = { + mode: "deep", + outputDir: f.scanDir, + resumeScanId: f.scanId, + maxCostUsd: 0.001, + ...f.recipe.deepScan, + }; + try { + const result = await client.run(f.repository, options); + expect(result.manifest.scan.id).toBe(f.scanId); + expect(result.manifest.scan.sealedAt).toBeString(); + expect(result.threadId).toBe(f.threadId); + expect(result.coverage.completeness).toBe("partial"); + expect(result.cost).toMatchObject(expectedCost); + expect(turns).toBe(0); + expect( + (await f.command(["get-scan", "--scan-id", f.scanId]))["scan"], + ).toMatchObject({ + progress: { status: "complete" }, + cost: expectedCost, + }); + expect( + await Promise.all( + artifactNames.map((name) => readFile(join(f.scanDir, name))), + ), + ).toEqual(artifacts); + } finally { + await client.close(); + } +}); + test.each([ - [false, false], - [true, false], - [false, true], -])( - "completed legacy discovery recovers partial results when its saved budget is exhausted (sealed: %p, restore logs: %p)", - async (sealed, restoreLogs) => { - const f = await interruptedScan( - "deep", - false, - { maxCostUsd: 0.001 }, - true, - false, - null, - ); - const cost = estimateScanCost("gpt-5.6-sol", { - input_tokens: 10000, - output_tokens: 2000, + "managed", + "native", + "wrong-claim", + "wrong-target", + "wrong-output", + "changed-artifact", +] as const)( + "sealed publication reads without authentication or execution (%s)", + async (scenario) => { + const childCost = estimateScanCost("gpt-5.6-sol", { + input_tokens: 375, + output_tokens: 3, })!; - const expectedCost = { - inputTokens: 10000, - outputTokens: 2000, - estimatedUsd: cost.estimatedUsd, - }; - await writeFile( - f.sessionPath, - JSON.stringify({ - type: "session_meta", - payload: { - id: f.threadId, - cwd: f.scanDir, - timestamp: ( - await f.command(["get-cli-scan-resume", "--scan-id", f.scanId]) - )["startedAt"] as string, - }, - }) + "\n", - ); + const expectedCost = estimateScanCost("gpt-5.6-sol", { + input_tokens: 1375, + output_tokens: 13, + })!; + const f = await interruptedScan("deep", false, {}, true, true, { + cost: childCost, + }); await appendFile( f.sessionPath, JSON.stringify({ @@ -1588,44 +1377,49 @@ test.each([ payload: { type: "token_count", info: { - total_token_usage: { input_tokens: 10000, output_tokens: 2000 }, + total_token_usage: { input_tokens: 1000, output_tokens: 10 }, }, }, }) + "\n", ); - const coverage = semanticCoverage({ - completeness: "partial", - surfaces: [], - explicitExclusions: [], - deferred: [{ reason: "Retained legacy discovery coverage." }], - }); - const checkpoint: DeepScanCheckpoint = { - version: 2, - startedAt: "2000-01-01T00:00:00Z", - passes: [], - mergedScanIds: [], - aggregate: { scanId: f.scanId, findings: [], coverage }, - noNewStreak: 0, - consecutiveErrors: 0, - terminalReason: "capped", - mergeFailures: 1, - legacy: { - discoveryRuns: 1, - coverage, - originThreadId: f.threadId, - ...(restoreLogs ? {} : { cost }), - }, - }; - await f.command( - [ - "save-scan-artifact", - "--scan-id", + await finishDiscovery(f); + await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); + const native = scenario !== "managed"; + const owner = "synthetic-native-owner"; + const claim = randomUUID(); + if (native) { + const nativeHome = join(f.root, "native-home"); + await rename(f.codexHome, nativeHome); + f.environment.CODEX_HOME = nativeHome; + execFileSync(f.python, [ + "-c", + `import sqlite3, sys +with sqlite3.connect(sys.argv[1]) as connection: + connection.execute("UPDATE scans SET deep_scan_owner_thread_id = ?, handoff_claim_token = ? WHERE id = ?", sys.argv[2:]) +`, + join(f.environment.CODEX_SECURITY_STATE_DIR, "workbench.sqlite3"), + owner, + claim, f.scanId, - "--artifact-path", - DEEP_SCAN_CHECKPOINT, - ], - JSON.stringify(checkpoint), - ); + ]); + } + let repository = f.repository; + let outputDir = f.scanDir; + if (scenario === "wrong-target") { + repository = join(f.root, "other-repository"); + await mkdir(repository); + await writeFile( + join(repository, "source.py"), + "# another synthetic source\n", + ); + } + if (scenario === "wrong-output") { + outputDir = join(f.root, "other-output"); + await mkdir(outputDir, { mode: 0o700 }); + await cp(f.scanDir, outputDir, { recursive: true }); + } + if (scenario === "changed-artifact") + await appendFile(join(f.scanDir, "findings.json"), "\n"); const artifactNames = [ "scan-manifest.json", "findings.json", @@ -1633,88 +1427,87 @@ test.each([ "report.md", DEEP_SCAN_CHECKPOINT, ]; - if (sealed) { - await writeDraft( - f.command, - f.registration, - "deep", - checkpoint.aggregate!, - ); - await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); - } - const artifacts = sealed - ? await Promise.all( - artifactNames.map((name) => readFile(join(f.scanDir, name))), - ) - : undefined; - let starts = 0; - let turns = 0; - const client = resumeClient(f, () => ({ - startThread() { - starts++; - return { - id: null, - async runStreamed() { - turns++; - throw new Error("Completed legacy discovery needs no model turn."); - }, - }; - }, - resumeThread() { - throw new Error("The retired coordinator must not resume."); + const artifacts = await Promise.all( + artifactNames.map((name) => readFile(join(f.scanDir, name))), + ); + const commands: string[] = []; + let authentications = 0; + const client = new TestClient( + { pluginPath: PLUGIN_ROOT }, + { + environment: f.environment, + ...(native + ? { + ambientExecution: { + command: { command: "synthetic-unused-codex" }, + configuration: {}, + environment: f.environment, + preserveProviderEnvironment: false, + pluginRoot: PLUGIN_ROOT, + }, + } + : {}), + prepareRuntime: async () => { + throw new Error( + "Saved publication must not prepare a Codex runtime.", + ); + }, + createCodex: () => { + throw new Error("Saved publication must not create a model client."); + }, + resolvePluginPython: async () => f.python, + runWorkbench: async (options, args, input) => { + commands.push(args[0]!); + return runWorkbench(options, args, input); + }, }, - }))({ codexOverrides: f.recipe.config }); - const options: ScanOptions = { - mode: "deep", - outputDir: f.scanDir, - resumeScanId: f.scanId, - maxCostUsd: 0.001, - ...f.recipe.deepScan, - }; + ); try { - if (restoreLogs) { - const savedSession = await readFile(f.sessionPath); - const checkpointPath = join(f.scanDir, DEEP_SCAN_CHECKPOINT); - const savedCheckpoint = await readFile(checkpointPath); - const before = await f.command(["get-scan", "--scan-id", f.scanId]); - await rm(f.sessionPath); - try { - await expect(client.run(f.repository, options)).rejects.toThrow( - "Restore the original Deep Scan session logs", - ); - expect(starts).toBe(0); - expect(turns).toBe(0); - expect(before["scan"]).toMatchObject({ - progress: { status: "running" }, - }); - expect(await f.command(["get-scan", "--scan-id", f.scanId])).toEqual( - before, - ); - expect(await readFile(checkpointPath)).toEqual(savedCheckpoint); - } finally { - await writeFile(f.sessionPath, savedSession); - } - } - const result = await client.run(f.repository, options); - expect(result.manifest.scan.id).toBe(f.scanId); - expect(result.manifest.scan.sealedAt).toBeString(); - expect(result.threadId).toBe(f.threadId); - expect(result.coverage.completeness).toBe("partial"); - expect(result.cost).toMatchObject(expectedCost); - expect(turns).toBe(0); - expect( - (await f.command(["get-scan", "--scan-id", f.scanId]))["scan"], - ).toMatchObject({ - progress: { status: "complete" }, - cost: expectedCost, + const pending = client.run(repository, { + mode: "deep", + outputDir, + maxCostUsd: 1, + ...(native + ? { + registeredScan: { + scanId: f.scanId, + scanDir: outputDir, + threadId: owner, + handoffClaimToken: + scenario === "wrong-claim" ? randomUUID() : claim, + }, + } + : { resumeScanId: f.scanId }), + onAuthentication() { + authentications++; + }, }); - if (sealed) { + if (scenario === "managed" || scenario === "native") { + const result = await pending; + expect(result.cost).toEqual(expectedCost); + expect(result.threadId).toBe(f.threadId); + expect(result.repositoryFindings).toEqual([]); + expect(commands).toContain("list-global-findings"); expect( - await Promise.all( - artifactNames.map((name) => readFile(join(f.scanDir, name))), - ), - ).toEqual(artifacts!); + commands.filter((command) => command === "complete-scan"), + ).toHaveLength(1); + } else { + await expect(pending).rejects.toThrow( + scenario === "wrong-claim" + ? "another continuation" + : scenario === "changed-artifact" + ? "Cannot resume sealed scan" + : "match its target, directory and mode", + ); + expect(commands).not.toContain("complete-scan"); } + expect(authentications).toBe(0); + expect(commands).not.toContain("get-scan-feedback"); + expect( + await Promise.all( + artifactNames.map((name) => readFile(join(f.scanDir, name))), + ), + ).toEqual(artifacts); } finally { await client.close(); } @@ -1722,21 +1515,16 @@ test.each([ ); test.each([ - ["unsealed", "other-directory", false], - ["unsealed", "predates-scan", false], - ["unsealed", "undated", false], - ["unsealed", "other-directory", true], - ["unsealed", "dedicated", true], - ["migrate", "predates-scan", false], - ["migrate", "dedicated", true], + ["sealed", "other-directory", false], ["sealed", "predates-scan", false], + ["sealed", "undated", false], + ["sealed", "other-directory", true], ["sealed", "dedicated", true], ["sealed-v1", "predates-scan", false], ["sealed-v1", "predates-scan", true], - ["failed", "predates-scan", false], - ["canceled", "predates-scan", false], + ["sealed-v1", "dedicated", true], ] as const)( - "legacy recovery attributes only scan-owned sessions (%s, %s, required: %p)", + "sealed legacy accounting includes only scan-owned sessions (%s, %s, required: %p)", async (state, origin, required) => { const f = await interruptedScan("deep", false, {}, true, false, null); const usage = { input_tokens: 1_000_000, output_tokens: 100 }; @@ -1782,15 +1570,14 @@ test.each([ aggregate, noNewStreak: 0, consecutiveErrors: 0, - terminalReason: - state === "failed" || state === "canceled" ? state : "capped", + terminalReason: "capped", legacy: { originThreadId: f.threadId, discoveryRuns: 1, coverage: aggregate.coverage, }, }; - if (state === "sealed-v1" || state === "migrate") { + if (state === "sealed-v1") { await f.command([ "set-scan-thread", "--scan-id", @@ -1828,16 +1615,6 @@ with sqlite3.connect(sys.argv[1]) as connection: JSON.stringify(checkpoint), ); } - if (state === "migrate") { - await writeDraft(f.command, f.registration, "deep", aggregate); - execFileSync(f.python, [ - "-c", - "import sqlite3,sys; c=sqlite3.connect(sys.argv[1]); c.execute('UPDATE scans SET deep_scan_owner_thread_id=continuation_thread_id, continuation_thread_id=NULL WHERE id=?',(sys.argv[2],)); c.commit()", - join(f.environment.CODEX_SECURITY_STATE_DIR, "workbench.sqlite3"), - f.scanId, - ]); - } - const sealed = state === "sealed" || state === "sealed-v1"; const names = [ "scan-manifest.json", "findings.json", @@ -1845,13 +1622,11 @@ with sqlite3.connect(sys.argv[1]) as connection: "report.md", ...(state === "sealed" ? [DEEP_SCAN_CHECKPOINT] : []), ]; - if (sealed) { - await writeDraft(f.command, f.registration, "deep", aggregate); - await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); - } - const artifacts = sealed - ? await Promise.all(names.map((name) => readFile(join(f.scanDir, name)))) - : null; + await publishDraft(f.command, f.registration, "deep", aggregate); + await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); + const artifacts = await Promise.all( + names.map((name) => readFile(join(f.scanDir, name))), + ); const before = await f.command(["get-scan", "--scan-id", f.scanId]); let turns = 0; const unusedThread = (id: string | null) => ({ @@ -1873,13 +1648,7 @@ with sqlite3.connect(sys.argv[1]) as connection: ...f.recipe.deepScan, ...(required ? { maxCostUsd: 10 } : {}), }); - if (state === "failed" || state === "canceled") { - const error: unknown = await pending.catch((error: unknown) => error); - expect(error).toBeInstanceOf(TerminalDeepScanError); - expect( - (error as TerminalDeepScanError).accounting.constituents, - ).toEqual([null]); - } else if (required && origin !== "dedicated") { + if (required && origin !== "dedicated") { await expect(pending).rejects.toThrow(/cost limit/); expect(await f.command(["get-scan", "--scan-id", f.scanId])).toEqual( before, @@ -1898,12 +1667,9 @@ with sqlite3.connect(sys.argv[1]) as connection: "scan" ] as JsonObject; if (origin !== "dedicated") expect(saved["cost"]).toBeUndefined(); - if (artifacts) - expect( - await Promise.all( - names.map((name) => readFile(join(f.scanDir, name))), - ), - ).toEqual(artifacts); + expect( + await Promise.all(names.map((name) => readFile(join(f.scanDir, name)))), + ).toEqual(artifacts); expect(turns).toBe(0); } finally { await client.close(); @@ -1913,6 +1679,7 @@ with sqlite3.connect(sys.argv[1]) as connection: test.each([ [null, false, false], + ["native-unbound", false, false], [undefined, false, false], [null, true, false], [null, true, true], @@ -1936,6 +1703,9 @@ test.each([ outputTokens: 13, estimatedUsd: cost.estimatedUsd, }; + const sessionStartedAt = ( + await f.command(["get-cli-scan-resume", "--scan-id", f.scanId]) + )["startedAt"] as string; await finishDiscovery(f); if (checkpoint !== "v2") { await rm(join(f.scanDir, DEEP_SCAN_CHECKPOINT)); @@ -1967,6 +1737,21 @@ with sqlite3.connect(sys.argv[1]) as connection: JSON.stringify(cost), ]); await f.command(["prepare-scan-completion", "--scan-id", f.scanId]); + if (checkpoint === "native-unbound") { + execFileSync(f.python, [ + "-c", + `import sqlite3, sys +with sqlite3.connect(sys.argv[1]) as connection: + connection.execute( + "UPDATE scans SET recipe_json = NULL, deep_scan_owner_thread_id = ? WHERE id = ?", + (sys.argv[3], sys.argv[2]), + ) +`, + join(f.environment.CODEX_SECURITY_STATE_DIR, "workbench.sqlite3"), + f.scanId, + f.threadId, + ]); + } const artifactNames = [ "scan-manifest.json", "findings.json", @@ -1977,15 +1762,11 @@ with sqlite3.connect(sys.argv[1]) as connection: const artifacts = await Promise.all( artifactNames.map((name) => readFile(join(f.scanDir, name))), ); - const savedCheckpoint = ( - await f.command(["get-scan", "--scan-id", f.scanId]) - )["compositionCheckpoint"]; + const saved = await f.command(["get-scan", "--scan-id", f.scanId]); + const savedCheckpoint = saved["compositionCheckpoint"]; if (checkpoint === "v2") expect(savedCheckpoint).toMatchObject({ version: 2 }); else expect(savedCheckpoint).toBeNull(); - const sessionStartedAt = ( - await f.command(["get-cli-scan-resume", "--scan-id", f.scanId]) - )["startedAt"] as string; for (const [threadId, cwd, inputTokens, outputTokens, timestamp] of [ [f.threadId, f.scanDir, 1000, 10, sessionStartedAt], [ @@ -2068,7 +1849,15 @@ with sqlite3.connect(sys.argv[1]) as connection: const pending = client.run(f.repository, { mode: "deep", outputDir: f.scanDir, - resumeScanId: f.scanId, + ...(checkpoint === "native-unbound" + ? { + registeredScan: { + scanId: f.scanId, + scanDir: f.scanDir, + threadId: f.threadId, + }, + } + : { resumeScanId: f.scanId }), ...f.recipe.deepScan, ...(requiredCost ? { maxCostUsd: 1 } : {}), onWarning: (warning) => warnings.push(warning), diff --git a/sdk/typescript/tests-ts/scan-semantics-preservation.test.ts b/sdk/typescript/tests-ts/scan-semantics-preservation.test.ts index 79b21de8b..a72ff3517 100644 --- a/sdk/typescript/tests-ts/scan-semantics-preservation.test.ts +++ b/sdk/typescript/tests-ts/scan-semantics-preservation.test.ts @@ -1,6 +1,5 @@ import { expect, test } from "bun:test"; import { - exactUnion, preserveFindingDetails, type JsonObject, } from "../src/scan-semantics.js"; @@ -14,26 +13,45 @@ const source = { id: "saved:0", finding: evidence }; const other = { id: "saved:1", finding: evidence }; const changed = { id: source.id, finding: { ...evidence, severity: "low" } }; -const cases: [unknown[], unknown[]][] = [ - [[source, other], [{ ...source }]], - [[source], [other, source]], +const cases: [unknown[], unknown[], unknown[]][] = [ + [[source, other], [{ ...source }], [source, other]], + [[source], [other, source], [source, other]], [ [source, source], [other, other], + [source, other], ], [ [changed, source], [structuredClone(source), structuredClone(changed)], + [changed, source], + ], + [ + [source], + [{ finding: evidence, id: source.id }], + [source, { finding: evidence, id: source.id }], + ], + [ + [source], + [{ ...source, annotation: "retain" }], + [source, { ...source, annotation: "retain" }], + ], + [ + [source], + [null, { finding: evidence }], + [source, null, { finding: evidence }], ], - [[source], [{ finding: evidence, id: source.id }]], - [[source], [{ ...source, annotation: "retain" }]], - [[source], [null, { finding: evidence }]], ]; test.each( - cases.map(([current, previous], index) => ({ current, previous, index })), + cases.map(([current, previous, expected], index) => ({ + current, + previous, + expected, + index, + })), )( "original union matches exact JSON equality and first-occurrence order (case $index)", - ({ current, previous }) => { + ({ current, previous, expected }) => { const saved: JsonObject = { summary: "saved", provenance: { sourceFindings: previous }, @@ -45,7 +63,7 @@ test.each( const before = structuredClone({ current, previous }); preserveFindingDetails(next, saved); expect((next["provenance"] as JsonObject)["sourceFindings"]).toEqual( - exactUnion(current, previous), + expected, ); expect({ current, previous }).toEqual(before); expect( diff --git a/sdk/typescript/tests-ts/skeleton.test.ts b/sdk/typescript/tests-ts/skeleton.test.ts index bbb4bb5f9..0e50250b5 100644 --- a/sdk/typescript/tests-ts/skeleton.test.ts +++ b/sdk/typescript/tests-ts/skeleton.test.ts @@ -58,12 +58,8 @@ function capture(): { } describe("TypeScript package skeleton", () => { - test("pins one Codex version across the CLI, MCP app, and evals", async () => { - const directories = [ - "sdk/typescript", - "plugins/codex-security/mcp-app", - "evals/triage-finding", - ]; + test("pins one Codex version across the CLI and evals", async () => { + const directories = ["sdk/typescript", "evals/triage-finding"]; const manifests = await Promise.all( directories.map(async (directory) => JSON.parse( diff --git a/sdk/typescript/tests-ts/stopped-scan-results.test.ts b/sdk/typescript/tests-ts/stopped-scan-results.test.ts index 67126fc9a..03276e04b 100644 --- a/sdk/typescript/tests-ts/stopped-scan-results.test.ts +++ b/sdk/typescript/tests-ts/stopped-scan-results.test.ts @@ -48,15 +48,11 @@ const stoppedScanProbe = [ "artifact_dir = scan_dir / 'artifacts' / 'deep-scan' / 'passes' / 'pass-1'", "for directory in (scan_dir / 'artifacts', scan_dir / 'artifacts' / 'deep-scan', artifact_dir.parent, artifact_dir): directory.mkdir(mode=0o700)", "child = run('register-cli-scan', '--repository', str(target), '--scan-dir', str(artifact_dir), '--parent-scan-id', scan_id, '--recipe-json', json.dumps({**recipe,'mode':'standard'}))", - "result_path = scan_dir / 'result.json'", "finding = json.loads((plugin / 'examples' / 'completed-scan' / 'findings.json').read_text(encoding='utf-8'))['findings'][0]", "for field in ('findingId','occurrenceId','fingerprints'): finding.pop(field, None)", "finding.setdefault('provenance', {})['candidateId'] = 'checkpoint-candidate'", "payload = {'scanId': scan_id, 'findings': [finding], 'coverage': {'completeness': 'partial', 'surfaces': [], 'explicitExclusions': [], 'deferred': [{'candidateId': 'pending-validation', 'reason': 'Validation stopped with the scan.', 'paths': ['src/extract.py']}]}, 'threatModel': {'summary': 'Synthetic stopped-scan threat model.'}}", - "if source == 'accepted':", - " result_path.write_text(json.dumps(payload), encoding='utf-8')", - - "else:", + "if source != 'accepted':", " checkpoint = {**payload, 'complete': False}", " checkpoint_dir = scan_dir / 'checkpoints'", " checkpoint_dir.mkdir()", @@ -69,7 +65,6 @@ const stoppedScanProbe = [ " for document in (later,):", " encoded = json.dumps(document).encode()", " (checkpoint_dir / f'{hashlib.sha256(encoded).hexdigest()}.json').write_bytes(encoded)", - " result_path.write_text(json.dumps(later), encoding='utf-8')", " else:", " if source == 'distinct-instances':", " first = json.loads(json.dumps(finding))", @@ -79,7 +74,6 @@ const stoppedScanProbe = [ " checkpoint['findings'] = [first, second]", " encoded = json.dumps(checkpoint).encode()", " (checkpoint_dir / f'{hashlib.sha256(encoded).hexdigest()}.json').write_bytes(encoded)", - " result_path.write_text('{incomplete', encoding='utf-8')", "documents = {name: json.loads((plugin / 'examples' / 'completed-scan' / name).read_text()) for name in ('scan-manifest.json','findings.json','coverage.json')}", "child_findings = later['findings'] if source == 'refined-checkpoint' else checkpoint['findings'] if source == 'distinct-instances' else payload['findings']", "manifest = documents['scan-manifest.json']['scan']", diff --git a/sdk/typescript/tests-ts/support/api-client.ts b/sdk/typescript/tests-ts/support/api-client.ts index 462183e44..a625eccb9 100644 --- a/sdk/typescript/tests-ts/support/api-client.ts +++ b/sdk/typescript/tests-ts/support/api-client.ts @@ -1,3 +1,6 @@ +import { mkdir } from "node:fs/promises"; +import { join } from "node:path"; +import { preparedRuntime } from "./api-events.js"; import { CodexSecurity } from "../../src/api.js"; import type { JsonObject } from "../../src/config.js"; @@ -92,3 +95,32 @@ export const SHELL_ENVIRONMENT_PREFIX = export function shellEnvironmentReference(name: string, suffix = ""): string { return `"${SHELL_ENVIRONMENT_PREFIX}${name}${suffix}"`; } + +export async function cancellationSetup(root: string) { + const repository = join(root, "repository"); + const codexHome = join(root, "codex-home"); + const scanDir = join(root, "scan"); + await Promise.all( + [repository, codexHome, scanDir].map((path) => + mkdir(path, { mode: 0o700 }), + ), + ); + const commands: Array = []; + const dependencies: Partial = { + prepareRuntime: async () => preparedRuntime(codexHome), + resolvePluginPython: async () => "/managed/python", + prepareOutputDir: async () => scanDir, + repositoryRevision: async () => "deadbeef", + runWorkbench: async (_options, args, input) => { + commands.push(args); + return mockWorkbench(args, input); + }, + }; + return { + repository, + scanDir, + commands, + controller: new AbortController(), + dependencies, + }; +} diff --git a/sdk/typescript/tests-ts/support/scan-publication.ts b/sdk/typescript/tests-ts/support/scan-publication.ts new file mode 100644 index 000000000..4931447fb --- /dev/null +++ b/sdk/typescript/tests-ts/support/scan-publication.ts @@ -0,0 +1,41 @@ +import { randomUUID } from "node:crypto"; +import { mkdir, writeFile } from "node:fs/promises"; +import { join } from "node:path"; +import type { JsonObject } from "../../src/config.js"; +import type { SemanticScan } from "../../src/semantic-models.js"; +import { prepareSemanticScanDraft } from "../../src/scan-semantics.js"; + +export async function publishDraft( + command: (args: readonly string[], input?: string) => Promise, + registration: JsonObject, + mode: "deep" | "standard", + draft: SemanticScan, +) { + const directory = registration["scanDir"] as string; + const documents = prepareSemanticScanDraft( + { + targetContract: registration["contract"] as JsonObject, + mode, + targetRevision: registration["targetRevision"] as string, + }, + draft, + ); + const draftPath = join(directory, "drafts", randomUUID() + ".json"); + const checkpointPath = join( + directory, + "drafts", + randomUUID() + ".checkpoint.json", + ); + await mkdir(join(directory, "drafts"), { recursive: true, mode: 0o700 }); + await writeFile(draftPath, JSON.stringify(documents)); + await writeFile(checkpointPath, JSON.stringify(draft)); + await command([ + "write-scan-draft", + "--scan-id", + registration["scanId"] as string, + "--draft-path", + draftPath, + "--checkpoint-path", + checkpointPath, + ]); +}