From 86c79d78d6fcb8481101d4d6ebbacbfea84d6bb3 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 14:47:37 +0530 Subject: [PATCH 1/7] feat(evals): add agent tool-use evaluation harness Add a deterministic eval layer for the agent harness. Scenarios script the OpenAI-compatible streaming tool loop with model turns and stub tool results, then score tool selection, planning, retrieval, and recovery without a provider key. - apps/sim/evals/agent-tool-use: 8 scenarios, scoring, JSON+Markdown report - `bun run test:evals` from apps/sim runs the suite and writes the report - picked up by the normal vitest run so a regression fails CI - README documents the contract and how to add a case --- apps/sim/evals/README.md | 88 +++++ .../agent-tool-use.eval.test.ts | 34 ++ apps/sim/evals/agent-tool-use/harness.ts | 366 ++++++++++++++++++ apps/sim/evals/agent-tool-use/report.ts | 63 +++ apps/sim/evals/agent-tool-use/scenarios.ts | 352 +++++++++++++++++ apps/sim/evals/agent-tool-use/types.ts | 125 ++++++ apps/sim/package.json | 1 + 7 files changed, 1029 insertions(+) create mode 100644 apps/sim/evals/README.md create mode 100644 apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts create mode 100644 apps/sim/evals/agent-tool-use/harness.ts create mode 100644 apps/sim/evals/agent-tool-use/report.ts create mode 100644 apps/sim/evals/agent-tool-use/scenarios.ts create mode 100644 apps/sim/evals/agent-tool-use/types.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md new file mode 100644 index 00000000000..604df341037 --- /dev/null +++ b/apps/sim/evals/README.md @@ -0,0 +1,88 @@ +# Agent harness evaluations + +Measurement for the agent harness — the code that turns a model's tool calls +into executed tools, feeds the results back, and keeps the turn alive when a +tool fails. Unit and integration tests prove the harness handles the cases we +already know about; evals measure whether it still behaves across a suite of +scenarios when the harness changes. + +## What runs + +The first suite lives in [`agent-tool-use/`](./agent-tool-use) and drives the +real OpenAI-compatible streaming tool loop +(`apps/sim/providers/openai-compat/streaming-tool-loop.ts`) — the loop that +serves OpenAI, DeepSeek, Groq, Cerebras, and the other OpenAI-compatible +providers. The model is **scripted**: each scenario supplies the assistant turns +(tool calls or a final answer) and the result of each tool call. That keeps the +suite deterministic and runnable in CI with no provider key, while the thing +being measured — tool dispatch, result feedback, error recovery — is real +production code. + +The suites cover four behaviors: + +| Category | What it measures | +| --- | --- | +| `tool-selection` | The loop dispatches the tool the model asked for, including from a set of distractors. | +| `planning` | Multi-turn, dependent and parallel tool calls execute in the right order and all results reach the next turn. | +| `retrieval` | Values returned by a tool survive into the final answer instead of being dropped or invented. | +| `recovery` | Tool errors, unknown tool names, and malformed argument JSON are fed back to the model rather than thrown out of the loop. | + +## Run it + +From `apps/sim`: + +```sh +bun run test:evals +``` + +The command writes a JSON report and a Markdown summary to +`test-results/evals/agent-tool-use.{json,md}` (gitignored) and fails the process +if any scenario fails. To point the report somewhere else, run Vitest directly: + +```sh +EVAL_REPORT_PATH=/tmp/agent-tool-use.json bunx vitest run evals/agent-tool-use +``` + +The suite is also picked up by the normal `bun run test` run, so a regression +fails CI even without the dedicated command. + +## Add a case + +1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add + an entry to `AGENT_TOOL_USE_SCENARIOS`. +2. Declare the `tools` the model may call and the `script` it produces. A + `tools` turn lists the calls the model emits; an `answer` turn ends the run. + Attach each call's stub `result` (or leave it to default to a successful + empty output). +3. Add the assertions you care about under `expect`: the ordered + `toolCallSequence`, `requiredTools`/`forbiddenTools`, `finalContent`, + `maxIterations`, and tool call counts. Every assertion becomes a named check + in the report. +4. Run `bun run test:evals`. + +A scenario is data, not code — there is no harness change needed for a new case. + +### Simulating a failure + +- **Tool error:** give the call `result: { success: false, error: '...' }`. +- **Unknown tool:** call a `name` that is not in `tools`; the loop returns a + tool-not-found error to the model. +- **Malformed arguments:** set `argumentsJson` to an invalid or non-object JSON + string. The loop must not execute the call and must return the parse error to + the model. + +## Report shape + +`report.json` is machine-readable for dashboards and trend tracking; `report.md` +is the same data as a table. Each result carries the scenario id, pass/fail, +every named check with a failure detail, the final content, the executed tool +invocations, and metrics: iterations, tool call counts (success/error), latency, +model/tool time, first-response time, and token usage. + +## Scope and next steps + +This suite evaluates the tool loop directly. The next layer is a scenario that +runs the same scripted model through the full `DAGExecutor` so agent block +wiring, variable resolution, and the executor's retry/fallback policy are +measured alongside the loop. The `AgentToolUseResult` shape is deliberately +independent of the harness entry point so both can share scoring and reporting. diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts new file mode 100644 index 00000000000..43f7cbaae1c --- /dev/null +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts @@ -0,0 +1,34 @@ +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { afterAll, describe, expect, it, vi } from 'vitest' +import { runScenario } from '@/evals/agent-tool-use/harness' +import { writeEvalReport } from '@/evals/agent-tool-use/report' +import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' +import type { AgentToolUseResult } from '@/evals/agent-tool-use/types' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) + +const results: AgentToolUseResult[] = [] + +afterAll(() => { + const reportPath = process.env.EVAL_REPORT_PATH + if (reportPath) writeEvalReport(results, reportPath) +}) + +describe('agent tool-use eval suite', () => { + it.each(AGENT_TOOL_USE_SCENARIOS)('$id: $name', async (scenario) => { + const result = await runScenario(scenario) + results.push(result) + + const failed = result.checks.filter((entry) => !entry.passed) + expect( + failed, + failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined + ).toEqual([]) + }) +}) diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts new file mode 100644 index 00000000000..1bb95744fe0 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -0,0 +1,366 @@ +import { createLogger } from '@sim/logger' +import { collectStream } from '@sim/testing/helpers/async' +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMockFns } from '@sim/testing/mocks/tools.mock' +import { isRecordLike } from '@sim/utils/object' +import type { ChatCompletionChunk } from 'openai/resources/chat/completions' +import type { CompletionUsage } from 'openai/resources/completions' +import { + createOpenAICompatStreamingToolLoopStream, + type OpenAICompatCreateCompletion, +} from '@/providers/openai-compat/streaming-tool-loop' +import type { AgentStreamEvent } from '@/providers/stream-events' +import type { StreamingToolLoopComplete } from '@/providers/streaming-tool-loop-shared' +import type { ProviderToolConfig, TimeSegment } from '@/providers/types' +import type { ToolResponse } from '@/tools/types' +import type { + AgentToolUseResult, + AgentToolUseScenario, + EvalCheck, + EvalToolDefinition, + EvalToolInvocation, + ScriptedModelTurn, + ScriptedToolCall, +} from './types' + +/** + * Runs one scenario through the real OpenAI-compatible streaming tool loop and + * scores the result. Vitest owns the `@/tools`, `@/providers` and + * `@/providers/utils` module mocks; this module only drives them. + */ + +const EVAL_MODEL = 'eval-model' +const EVAL_PROVIDER = 'Eval' +const MAX_TOOL_ITERATIONS = 20 + +const logger = createLogger('AgentToolUseEval') + +interface CapturedToolCall { + name: string + arguments: Record + success: boolean + result?: unknown + duration?: number +} + +interface ToolCallList { + list: CapturedToolCall[] + count: number +} + +/** Raw call counts as a value the loop never reads; scenarios only assert on it. */ +const COMPLETION_USAGE = (): CompletionUsage => ({ + prompt_tokens: 10, + completion_tokens: 5, + total_tokens: 15, +}) + +let chunkCounter = 0 + +function chunk( + delta: ChatCompletionChunk.Choice['delta'] & { reasoning_content?: string }, + finishReason: ChatCompletionChunk.Choice['finish_reason'] = null, + usage?: CompletionUsage +): ChatCompletionChunk { + chunkCounter += 1 + return { + id: `eval-chunk-${chunkCounter}`, + object: 'chat.completion.chunk', + created: 0, + model: EVAL_MODEL, + choices: [{ index: 0, delta, finish_reason: finishReason, logprobs: null }], + ...(usage ? { usage } : {}), + } +} + +function turnToChunks(turn: ScriptedModelTurn): ChatCompletionChunk[] { + if (turn.kind === 'answer') { + return [ + ...(turn.thinking ? [chunk({ reasoning_content: turn.thinking })] : []), + chunk({ content: turn.content }, 'stop', COMPLETION_USAGE()), + ] + } + + const chunks: ChatCompletionChunk[] = [] + if (turn.thinking) chunks.push(chunk({ reasoning_content: turn.thinking })) + turn.calls.forEach((call, index) => { + chunks.push( + chunk({ + tool_calls: [ + { + index, + id: `call_${index}`, + type: 'function', + function: { + name: call.name, + arguments: call.argumentsJson ?? JSON.stringify(call.args ?? {}), + }, + }, + ], + }) + ) + }) + chunks.push(chunk({}, 'tool_calls', COMPLETION_USAGE())) + return chunks +} + +function createScriptedCompletion(scenario: AgentToolUseScenario): OpenAICompatCreateCompletion { + let turnIndex = 0 + return async () => { + const turn = scenario.script[turnIndex] + turnIndex += 1 + if (!turn) { + throw new Error( + `Scenario "${scenario.id}" requested model turn ${turnIndex} but only ${scenario.script.length} are scripted` + ) + } + return (async function* () { + for (const next of turnToChunks(turn)) yield next + })() + } +} + +function toProviderTools(tools: EvalToolDefinition[]): ProviderToolConfig[] { + return tools.map((tool) => ({ + id: tool.name, + description: tool.description, + params: {}, + parameters: { + type: tool.parameters?.type ?? 'object', + properties: tool.parameters?.properties ?? {}, + required: tool.parameters?.required ?? [], + }, + })) +} + +/** True when the loop will execute the call, so its result must be queued. */ +function isExecutable(call: ScriptedToolCall, toolNames: Set): boolean { + if (!toolNames.has(call.name)) return false + if (call.argumentsJson === undefined) return true + try { + return isRecordLike(JSON.parse(call.argumentsJson)) + } catch { + return false + } +} + +/** + * Results are queued per tool in script order. The loop may execute calls from + * one turn in any completion order, so keying by name keeps every call matched + * to the result the scenario intended. + */ +function buildResultQueues( + scenario: AgentToolUseScenario, + toolNames: Set +): Map { + const queues = new Map() + for (const turn of scenario.script) { + if (turn.kind !== 'tools') continue + for (const call of turn.calls) { + if (!isExecutable(call, toolNames)) continue + const spec = call.result + const queue = queues.get(call.name) ?? [] + queue.push({ + success: spec?.success ?? true, + output: spec?.output ?? {}, + ...(spec?.error ? { error: spec.error } : {}), + }) + queues.set(call.name, queue) + } + } + return queues +} + +function check(name: string, passed: boolean, detail: string): EvalCheck { + return { name, passed, detail } +} + +function sameSequence(actual: string[], expected: string[]): boolean { + return actual.length === expected.length && actual.every((name, i) => name === expected[i]) +} + +function matchesContent(content: string, expected: string | RegExp): boolean { + return typeof expected === 'string' ? content.includes(expected) : expected.test(content) +} + +function score( + scenario: AgentToolUseScenario, + toolCalls: CapturedToolCall[], + finalContent: string, + iterations: number, + error: unknown +): EvalCheck[] { + const expected = scenario.expect + const actualSequence = toolCalls.map((call) => call.name) + const checks: EvalCheck[] = [] + + if (expected.toolCallSequence) { + checks.push( + check( + 'tool-call-sequence', + sameSequence(actualSequence, expected.toolCallSequence), + `expected [${expected.toolCallSequence.join(', ')}], got [${actualSequence.join(', ')}]` + ) + ) + } + + if (expected.requiredTools) { + const missing = expected.requiredTools.filter((name) => !actualSequence.includes(name)) + checks.push( + check( + 'required-tools', + missing.length === 0, + missing.length === 0 ? 'all required tools called' : `missing [${missing.join(', ')}]` + ) + ) + } + + if (expected.forbiddenTools) { + const called = expected.forbiddenTools.filter((name) => actualSequence.includes(name)) + checks.push( + check( + 'forbidden-tools', + called.length === 0, + called.length === 0 ? 'no forbidden tools called' : `called [${called.join(', ')}]` + ) + ) + } + + if (expected.finalContent !== undefined) { + checks.push( + check( + 'final-content', + matchesContent(finalContent, expected.finalContent), + `final content ${JSON.stringify(finalContent)}` + ) + ) + } + + if (expected.maxIterations !== undefined) { + checks.push( + check( + 'max-iterations', + iterations <= expected.maxIterations, + `iterations ${iterations} (max ${expected.maxIterations})` + ) + ) + } + + const successful = toolCalls.filter((call) => call.success).length + if (expected.successfulToolCalls !== undefined) { + checks.push( + check( + 'successful-tool-calls', + successful === expected.successfulToolCalls, + `expected ${expected.successfulToolCalls}, got ${successful}` + ) + ) + } + + const errored = toolCalls.length - successful + if (expected.erroredToolCalls !== undefined) { + checks.push( + check( + 'errored-tool-calls', + errored === expected.erroredToolCalls, + `expected ${expected.erroredToolCalls}, got ${errored}` + ) + ) + } + + if (expected.completesWithoutError !== false) { + checks.push( + check( + 'completes-without-error', + error === undefined, + error === undefined ? 'loop settled' : String(error) + ) + ) + } + + return checks +} + +/** Runs and scores one scenario. */ +export async function runScenario(scenario: AgentToolUseScenario): Promise { + const toolNames = new Set(scenario.tools.map((tool) => tool.name)) + const resultQueues = buildResultQueues(scenario, toolNames) + const invocations: EvalToolInvocation[] = [] + + providersMock.MAX_TOOL_ITERATIONS = MAX_TOOL_ITERATIONS + providersUtilsMockFns.mockCalculateCost.mockReturnValue({ input: 0, output: 0, total: 0 }) + toolsMockFns.mockExecuteTool.mockImplementation( + async (toolId: string, params: Record): Promise => { + const startedAt = Date.now() + const response = resultQueues.get(toolId)?.shift() ?? { success: true, output: {} } + invocations.push({ + name: toolId, + arguments: params ?? {}, + success: response.success, + ...(response.error ? { error: response.error } : {}), + durationMs: Date.now() - startedAt, + }) + return response + } + ) + + const timeSegments: TimeSegment[] = [] + let completed: StreamingToolLoopComplete | undefined + let streamError: unknown + + const startedAt = Date.now() + try { + const stream = createOpenAICompatStreamingToolLoopStream({ + providerName: EVAL_PROVIDER, + request: { + model: EVAL_MODEL, + apiKey: 'eval-key', + messages: [], + tools: toProviderTools(scenario.tools), + }, + basePayload: { model: EVAL_MODEL }, + messages: [{ role: 'user', content: scenario.userMessage }], + createStream: createScriptedCompletion(scenario), + logger, + timeSegments, + onComplete: (result) => { + completed = result + }, + }) + await collectStream(stream as ReadableStream) + } catch (error) { + streamError = error + } + const latencyMs = Date.now() - startedAt + + const toolCalls = ((completed?.toolCalls as ToolCallList | undefined)?.list ?? []).slice() + const finalContent = completed?.content ?? '' + const iterations = completed?.iterations ?? 0 + const checks = score(scenario, toolCalls, finalContent, iterations, streamError) + const successful = toolCalls.filter((call) => call.success).length + + return { + id: scenario.id, + name: scenario.name, + category: scenario.category, + passed: checks.every((entry) => entry.passed), + checks, + finalContent, + toolInvocations: invocations, + metrics: { + iterations, + toolCalls: toolCalls.length, + successfulToolCalls: successful, + erroredToolCalls: toolCalls.length - successful, + latencyMs, + modelTimeMs: completed?.modelTime ?? 0, + toolsTimeMs: completed?.toolsTime ?? 0, + firstResponseTimeMs: completed?.firstResponseTime ?? 0, + inputTokens: completed?.tokens.input ?? 0, + outputTokens: completed?.tokens.output ?? 0, + totalTokens: completed?.tokens.total ?? 0, + }, + ...(streamError ? { error: String(streamError) } : {}), + } +} diff --git a/apps/sim/evals/agent-tool-use/report.ts b/apps/sim/evals/agent-tool-use/report.ts new file mode 100644 index 00000000000..494513d3948 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/report.ts @@ -0,0 +1,63 @@ +import { mkdirSync, writeFileSync } from 'node:fs' +import { dirname } from 'node:path' +import type { AgentToolUseResult } from './types' + +/** Machine-readable summary of one eval run. */ +export interface AgentToolUseEvalReport { + suite: 'agent-tool-use' + generatedAt: string + total: number + passed: number + failed: number + results: AgentToolUseResult[] +} + +export function buildEvalReport(results: AgentToolUseResult[]): AgentToolUseEvalReport { + const passed = results.filter((result) => result.passed).length + return { + suite: 'agent-tool-use', + generatedAt: new Date().toISOString(), + total: results.length, + passed, + failed: results.length - passed, + results, + } +} + +function escapeCell(value: string): string { + return value.replaceAll('|', '\\|').replaceAll('\n', ' ') +} + +function renderMarkdown(report: AgentToolUseEvalReport): string { + const header = [ + '# Agent tool-use eval report', + '', + `Generated: ${report.generatedAt}`, + '', + `**${report.passed}/${report.total} passed**`, + '', + '| Scenario | Category | Status | Iterations | Tools (ok/error) | Latency | Failed checks |', + '| --- | --- | --- | ---: | ---: | ---: | --- |', + ] + + const rows = report.results.map((result) => { + const failed = result.checks + .filter((entry) => !entry.passed) + .map((entry) => entry.name) + .join(', ') + return `| ${escapeCell(result.id)} | ${result.category} | ${result.passed ? '✅ pass' : '❌ fail'} | ${result.metrics.iterations} | ${result.metrics.successfulToolCalls}/${result.metrics.erroredToolCalls} | ${result.metrics.latencyMs}ms | ${failed || '—'} |` + }) + + return [...header, ...rows, ''].join('\n') +} + +/** + * Writes the JSON report to `reportPath` and a sibling Markdown summary. The + * caller supplies the path (`EVAL_REPORT_PATH`) so CI can upload it. + */ +export function writeEvalReport(results: AgentToolUseResult[], reportPath: string): void { + const report = buildEvalReport(results) + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) + writeFileSync(reportPath.replace(/\.json$/, '.md'), renderMarkdown(report)) +} diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts new file mode 100644 index 00000000000..a0042b33fb6 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -0,0 +1,352 @@ +import type { AgentToolUseScenario, EvalToolDefinition } from './types' + +const searchDocs: EvalToolDefinition = { + name: 'search_docs', + description: 'Search the product documentation for a query and return matching passages.', + parameters: { + type: 'object', + properties: { query: { type: 'string', description: 'Search query' } }, + required: ['query'], + }, +} + +const getWeather: EvalToolDefinition = { + name: 'get_weather', + description: 'Get the current weather for a city.', + parameters: { + type: 'object', + properties: { city: { type: 'string', description: 'City name' } }, + required: ['city'], + }, +} + +const sendEmail: EvalToolDefinition = { + name: 'send_email', + description: 'Send an email to a recipient.', + parameters: { + type: 'object', + properties: { + to: { type: 'string' }, + subject: { type: 'string' }, + body: { type: 'string' }, + }, + required: ['to', 'subject', 'body'], + }, +} + +const lookupOrder: EvalToolDefinition = { + name: 'lookup_order', + description: 'Look up an order by its customer-facing order number.', + parameters: { + type: 'object', + properties: { orderNumber: { type: 'string' } }, + required: ['orderNumber'], + }, +} + +const listFiles: EvalToolDefinition = { + name: 'list_files', + description: 'List files in a directory.', + parameters: { + type: 'object', + properties: { directory: { type: 'string' } }, + required: ['directory'], + }, +} + +const readFile: EvalToolDefinition = { + name: 'read_file', + description: 'Read the contents of a file.', + parameters: { + type: 'object', + properties: { path: { type: 'string' } }, + required: ['path'], + }, +} + +const flakyApi: EvalToolDefinition = { + name: 'flaky_api', + description: 'Fetch a value from an upstream API that intermittently returns 503.', + parameters: { + type: 'object', + properties: { resource: { type: 'string' } }, + required: ['resource'], + }, +} + +const getNews: EvalToolDefinition = { + name: 'get_news', + description: 'Get the top news headlines for a topic.', + parameters: { + type: 'object', + properties: { topic: { type: 'string' } }, + required: ['topic'], + }, +} + +/** + * The first suite: eight agent tool-use reliability cases. Each script is the + * model transcript; the loop, not the model, is what the assertions score. + */ +export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ + { + id: 'single-tool-lookup', + name: 'calls the one relevant tool and answers from its result', + category: 'tool-selection', + description: + 'A single retrieval tool is available. The loop must dispatch it once and surface the answer built from its result.', + userMessage: 'What is the API rate limit?', + tools: [searchDocs], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'api rate limit' }, + result: { + success: true, + output: { snippet: 'The API rate limit is 100 requests per minute.' }, + }, + }, + ], + }, + { kind: 'answer', content: 'The API rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['search_docs'], + finalContent: '100 requests per minute', + maxIterations: 3, + successfulToolCalls: 1, + erroredToolCalls: 0, + }, + }, + { + id: 'select-correct-tool', + name: 'selects the relevant tool and leaves the irrelevant ones unused', + category: 'tool-selection', + description: + 'Four tools are exposed; only the weather tool answers the question. The loop must not invoke the distractors.', + userMessage: 'What is the weather in Berlin right now?', + tools: [searchDocs, getWeather, sendEmail, lookupOrder], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'get_weather', + args: { city: 'Berlin' }, + result: { success: true, output: { temperatureC: 17, conditions: 'cloudy' } }, + }, + ], + }, + { kind: 'answer', content: 'It is 17°C and cloudy in Berlin.' }, + ], + expect: { + requiredTools: ['get_weather'], + forbiddenTools: ['search_docs', 'send_email', 'lookup_order'], + finalContent: '17°C', + maxIterations: 3, + }, + }, + { + id: 'multi-step-planning', + name: 'chains two dependent tools in order before answering', + category: 'planning', + description: + 'The model must list a directory, then read the file it found. The loop must preserve order and feed the first result into the second turn.', + userMessage: 'Summarize the notes file in /docs.', + tools: [listFiles, readFile], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'list_files', + args: { directory: '/docs' }, + result: { success: true, output: { files: ['notes.md'] } }, + }, + ], + }, + { + kind: 'tools', + calls: [ + { + name: 'read_file', + args: { path: '/docs/notes.md' }, + result: { success: true, output: { content: 'Ship the eval harness.' } }, + }, + ], + }, + { kind: 'answer', content: 'The notes say: ship the eval harness.' }, + ], + expect: { + toolCallSequence: ['list_files', 'read_file'], + finalContent: 'ship the eval harness', + maxIterations: 4, + successfulToolCalls: 2, + }, + }, + { + id: 'uses-retrieved-value', + name: 'answers from the retrieved value rather than inventing one', + category: 'retrieval', + description: + 'The order lookup returns a specific identifier. The final answer must carry that retrieved value through the loop.', + userMessage: 'What is the status of order 4471?', + tools: [lookupOrder], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'lookup_order', + args: { orderNumber: '4471' }, + result: { + success: true, + output: { id: 'A-1937', status: 'shipped', carrier: 'DHL' }, + }, + }, + ], + }, + { kind: 'answer', content: 'Order A-1937 has shipped with DHL.' }, + ], + expect: { + requiredTools: ['lookup_order'], + finalContent: /A-1937.*shipped/, + maxIterations: 3, + }, + }, + { + id: 'parallel-independent-tools', + name: 'dispatches independent tools from one turn', + category: 'planning', + description: + 'A single model turn requests two independent tools. The loop must execute both and fold both results back into the next turn.', + userMessage: 'Give me the weather and the news for Berlin.', + tools: [getWeather, getNews], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'get_weather', + args: { city: 'Berlin' }, + result: { success: true, output: { temperatureC: 12 } }, + }, + { + name: 'get_news', + args: { topic: 'Berlin' }, + result: { success: true, output: { headline: 'Transit strike ends' } }, + }, + ], + }, + { kind: 'answer', content: 'It is 12°C in Berlin. Top story: transit strike ends.' }, + ], + expect: { + toolCallSequence: ['get_weather', 'get_news'], + finalContent: /12°C.*transit strike ends/, + maxIterations: 3, + successfulToolCalls: 2, + }, + }, + { + id: 'recovers-from-tool-error', + name: 'retries a failing tool and completes the turn', + category: 'recovery', + description: + 'The first call errors with a 503 and the second succeeds. The error must be fed back to the model, not thrown out of the loop.', + userMessage: 'Fetch the current exchange rate from flaky_api.', + tools: [flakyApi], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'flaky_api', + args: { resource: 'exchange-rate' }, + result: { success: false, error: '503 Service Unavailable' }, + }, + ], + }, + { + kind: 'tools', + calls: [ + { + name: 'flaky_api', + args: { resource: 'exchange-rate' }, + result: { success: true, output: { usdToEur: 0.92 } }, + }, + ], + }, + { kind: 'answer', content: 'The exchange rate is 0.92 USD to EUR after a retry.' }, + ], + expect: { + toolCallSequence: ['flaky_api', 'flaky_api'], + finalContent: '0.92', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, + { + id: 'recovers-from-unknown-tool', + name: 'recovers when the model asks for a tool that does not exist', + category: 'recovery', + description: + 'The model hallucinates a tool name first. The loop must return a tool-not-found error to the model instead of failing the run.', + userMessage: 'Search the docs for the rate limit.', + tools: [searchDocs], + script: [ + { kind: 'tools', calls: [{ name: 'nonexistent_tool', args: { query: 'rate limit' } }] }, + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'rate limit' }, + result: { success: true, output: { snippet: '100 requests per minute.' } }, + }, + ], + }, + { kind: 'answer', content: 'The rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['nonexistent_tool', 'search_docs'], + finalContent: '100 requests per minute', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, + { + id: 'recovers-from-malformed-arguments', + name: 'does not execute a tool with malformed argument JSON and recovers', + category: 'recovery', + description: + 'The first call emits truncated JSON. The loop must skip execution, return the parse error, and let the corrected second call succeed.', + userMessage: 'Search the docs for the rate limit.', + tools: [searchDocs], + script: [ + { kind: 'tools', calls: [{ name: 'search_docs', argumentsJson: '{"query":' }] }, + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'rate limit' }, + result: { success: true, output: { snippet: '100 requests per minute.' } }, + }, + ], + }, + { kind: 'answer', content: 'The rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['search_docs', 'search_docs'], + finalContent: '100 requests per minute', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, +] diff --git a/apps/sim/evals/agent-tool-use/types.ts b/apps/sim/evals/agent-tool-use/types.ts new file mode 100644 index 00000000000..117f75c5bf6 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/types.ts @@ -0,0 +1,125 @@ +/** + * Declarative contract for the agent tool-use eval suite. + * + * A scenario is a scripted model transcript against the real OpenAI-compatible + * streaming tool loop (`providers/openai-compat/streaming-tool-loop.ts`). The + * loop is the harness under test: it must dispatch the tools the model asks + * for, feed results back, and recover from tool failures without losing the + * turn. The model is an input, so scenarios stay deterministic and run in CI + * with no provider key. + */ + +/** The agent behavior a scenario measures. */ +export type EvalCategory = 'tool-selection' | 'planning' | 'retrieval' | 'recovery' + +/** A tool exposed to the scripted model for one scenario. */ +export interface EvalToolDefinition { + name: string + description: string + parameters?: { + type?: string + properties?: Record + required?: string[] + } +} + +/** Result the stub tool returns for one scripted call. */ +export interface EvalToolResult { + success: boolean + output?: Record + error?: string +} + +/** One tool invocation the scripted model asks for. */ +export interface ScriptedToolCall { + name: string + args?: Record + /** + * Raw JSON emitted instead of serializing {@link args}. Used to exercise the + * loop's malformed-arguments guard, which must not execute the tool. + */ + argumentsJson?: string + /** Stub result for this call; defaults to `{ success: true, output: {} }`. */ + result?: EvalToolResult +} + +/** One model turn: either a set of tool calls or a final answer. */ +export type ScriptedModelTurn = + | { kind: 'tools'; calls: ScriptedToolCall[]; thinking?: string } + | { kind: 'answer'; content: string; thinking?: string } + +/** Assertions applied to a completed run. */ +export interface AgentToolUseExpectations { + /** Exact ordered sequence of tool names the model asked for. */ + toolCallSequence?: string[] + /** Tool names that must appear at least once. */ + requiredTools?: string[] + /** Tool names that must never be called. */ + forbiddenTools?: string[] + /** Substring or pattern the final assistant content must match. */ + finalContent?: string | RegExp + /** Upper bound on tool iterations. */ + maxIterations?: number + /** Exact count of tool calls that returned success. */ + successfulToolCalls?: number + /** Exact count of tool calls that returned an error. */ + erroredToolCalls?: number + /** Whether the loop must settle without throwing. Defaults to `true`. */ + completesWithoutError?: boolean +} + +/** A single agent behavior case. */ +export interface AgentToolUseScenario { + id: string + name: string + category: EvalCategory + description: string + userMessage: string + tools: EvalToolDefinition[] + script: ScriptedModelTurn[] + expect: AgentToolUseExpectations +} + +/** One scored expectation. */ +export interface EvalCheck { + name: string + passed: boolean + detail: string +} + +/** A tool call the loop actually executed through `executeTool`. */ +export interface EvalToolInvocation { + name: string + arguments: Record + success: boolean + error?: string + durationMs: number +} + +/** Measured properties of one completed run. */ +export interface AgentToolUseMetrics { + iterations: number + toolCalls: number + successfulToolCalls: number + erroredToolCalls: number + latencyMs: number + modelTimeMs: number + toolsTimeMs: number + firstResponseTimeMs: number + inputTokens: number + outputTokens: number + totalTokens: number +} + +/** The scored outcome of one scenario. */ +export interface AgentToolUseResult { + id: string + name: string + category: EvalCategory + passed: boolean + checks: EvalCheck[] + finalContent: string + toolInvocations: EvalToolInvocation[] + metrics: AgentToolUseMetrics + error?: string +} diff --git a/apps/sim/package.json b/apps/sim/package.json index edb48329506..67cdb6cb92d 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -24,6 +24,7 @@ "test:scim:e2e": "bun run scripts/test-scim-e2e.ts", "test:watch": "vitest", "test:coverage": "vitest run --coverage", + "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .", From af2e6cab2903483b994b07cce5a7c237f85dd329 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 16:18:55 +0530 Subject: [PATCH 2/7] feat(evals): add live DeepSeek model runs to the agent eval suite Replay the same scenarios against a real model. The model is the only thing that changes: runScenario now takes an optional completion transport and a live mode that relaxes exact assertions (ordered subsequence, minimum successes) and skips scripted-only recovery cases. - live.ts: OpenAI-compatible transport + DeepSeek factory - agent-tool-use.live.test.ts: K trials per scenario, gated on EVAL_LIVE=1 and DEEPSEEK_API_KEY, never runs in CI - live report with pass rates, avg iterations, latency, failed checks - test:evals:live script and README knobs --- apps/sim/evals/README.md | 29 +++++++ .../agent-tool-use.live.test.ts | 83 +++++++++++++++++++ apps/sim/evals/agent-tool-use/harness.ts | 74 +++++++++++++---- apps/sim/evals/agent-tool-use/live.ts | 61 ++++++++++++++ apps/sim/evals/agent-tool-use/report.ts | 68 ++++++++++++++- apps/sim/evals/agent-tool-use/scenarios.ts | 6 ++ apps/sim/evals/agent-tool-use/types.ts | 22 +++++ apps/sim/package.json | 1 + 8 files changed, 329 insertions(+), 15 deletions(-) create mode 100644 apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts create mode 100644 apps/sim/evals/agent-tool-use/live.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 604df341037..8ff4f1fb843 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -46,6 +46,35 @@ EVAL_REPORT_PATH=/tmp/agent-tool-use.json bunx vitest run evals/agent-tool-use The suite is also picked up by the normal `bun run test` run, so a regression fails CI even without the dedicated command. +## Run against a real model (live) + +The same scenarios can be replayed against a live model. This is opt-in and +never runs in CI. DeepSeek is wired first; any OpenAI-compatible provider works +through `createOpenAICompatLiveCompletion` in `live.ts`. + +```sh +cd apps/sim +DEEPSEEK_API_KEY=... bun run test:evals:live +``` + +Useful knobs: + +| Variable | Default | Meaning | +| --- | --- | --- | +| `EVAL_TRIALS` | `3` | Runs per scenario. Models are nondeterministic, so results are pass rates. | +| `EVAL_MIN_PASS_RATE` | `0` | When > 0, fail a scenario below this pass rate (0–1). | +| `EVAL_MODEL` | `deepseek-chat` | Model id sent to the provider. | +| `EVAL_TIMEOUT_MS` | `180000` | Per-request timeout. | +| `EVAL_REPORT_PATH` | `test-results/evals/agent-tool-use-live.json` | Report location. | + +Live runs relax exact assertions: `toolCallSequence` becomes an ordered +subsequence, `successfulToolCalls` becomes a minimum, and scripted-only cases +(malformed JSON, unknown tool) are skipped. A scenario-level `liveExpect` +overrides the scripted expectation where a real model cannot reproduce it (for +example, an exact retry count). The report is at +`test-results/evals/agent-tool-use-live.{json,md}` with pass rates, average +iterations, latency, and the failed check names. + ## Add a case 1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts new file mode 100644 index 00000000000..023ca29ecba --- /dev/null +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts @@ -0,0 +1,83 @@ +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { afterAll, describe, expect, it, vi } from 'vitest' +import { runScenario } from '@/evals/agent-tool-use/harness' +import { createDeepSeekLiveCompletion } from '@/evals/agent-tool-use/live' +import { writeLiveEvalReport } from '@/evals/agent-tool-use/report' +import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' +import type { AgentToolUseResult, LiveScenarioSummary } from '@/evals/agent-tool-use/types' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) + +/** + * Live agent tool-use evals. Opt-in only: + * + * EVAL_LIVE=1 DEEPSEEK_API_KEY=... \ + * bun run --cwd apps/sim test --mode live evals/agent-tool-use/agent-tool-use.live.test.ts + * + * Each scenario runs `EVAL_TRIALS` times (default 3) because a real model is + * nondeterministic. The report carries pass rates, not a single boolean. Set + * `EVAL_MIN_PASS_RATE` (0–1) to turn a pass-rate floor into a failing gate. + */ +const LIVE = process.env.EVAL_LIVE === '1' && Boolean(process.env.DEEPSEEK_API_KEY) +const TRIALS = Number(process.env.EVAL_TRIALS ?? '3') +const MIN_PASS_RATE = Number(process.env.EVAL_MIN_PASS_RATE ?? '0') +const MODEL = process.env.EVAL_MODEL ?? 'deepseek-chat' +const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '180000') + +const liveScenarios = AGENT_TOOL_USE_SCENARIOS.filter((scenario) => !scenario.scriptedOnly) +const summaries: LiveScenarioSummary[] = [] + +afterAll(() => { + if (!LIVE) return + writeLiveEvalReport( + summaries, + process.env.EVAL_REPORT_PATH ?? 'test-results/evals/agent-tool-use-live.json' + ) +}) + +describe.skipIf(!LIVE)('agent tool-use eval suite (live DeepSeek)', () => { + it.each(liveScenarios)( + '$id: $name', + async (scenario) => { + const completion = createDeepSeekLiveCompletion(MODEL) + const results: AgentToolUseResult[] = [] + + for (let trial = 0; trial < TRIALS; trial++) { + results.push( + await runScenario(scenario, { + completion, + mode: 'live', + model: MODEL, + providerName: 'DeepSeek', + }) + ) + } + + const passed = results.filter((result) => result.passed).length + const passRate = results.length === 0 ? 0 : passed / results.length + summaries.push({ + id: scenario.id, + name: scenario.name, + category: scenario.category, + trials: results.length, + passed, + passRate, + results, + }) + + if (MIN_PASS_RATE > 0) { + expect( + passRate, + `${scenario.id} passed ${passed}/${results.length} trials` + ).toBeGreaterThanOrEqual(MIN_PASS_RATE) + } + }, + TIMEOUT_MS + ) +}) diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts index 1bb95744fe0..631b422698e 100644 --- a/apps/sim/evals/agent-tool-use/harness.ts +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -12,9 +12,11 @@ import { } from '@/providers/openai-compat/streaming-tool-loop' import type { AgentStreamEvent } from '@/providers/stream-events' import type { StreamingToolLoopComplete } from '@/providers/streaming-tool-loop-shared' +import { adaptOpenAIChatToolSchema } from '@/providers/tool-schema-adapter' import type { ProviderToolConfig, TimeSegment } from '@/providers/types' import type { ToolResponse } from '@/tools/types' import type { + AgentToolUseExpectations, AgentToolUseResult, AgentToolUseScenario, EvalCheck, @@ -49,6 +51,20 @@ interface ToolCallList { count: number } +/** Scripted runs assert exact behavior; live runs assert outcomes across trials. */ +export type EvalRunMode = 'scripted' | 'live' + +/** Options for {@link runScenario}. */ +export interface RunScenarioOptions { + /** Model turns. Defaults to the scenario's scripted turns. */ + completion?: OpenAICompatCreateCompletion + mode?: EvalRunMode + /** Model id sent to a live provider and recorded in the run. */ + model?: string + /** Provider label used in loop diagnostics. */ + providerName?: string +} + /** Raw call counts as a value the loop never reads; scenarios only assert on it. */ const COMPLETION_USAGE = (): CompletionUsage => ({ prompt_tokens: 10, @@ -184,22 +200,34 @@ function matchesContent(content: string, expected: string | RegExp): boolean { return typeof expected === 'string' ? content.includes(expected) : expected.test(content) } +function isOrderedSubsequence(actual: string[], expected: string[]): boolean { + let index = 0 + for (const name of actual) { + if (name === expected[index]) index += 1 + } + return index === expected.length +} + function score( - scenario: AgentToolUseScenario, + expected: AgentToolUseExpectations, toolCalls: CapturedToolCall[], finalContent: string, iterations: number, - error: unknown + error: unknown, + mode: EvalRunMode ): EvalCheck[] { - const expected = scenario.expect const actualSequence = toolCalls.map((call) => call.name) const checks: EvalCheck[] = [] if (expected.toolCallSequence) { + const sequenceMatches = + mode === 'live' + ? isOrderedSubsequence(actualSequence, expected.toolCallSequence) + : sameSequence(actualSequence, expected.toolCallSequence) checks.push( check( 'tool-call-sequence', - sameSequence(actualSequence, expected.toolCallSequence), + sequenceMatches, `expected [${expected.toolCallSequence.join(', ')}], got [${actualSequence.join(', ')}]` ) ) @@ -249,17 +277,22 @@ function score( const successful = toolCalls.filter((call) => call.success).length if (expected.successfulToolCalls !== undefined) { + const successMatches = + mode === 'live' + ? successful >= expected.successfulToolCalls + : successful === expected.successfulToolCalls checks.push( check( 'successful-tool-calls', - successful === expected.successfulToolCalls, - `expected ${expected.successfulToolCalls}, got ${successful}` + successMatches, + `expected ${mode === 'live' ? 'at least ' : ''}${expected.successfulToolCalls}, got ${successful}` ) ) } const errored = toolCalls.length - successful - if (expected.erroredToolCalls !== undefined) { + /** A live model chooses its own retry count, so an exact error count is scripted-only. */ + if (expected.erroredToolCalls !== undefined && mode !== 'live') { checks.push( check( 'errored-tool-calls', @@ -283,9 +316,18 @@ function score( } /** Runs and scores one scenario. */ -export async function runScenario(scenario: AgentToolUseScenario): Promise { +export async function runScenario( + scenario: AgentToolUseScenario, + options: RunScenarioOptions = {} +): Promise { + const mode = options.mode ?? 'scripted' + const model = options.model ?? EVAL_MODEL + const providerName = options.providerName ?? EVAL_PROVIDER + const expected = + mode === 'live' ? { ...scenario.expect, ...scenario.liveExpect } : scenario.expect const toolNames = new Set(scenario.tools.map((tool) => tool.name)) const resultQueues = buildResultQueues(scenario, toolNames) + const providerTools = toProviderTools(scenario.tools) const invocations: EvalToolInvocation[] = [] providersMock.MAX_TOOL_ITERATIONS = MAX_TOOL_ITERATIONS @@ -312,18 +354,22 @@ export async function runScenario(scenario: AgentToolUseScenario): Promise adaptOpenAIChatToolSchema(tool)), }, - basePayload: { model: EVAL_MODEL }, messages: [{ role: 'user', content: scenario.userMessage }], - createStream: createScriptedCompletion(scenario), + createStream: options.completion ?? createScriptedCompletion(scenario), logger, timeSegments, + preserveAssistantReasoning: true, onComplete: (result) => { completed = result }, @@ -337,7 +383,7 @@ export async function runScenario(scenario: AgentToolUseScenario): Promise call.success).length return { diff --git a/apps/sim/evals/agent-tool-use/live.ts b/apps/sim/evals/agent-tool-use/live.ts new file mode 100644 index 00000000000..94ce8ae6da4 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/live.ts @@ -0,0 +1,61 @@ +import OpenAI from 'openai' +import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop' + +/** + * Real-model transport for the eval harness. Any OpenAI-compatible provider + * (OpenAI, DeepSeek, Groq, OpenRouter, …) works by passing its `baseURL`. + * + * The scripted suite stays the CI gate; this exists so the same scenarios can + * be replayed against a live model on demand. + */ +export interface OpenAICompatLiveModelOptions { + apiKey: string + model: string + baseURL?: string + /** Per-request timeout in ms. Live model calls routinely exceed the default. */ + timeoutMs?: number +} + +const DEFAULT_TIMEOUT_MS = 120_000 + +export function createOpenAICompatLiveCompletion( + options: OpenAICompatLiveModelOptions +): OpenAICompatCreateCompletion { + const client = new OpenAI({ + apiKey: options.apiKey, + ...(options.baseURL ? { baseURL: options.baseURL } : {}), + timeout: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, + maxRetries: 2, + }) + + return async (params, requestOptions) => + client.chat.completions.create( + { + ...params, + model: options.model, + stream: true, + // OpenAI-compatible providers require an opt-in to emit usage on streams. + stream_options: { include_usage: true }, + }, + requestOptions + ) +} + +/** + * DeepSeek's OpenAI-compatible endpoint. Reads `DEEPSEEK_API_KEY` and the + * optional `DEEPSEEK_BASE_URL` / `EVAL_MODEL` overrides. + */ +export function createDeepSeekLiveCompletion( + model = process.env.EVAL_MODEL ?? 'deepseek-chat' +): OpenAICompatCreateCompletion { + const apiKey = process.env.DEEPSEEK_API_KEY + if (!apiKey) { + throw new Error('DEEPSEEK_API_KEY is required for the live agent eval') + } + return createOpenAICompatLiveCompletion({ + apiKey, + baseURL: process.env.DEEPSEEK_BASE_URL ?? 'https://api.deepseek.com', + model, + ...(process.env.EVAL_TIMEOUT_MS ? { timeoutMs: Number(process.env.EVAL_TIMEOUT_MS) } : {}), + }) +} diff --git a/apps/sim/evals/agent-tool-use/report.ts b/apps/sim/evals/agent-tool-use/report.ts index 494513d3948..42d8983c6f0 100644 --- a/apps/sim/evals/agent-tool-use/report.ts +++ b/apps/sim/evals/agent-tool-use/report.ts @@ -1,6 +1,6 @@ import { mkdirSync, writeFileSync } from 'node:fs' import { dirname } from 'node:path' -import type { AgentToolUseResult } from './types' +import type { AgentToolUseResult, LiveScenarioSummary } from './types' /** Machine-readable summary of one eval run. */ export interface AgentToolUseEvalReport { @@ -61,3 +61,69 @@ export function writeEvalReport(results: AgentToolUseResult[], reportPath: strin writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) writeFileSync(reportPath.replace(/\.json$/, '.md'), renderMarkdown(report)) } + +/** Machine-readable summary of one live eval run across trials. */ +export interface LiveEvalReport { + suite: 'agent-tool-use-live' + generatedAt: string + trials: number + scenarios: number + totalPassRate: number + results: LiveScenarioSummary[] +} + +export function buildLiveEvalReport(summaries: LiveScenarioSummary[]): LiveEvalReport { + const trials = summaries.reduce((sum, summary) => sum + summary.trials, 0) + const passed = summaries.reduce((sum, summary) => sum + summary.passed, 0) + return { + suite: 'agent-tool-use-live', + generatedAt: new Date().toISOString(), + trials, + scenarios: summaries.length, + totalPassRate: trials === 0 ? 0 : passed / trials, + results: summaries, + } +} + +function average(values: number[]): number { + if (values.length === 0) return 0 + return values.reduce((sum, value) => sum + value, 0) / values.length +} + +function renderLiveMarkdown(report: LiveEvalReport): string { + const header = [ + '# Agent tool-use live eval report', + '', + `Generated: ${report.generatedAt}`, + '', + `**${(report.totalPassRate * 100).toFixed(0)}% pass across ${report.trials} trials / ${report.scenarios} scenarios**`, + '', + '| Scenario | Category | Pass rate | Trials | Avg iterations | Avg latency | Failed checks |', + '| --- | --- | ---: | ---: | ---: | ---: | --- |', + ] + + const rows = report.results.map((summary) => { + const failed = new Set() + for (const result of summary.results) { + for (const entry of result.checks) { + if (!entry.passed) failed.add(entry.name) + } + } + const avgIterations = average(summary.results.map((result) => result.metrics.iterations)) + const avgLatency = average(summary.results.map((result) => result.metrics.latencyMs)) + return `| ${escapeCell(summary.id)} | ${summary.category} | ${(summary.passRate * 100).toFixed(0)}% (${summary.passed}/${summary.trials}) | ${summary.trials} | ${avgIterations.toFixed(1)} | ${Math.round(avgLatency)}ms | ${[...failed].join(', ') || '—'} |` + }) + + return [...header, ...rows, ''].join('\n') +} + +/** + * Writes the live JSON report to `reportPath` and a sibling Markdown summary. + * Live results are statistical, so the report carries pass rates, not a boolean. + */ +export function writeLiveEvalReport(summaries: LiveScenarioSummary[], reportPath: string): void { + const report = buildLiveEvalReport(summaries) + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) + writeFileSync(reportPath.replace(/\.json$/, '.md'), renderLiveMarkdown(report)) +} diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts index a0042b33fb6..35f34cbc2d9 100644 --- a/apps/sim/evals/agent-tool-use/scenarios.ts +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -249,6 +249,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ maxIterations: 3, successfulToolCalls: 2, }, + /** The two tools are independent; a real model may emit them in either order. */ + liveExpect: { toolCallSequence: undefined, requiredTools: ['get_weather', 'get_news'] }, }, { id: 'recovers-from-tool-error', @@ -288,6 +290,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ successfulToolCalls: 1, erroredToolCalls: 1, }, + /** A live model decides its own retry count; only the grounded answer is asserted. */ + liveExpect: { toolCallSequence: undefined, requiredTools: ['flaky_api'] }, }, { id: 'recovers-from-unknown-tool', @@ -297,6 +301,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ 'The model hallucinates a tool name first. The loop must return a tool-not-found error to the model instead of failing the run.', userMessage: 'Search the docs for the rate limit.', tools: [searchDocs], + scriptedOnly: true, script: [ { kind: 'tools', calls: [{ name: 'nonexistent_tool', args: { query: 'rate limit' } }] }, { @@ -327,6 +332,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ 'The first call emits truncated JSON. The loop must skip execution, return the parse error, and let the corrected second call succeed.', userMessage: 'Search the docs for the rate limit.', tools: [searchDocs], + scriptedOnly: true, script: [ { kind: 'tools', calls: [{ name: 'search_docs', argumentsJson: '{"query":' }] }, { diff --git a/apps/sim/evals/agent-tool-use/types.ts b/apps/sim/evals/agent-tool-use/types.ts index 117f75c5bf6..036533acbe5 100644 --- a/apps/sim/evals/agent-tool-use/types.ts +++ b/apps/sim/evals/agent-tool-use/types.ts @@ -78,6 +78,17 @@ export interface AgentToolUseScenario { tools: EvalToolDefinition[] script: ScriptedModelTurn[] expect: AgentToolUseExpectations + /** + * Overrides applied only to live runs, merged over {@link expect}. Use when a + * scripted assertion (an exact retry count, a parallel call order) is not + * meaningful once a real model chooses the calls. + */ + liveExpect?: Partial + /** + * True when the case only makes sense with a scripted model (e.g. it requires + * the model to emit malformed JSON on demand). Excluded from live runs. + */ + scriptedOnly?: boolean } /** One scored expectation. */ @@ -111,6 +122,17 @@ export interface AgentToolUseMetrics { totalTokens: number } +/** One live scenario across its trials. */ +export interface LiveScenarioSummary { + id: string + name: string + category: EvalCategory + trials: number + passed: number + passRate: number + results: AgentToolUseResult[] +} + /** The scored outcome of one scenario. */ export interface AgentToolUseResult { id: string diff --git a/apps/sim/package.json b/apps/sim/package.json index 67cdb6cb92d..b9542513782 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -25,6 +25,7 @@ "test:watch": "vitest", "test:coverage": "vitest run --coverage", "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", + "test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .", From 433fd0f81f7f19425ea838a914136489eda4ffe1 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 16:24:20 +0530 Subject: [PATCH 3/7] test(evals): assert grounded behavior instead of exact phrasing in live mode The first live DeepSeek run exposed brittle assertions, not harness bugs: the model chained the tools correctly but the checks were case-sensitive and required an internal order id. Match the retrieved value case-insensitively and let live runs accept the grounded status rather than the internal id. --- apps/sim/evals/agent-tool-use/scenarios.ts | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts index 35f34cbc2d9..01a66099df1 100644 --- a/apps/sim/evals/agent-tool-use/scenarios.ts +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -182,7 +182,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ ], expect: { toolCallSequence: ['list_files', 'read_file'], - finalContent: 'ship the eval harness', + finalContent: /ship the eval harness/i, maxIterations: 4, successfulToolCalls: 2, }, @@ -216,6 +216,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ finalContent: /A-1937.*shipped/, maxIterations: 3, }, + /** A live model may answer with the user-facing order number and the grounded status. */ + liveExpect: { finalContent: /shipped/i }, }, { id: 'parallel-independent-tools', @@ -245,7 +247,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ ], expect: { toolCallSequence: ['get_weather', 'get_news'], - finalContent: /12°C.*transit strike ends/, + finalContent: /12°C[\s\S]*transit strike ends/i, maxIterations: 3, successfulToolCalls: 2, }, From 85525a412210c54290a9583b060bd4b3aa7db238 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:13:54 +0530 Subject: [PATCH 4/7] feat(evals): run agent scenarios through the DAGExecutor Add an executor-level harness: a real Start -> Agent workflow on DAGExecutor, with only executeProviderRequest mocked at the provider boundary. This covers agent-block input wiring, variable resolution from Start outputs, and executor run/error handling, which the direct loop harness cannot see. - executor-harness.ts: workflow builder + runExecutorScenario - shares the scorer (scoreExpectations) and report with the loop suite - two scenarios: Start->Agent output, and resolution - README documents adding an executor-level scenario --- apps/sim/evals/README.md | 32 ++- .../agent-tool-use.eval.test.ts | 51 +++- .../evals/agent-tool-use/executor-harness.ts | 262 ++++++++++++++++++ apps/sim/evals/agent-tool-use/harness.ts | 14 +- 4 files changed, 348 insertions(+), 11 deletions(-) create mode 100644 apps/sim/evals/agent-tool-use/executor-harness.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 8ff4f1fb843..f5115681220 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -100,6 +100,27 @@ A scenario is data, not code — there is no harness change needed for a new cas string. The loop must not execute the call and must return the parse error to the model. +## Executor-level scenarios + +[`agent-tool-use/executor-harness.ts`](./agent-tool-use/executor-harness.ts) +runs a case through a real `DAGExecutor`: a Start block → Agent block workflow, +with only the provider boundary (`executeProviderRequest`) mocked. This covers +what the loop harness cannot — agent-block input wiring, variable resolution +from Start outputs, and the executor's run/error handling. Tool dispatch stays +covered by the loop suite. + +Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: + +- `workflowInput` is exposed on the Start block; reference an output with + `` from the Agent prompt. +- `agent` is the Agent block config (`model`, `systemPrompt`, `userPrompt`). +- `providerResponse` is what the mocked provider returns (`content`, + `toolCalls`, `tokens`). +- `expect` uses the loop's checks plus `resolvedInput` (a substring that must + reach the provider messages) and `succeeds` (expected `ExecutionResult.success`). + +Both suites write one report, so executor rows appear alongside loop rows. + ## Report shape `report.json` is machine-readable for dashboards and trend tracking; `report.md` @@ -110,8 +131,9 @@ model/tool time, first-response time, and token usage. ## Scope and next steps -This suite evaluates the tool loop directly. The next layer is a scenario that -runs the same scripted model through the full `DAGExecutor` so agent block -wiring, variable resolution, and the executor's retry/fallback policy are -measured alongside the loop. The `AgentToolUseResult` shape is deliberately -independent of the harness entry point so both can share scoring and reporting. +Two harnesses share one result shape and report: the tool loop and the +`DAGExecutor`. The executor harness mocks the provider boundary, so the +executor's retry/fallback policy is not yet asserted; add a scenario with a +first-call rejection and a block retry config to cover it. Further expansion +(context/memory, model routing, subagent orchestration) is tracked as +follow-up work. diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts index 43f7cbaae1c..f13027a5d66 100644 --- a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts @@ -1,8 +1,14 @@ +import { + permissionCheckMock, + permissionCheckMockFns, +} from '@sim/testing/mocks/permission-check.mock' import { providersMock } from '@sim/testing/mocks/providers.mock' import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' -import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { providersUtilsMock, providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock' import { toolsMock } from '@sim/testing/mocks/tools.mock' -import { afterAll, describe, expect, it, vi } from 'vitest' +import { workspaceFileSecretProvenanceMock } from '@sim/testing/mocks/workspace-file-secret-provenance.mock' +import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' +import { EXECUTOR_SCENARIOS, runExecutorScenario } from '@/evals/agent-tool-use/executor-harness' import { runScenario } from '@/evals/agent-tool-use/harness' import { writeEvalReport } from '@/evals/agent-tool-use/report' import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' @@ -12,9 +18,37 @@ vi.mock('@/providers/conversation-history', () => providersConversationHistoryMo vi.mock('@/tools', () => toolsMock) vi.mock('@/providers/utils', () => providersUtilsMock) vi.mock('@/providers', () => providersMock) +vi.mock('@/ee/access-control/utils/permission-check', () => permissionCheckMock) +vi.mock( + '@/lib/uploads/contexts/workspace/workspace-file-secret-provenance', + () => workspaceFileSecretProvenanceMock +) +vi.mock('@/lib/memory/agent-turn-session', () => ({ + openAgentTurnSession: vi.fn(async () => undefined), +})) +vi.mock('@/lib/internal/mcp/discover-tools', () => ({ + discoverMcpServerToolsAsExecutor: vi.fn(async () => []), +})) +vi.mock('@/lib/internal/custom-tools/read-available-by-id-or-title', () => ({ + readAvailableCustomToolByIdOrTitleAsExecutor: vi.fn(async () => undefined), +})) +vi.mock('@/executor/utils/http', () => ({ + buildAuthHeaders: vi.fn(async () => ({ 'Content-Type': 'application/json' })), + buildAPIUrl: vi.fn((path: string) => path), + extractAPIErrorMessage: vi.fn(async () => 'request failed'), +})) +vi.mock('@/lib/execution/cancellation', () => ({ + subscribeToExecutionCancellation: vi.fn(async () => () => {}), + isExecutionCancelled: vi.fn(async () => false), +})) const results: AgentToolUseResult[] = [] +beforeEach(() => { + permissionCheckMockFns.mockValidateModelProvider.mockResolvedValue(undefined) + providersUtilsMockFns.mockGetProviderFromModel.mockReturnValue('mock-provider') +}) + afterAll(() => { const reportPath = process.env.EVAL_REPORT_PATH if (reportPath) writeEvalReport(results, reportPath) @@ -32,3 +66,16 @@ describe('agent tool-use eval suite', () => { ).toEqual([]) }) }) + +describe('agent executor eval suite', () => { + it.each(EXECUTOR_SCENARIOS)('$id: $name', async (scenario) => { + const result = await runExecutorScenario(scenario) + results.push(result) + + const failed = result.checks.filter((entry) => !entry.passed) + expect( + failed, + failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined + ).toEqual([]) + }) +}) diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts new file mode 100644 index 00000000000..1f6b5d67bdb --- /dev/null +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -0,0 +1,262 @@ +import { + createSerializedBlock, + createSerializedWorkflow, +} from '@sim/testing/factories/serialized-block.factory' +import { providersMockFns } from '@sim/testing/mocks/providers.mock' +import { DAGExecutor } from '@/executor/execution/executor' +import type { SerializedWorkflow } from '@/serializer/types' +import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' +import type { + AgentToolUseExpectations, + AgentToolUseResult, + EvalCategory, + EvalToolInvocation, +} from './types' + +/** + * Executor-level harness. + * + * Drives a real `DAGExecutor` run: Start block → Agent block. The provider + * boundary (`executeProviderRequest`) is the only thing mocked — the Agent + * block handler, input/variable resolution, and the executor run/error handling + * are real. Tool calls are what the mocked provider returns; tool *dispatch* is + * covered by the loop harness. + */ + +/** One tool call the mocked provider reports in its response. */ +export interface ExecutorProviderToolCall { + name: string + arguments?: Record + result?: unknown +} + +/** The provider response `executeProviderRequest` returns for one model call. */ +export interface ExecutorProviderResponse { + content: string + model?: string + tokens?: { input?: number; output?: number; total?: number } + toolCalls?: ExecutorProviderToolCall[] + cost?: unknown + timing?: unknown +} + +export interface ExecutorScenario { + id: string + name: string + category: EvalCategory + description: string + /** Exposed on the Start block and referenced from the Agent block. */ + workflowInput: Record + agent: { + model: string + systemPrompt?: string + userPrompt?: string + temperature?: number + } + /** One entry per model call; the last entry serves any extra fallback calls. */ + providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] + expect: AgentToolUseExpectations & { + /** Substring that must appear in the messages sent to the provider. */ + resolvedInput?: string + /** Expected `ExecutionResult.success`. */ + succeeds?: boolean + } +} + +function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { + const start = createSerializedBlock({ + id: 'start', + type: 'start_trigger', + name: 'Start', + }) + /** The trigger handler claims a block whose metadata says it is a trigger. */ + if (start.metadata) start.metadata.category = 'triggers' + const agent = createSerializedBlock({ + id: 'agent', + type: 'agent', + name: 'Eval Agent', + }) + agent.config.tool = 'agent' + agent.config.params = { + model: scenario.agent.model, + systemPrompt: scenario.agent.systemPrompt, + userPrompt: scenario.agent.userPrompt, + ...(scenario.agent.temperature !== undefined + ? { temperature: scenario.agent.temperature } + : {}), + } + + return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }]) +} + +/** + * Runs one executor scenario and scores it with the shared scorer, returning + * the same result shape as the loop harness so both land in one report. + */ +export async function runExecutorScenario( + scenario: ExecutorScenario, + options: { mode?: EvalRunMode } = {} +): Promise { + const mode = options.mode ?? 'scripted' + const responses = Array.isArray(scenario.providerResponse) + ? [...scenario.providerResponse] + : [scenario.providerResponse] + const requests: Array> = [] + let callIndex = 0 + + providersMockFns.mockExecuteProviderRequest.mockImplementation( + async (_providerId: string, request: Record) => { + requests.push(request) + const response = responses[Math.min(callIndex, responses.length - 1)] + callIndex += 1 + return { + content: response.content, + model: response.model ?? scenario.agent.model, + tokens: response.tokens ?? { input: 0, output: 0, total: 0 }, + toolCalls: response.toolCalls ?? [], + cost: response.cost ?? 0, + timing: response.timing ?? { total: 0 }, + } + } + ) + + const executor = new DAGExecutor({ + workflow: buildWorkflow(scenario), + workflowInput: scenario.workflowInput, + contextExtensions: { + workspaceId: 'eval-workspace', + executionId: 'eval-execution', + userId: 'eval-user', + }, + }) + + let result: { success?: boolean; output?: Record } | undefined + let runError: unknown + const startedAt = Date.now() + try { + result = (await executor.execute('eval-workflow')) as typeof result + } catch (error) { + runError = error + } + const latencyMs = Date.now() - startedAt + + const output = (result?.output ?? {}) as Record + const finalContent = typeof output.content === 'string' ? output.content : '' + const rawToolCalls = ((output.toolCalls as { list?: unknown[] } | undefined)?.list ?? + []) as Array> + + const toolCalls: ScoredToolCall[] = rawToolCalls.map((call) => ({ + name: typeof call.name === 'string' ? call.name : 'unknown', + success: true, + })) + const toolInvocations: EvalToolInvocation[] = rawToolCalls.map((call) => ({ + name: typeof call.name === 'string' ? call.name : 'unknown', + arguments: (call.arguments ?? {}) as Record, + success: true, + durationMs: typeof call.duration === 'number' ? call.duration : 0, + })) + + const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode) + + if (scenario.expect.resolvedInput !== undefined) { + const sent = JSON.stringify(requests) + checks.push({ + name: 'resolved-input', + passed: sent.includes(scenario.expect.resolvedInput), + detail: `looking for ${JSON.stringify(scenario.expect.resolvedInput)} in provider messages`, + }) + } + + if (scenario.expect.succeeds !== undefined) { + checks.push({ + name: 'workflow-success', + passed: result?.success === scenario.expect.succeeds, + detail: `success=${String(result?.success)}`, + }) + } + + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } + + return { + id: scenario.id, + name: scenario.name, + category: scenario.category, + passed: checks.every((entry) => entry.passed), + checks, + finalContent, + toolInvocations, + metrics: { + iterations: requests.length, + toolCalls: toolCalls.length, + successfulToolCalls: toolCalls.filter((call) => call.success).length, + erroredToolCalls: 0, + latencyMs, + modelTimeMs: 0, + toolsTimeMs: 0, + firstResponseTimeMs: 0, + inputTokens: tokens.input ?? 0, + outputTokens: tokens.output ?? 0, + totalTokens: tokens.total ?? 0, + }, + ...(runError ? { error: String(runError) } : {}), + } +} + +/** + * Executor-level scenarios. Two cover the wiring the loop suite cannot see: + * Start → Agent execution, and variable resolution from a Start output into the + * Agent's prompt. + */ +export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ + { + id: 'executor-agent-runs', + name: 'runs a Start → Agent workflow and surfaces the Agent output', + category: 'tool-selection', + description: + 'The real Agent block handler runs inside the DAG. The mocked provider reports one tool call; the executor result must carry the content and the tool call through.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + systemPrompt: 'You are a documentation assistant.', + userPrompt: 'What is the API rate limit?', + }, + providerResponse: { + content: 'The API rate limit is 100 requests per minute.', + toolCalls: [ + { + name: 'search_docs', + arguments: { query: 'api rate limit' }, + result: { snippet: 'The API rate limit is 100 requests per minute.' }, + }, + ], + tokens: { input: 10, output: 20, total: 30 }, + }, + expect: { + succeeds: true, + finalContent: '100 requests per minute', + toolCallSequence: ['search_docs'], + successfulToolCalls: 1, + }, + }, + { + id: 'executor-resolves-start-input', + name: 'resolves a Start output into the Agent prompt before the provider call', + category: 'planning', + description: + 'The Agent userPrompt references . The value must be resolved by the executor and reach the provider request, not passed through verbatim.', + workflowInput: { message: 'Summarize order A-1937' }, + agent: { + model: 'gpt-4o', + userPrompt: '', + }, + providerResponse: { + content: 'Order A-1937 shipped via DHL.', + tokens: { input: 8, output: 12, total: 20 }, + }, + expect: { + succeeds: true, + resolvedInput: 'Summarize order A-1937', + finalContent: /A-1937/, + }, + }, +] diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts index 631b422698e..0b369a06992 100644 --- a/apps/sim/evals/agent-tool-use/harness.ts +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -208,9 +208,15 @@ function isOrderedSubsequence(actual: string[], expected: string[]): boolean { return index === expected.length } -function score( +/** Minimal tool-call shape the scorer needs; both harnesses produce it. */ +export interface ScoredToolCall { + name: string + success: boolean +} + +export function scoreExpectations( expected: AgentToolUseExpectations, - toolCalls: CapturedToolCall[], + toolCalls: ScoredToolCall[], finalContent: string, iterations: number, error: unknown, @@ -307,7 +313,7 @@ function score( check( 'completes-without-error', error === undefined, - error === undefined ? 'loop settled' : String(error) + error === undefined ? 'completed without error' : String(error) ) ) } @@ -383,7 +389,7 @@ export async function runScenario( const toolCalls = ((completed?.toolCalls as ToolCallList | undefined)?.list ?? []).slice() const finalContent = completed?.content ?? '' const iterations = completed?.iterations ?? 0 - const checks = score(expected, toolCalls, finalContent, iterations, streamError, mode) + const checks = scoreExpectations(expected, toolCalls, finalContent, iterations, streamError, mode) const successful = toolCalls.filter((call) => call.success).length return { From fb7f54c73b9a8dbfe36fa951f3a500f3ec4967e2 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:48:02 +0530 Subject: [PATCH 5/7] test(evals): assert executor block retry on agent failure Add executor-retries-failed-block: the first provider call rejects, the Agent block has retry enabled, and the executor replays it. The run must complete with the second response. Verifies providerCalls === 2, and fails without the retry policy (checked locally: expected 2, got 1). --- apps/sim/evals/README.md | 15 +++-- .../evals/agent-tool-use/executor-harness.ts | 60 +++++++++++++++++-- 2 files changed, 63 insertions(+), 12 deletions(-) diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index f5115681220..ba9f662705f 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -117,7 +117,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: - `providerResponse` is what the mocked provider returns (`content`, `toolCalls`, `tokens`). - `expect` uses the loop's checks plus `resolvedInput` (a substring that must - reach the provider messages) and `succeeds` (expected `ExecutionResult.success`). + reach the provider messages), `succeeds` (expected `ExecutionResult.success`), + and `providerCalls` (exact provider call count). +- Set `agent.retry` to exercise the executor's per-block retry policy; make the + first `providerResponse` a `reject` and the retry lands on the next one. Both suites write one report, so executor rows appear alongside loop rows. @@ -132,8 +135,8 @@ model/tool time, first-response time, and token usage. ## Scope and next steps Two harnesses share one result shape and report: the tool loop and the -`DAGExecutor`. The executor harness mocks the provider boundary, so the -executor's retry/fallback policy is not yet asserted; add a scenario with a -first-call rejection and a block retry config to cover it. Further expansion -(context/memory, model routing, subagent orchestration) is tracked as -follow-up work. +`DAGExecutor`. The executor suite covers block retry — +`executor-retries-failed-block` makes the first provider call reject, the block +is replayed, and the run completes. Model fallback (`fallbackModels`) is not +asserted yet. Further expansion (context/memory, model routing, subagent +orchestration) is tracked as follow-up work. diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts index 1f6b5d67bdb..9cae3471870 100644 --- a/apps/sim/evals/agent-tool-use/executor-harness.ts +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -4,7 +4,7 @@ import { } from '@sim/testing/factories/serialized-block.factory' import { providersMockFns } from '@sim/testing/mocks/providers.mock' import { DAGExecutor } from '@/executor/execution/executor' -import type { SerializedWorkflow } from '@/serializer/types' +import type { SerializedBlock, SerializedWorkflow } from '@/serializer/types' import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' import type { AgentToolUseExpectations, @@ -30,8 +30,10 @@ export interface ExecutorProviderToolCall { result?: unknown } -/** The provider response `executeProviderRequest` returns for one model call. */ +/** One model call: either a response, or a rejection the block must recover from. */ export interface ExecutorProviderResponse { + /** When set, the call rejects with this message instead of resolving. */ + reject?: string content: string model?: string tokens?: { input?: number; output?: number; total?: number } @@ -52,26 +54,30 @@ export interface ExecutorScenario { systemPrompt?: string userPrompt?: string temperature?: number + /** Enables the executor's per-block retry policy for the Agent block. */ + retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number } } - /** One entry per model call; the last entry serves any extra fallback calls. */ + /** One entry per model call; the last entry serves any extra/retry calls. */ providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] expect: AgentToolUseExpectations & { /** Substring that must appear in the messages sent to the provider. */ resolvedInput?: string /** Expected `ExecutionResult.success`. */ succeeds?: boolean + /** Exact number of provider calls the executor made. */ + providerCalls?: number } } function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { - const start = createSerializedBlock({ + const start: SerializedBlock = createSerializedBlock({ id: 'start', type: 'start_trigger', name: 'Start', }) /** The trigger handler claims a block whose metadata says it is a trigger. */ if (start.metadata) start.metadata.category = 'triggers' - const agent = createSerializedBlock({ + const agent: SerializedBlock = createSerializedBlock({ id: 'agent', type: 'agent', name: 'Eval Agent', @@ -85,6 +91,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { ? { temperature: scenario.agent.temperature } : {}), } + if (scenario.agent.retry) agent.retry = scenario.agent.retry return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }]) } @@ -109,6 +116,7 @@ export async function runExecutorScenario( requests.push(request) const response = responses[Math.min(callIndex, responses.length - 1)] callIndex += 1 + if (response.reject) throw new Error(response.reject) return { content: response.content, model: response.model ?? scenario.agent.model, @@ -156,7 +164,14 @@ export async function runExecutorScenario( durationMs: typeof call.duration === 'number' ? call.duration : 0, })) - const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode) + const checks = scoreExpectations( + scenario.expect, + toolCalls, + finalContent, + requests.length, + runError, + mode + ) if (scenario.expect.resolvedInput !== undefined) { const sent = JSON.stringify(requests) @@ -175,6 +190,14 @@ export async function runExecutorScenario( }) } + if (scenario.expect.providerCalls !== undefined) { + checks.push({ + name: 'provider-calls', + passed: requests.length === scenario.expect.providerCalls, + detail: `expected ${scenario.expect.providerCalls}, got ${requests.length}`, + }) + } + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } return { @@ -259,4 +282,29 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ finalContent: /A-1937/, }, }, + { + id: 'executor-retries-failed-block', + name: 'retries a failed Agent block and completes the run', + category: 'recovery', + description: + 'The first provider call rejects with a 503. The block has retry enabled, so the executor replays it and the second call succeeds — proving the executor retry policy, not the agent handler, recovered the turn.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + userPrompt: 'What is the API rate limit?', + retry: { enabled: true, maxTries: 3, waitBetweenTriesMs: 0 }, + }, + providerResponse: [ + { reject: '503 Service Unavailable', content: '' }, + { + content: 'The API rate limit is 100 requests per minute.', + tokens: { input: 10, output: 20, total: 30 }, + }, + ], + expect: { + succeeds: true, + finalContent: '100 requests per minute', + providerCalls: 2, + }, + }, ] From c93426b3a6930d26ef9fdb76f0b1ddb815d09f00 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:59:53 +0530 Subject: [PATCH 6/7] test(evals): assert executor model fallback on primary failure Add executor-falls-back-to-secondary-model: the primary call rejects, the Agent block has a fallback model, and the handler serves the answer from gpt-4o-mini. Asserts providerCalls === 2 and lastRequestModel, and fails without the fallback row (checked locally: got gpt-4o, run errored). --- apps/sim/evals/README.md | 15 +++---- .../evals/agent-tool-use/executor-harness.ts | 41 +++++++++++++++++++ 2 files changed, 49 insertions(+), 7 deletions(-) diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index ba9f662705f..6c38a5f14d7 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -119,8 +119,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: - `expect` uses the loop's checks plus `resolvedInput` (a substring that must reach the provider messages), `succeeds` (expected `ExecutionResult.success`), and `providerCalls` (exact provider call count). -- Set `agent.retry` to exercise the executor's per-block retry policy; make the - first `providerResponse` a `reject` and the retry lands on the next one. +- Set `agent.retry` to exercise the executor's per-block retry policy, or + `agent.fallbackModels` to exercise model fallback. Make the first + `providerResponse` a `reject` and the next one succeeds; assert + `providerCalls` and `lastRequestModel` to prove which path recovered. Both suites write one report, so executor rows appear alongside loop rows. @@ -135,8 +137,7 @@ model/tool time, first-response time, and token usage. ## Scope and next steps Two harnesses share one result shape and report: the tool loop and the -`DAGExecutor`. The executor suite covers block retry — -`executor-retries-failed-block` makes the first provider call reject, the block -is replayed, and the run completes. Model fallback (`fallbackModels`) is not -asserted yet. Further expansion (context/memory, model routing, subagent -orchestration) is tracked as follow-up work. +`DAGExecutor`. The executor suite covers both recovery paths — block retry +(`executor-retries-failed-block`) and model fallback +(`executor-falls-back-to-secondary-model`). Further expansion (context/memory, +model routing, subagent orchestration) is tracked as follow-up work. diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts index 9cae3471870..52bc329a0f3 100644 --- a/apps/sim/evals/agent-tool-use/executor-harness.ts +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -56,6 +56,8 @@ export interface ExecutorScenario { temperature?: number /** Enables the executor's per-block retry policy for the Agent block. */ retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number } + /** Ordered models the Agent handler tries after the primary fails. */ + fallbackModels?: Array<{ model: string }> } /** One entry per model call; the last entry serves any extra/retry calls. */ providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] @@ -66,6 +68,8 @@ export interface ExecutorScenario { succeeds?: boolean /** Exact number of provider calls the executor made. */ providerCalls?: number + /** Model id sent on the final provider call (proves which candidate served). */ + lastRequestModel?: string } } @@ -90,6 +94,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { ...(scenario.agent.temperature !== undefined ? { temperature: scenario.agent.temperature } : {}), + ...(scenario.agent.fallbackModels ? { fallbackModels: scenario.agent.fallbackModels } : {}), } if (scenario.agent.retry) agent.retry = scenario.agent.retry @@ -198,6 +203,15 @@ export async function runExecutorScenario( }) } + if (scenario.expect.lastRequestModel !== undefined) { + const lastModel = (requests.at(-1) as { model?: string } | undefined)?.model + checks.push({ + name: 'last-request-model', + passed: lastModel === scenario.expect.lastRequestModel, + detail: `expected ${scenario.expect.lastRequestModel}, got ${String(lastModel)}`, + }) + } + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } return { @@ -307,4 +321,31 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ providerCalls: 2, }, }, + { + id: 'executor-falls-back-to-secondary-model', + name: 'falls back to the secondary model when the primary fails', + category: 'recovery', + description: + 'The primary model call rejects and the Agent block has a fallback model. The handler must serve the answer from the fallback and the run must complete.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + userPrompt: 'What is the API rate limit?', + fallbackModels: [{ model: 'gpt-4o-mini' }], + }, + providerResponse: [ + { reject: '429 rate limited', content: '' }, + { + content: 'The API rate limit is 100 requests per minute.', + model: 'gpt-4o-mini', + tokens: { input: 10, output: 20, total: 30 }, + }, + ], + expect: { + succeeds: true, + finalContent: '100 requests per minute', + providerCalls: 2, + lastRequestModel: 'gpt-4o-mini', + }, + }, ] From c12342a6701a570320e93bbf694fb174229f66ed Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 01:24:33 +0530 Subject: [PATCH 7/7] feat(evals): add agent context eval suite Drive the Agent block through the executor with conversation memory on. The memory read is stubbed per conversation id, so the provider request shows what the handler assembled: prior history, then the new prompt, system prompt preserved, correct conversation id. A wrong id surfaces as missing history and fails (checked locally). - agent-context/scenarios.ts: two context scenarios - executor-harness.ts: memory seam + assembly/isolation checks - test:evals:context script; README documents the suite --- apps/sim/evals/README.md | 19 +++++ .../agent-context/agent-context.eval.test.ts | 67 +++++++++++++++ apps/sim/evals/agent-context/scenarios.ts | 71 ++++++++++++++++ .../evals/agent-tool-use/executor-harness.ts | 84 +++++++++++++++++++ apps/sim/package.json | 1 + 5 files changed, 242 insertions(+) create mode 100644 apps/sim/evals/agent-context/agent-context.eval.test.ts create mode 100644 apps/sim/evals/agent-context/scenarios.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 6c38a5f14d7..98ba29f4cf2 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -126,6 +126,25 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: Both suites write one report, so executor rows appear alongside loop rows. +## Context evals + +[`agent-context/`](./agent-context/) drives the Agent block through the executor +with conversation memory on. The memory read is stubbed per conversation id, so +the provider request shows exactly what the handler assembled: prior history, +then the new user prompt, with the system prompt preserved, and the conversation +id must match. Windowing inside the memory service (`sliding_window`, token +budgets) is covered by its unit tests; this suite covers the assembly the model +sees. + +```sh +cd apps/sim +bun run test:evals:context # writes test-results/evals/agent-context.{json,md} +``` + +Scenarios live in [`agent-context/scenarios.ts`](./agent-context/scenarios.ts) +and reuse the executor harness, so a case is the same shape as an executor case +plus `agent.memory`. + ## Report shape `report.json` is machine-readable for dashboards and trend tracking; `report.md` diff --git a/apps/sim/evals/agent-context/agent-context.eval.test.ts b/apps/sim/evals/agent-context/agent-context.eval.test.ts new file mode 100644 index 00000000000..c5d59af8b97 --- /dev/null +++ b/apps/sim/evals/agent-context/agent-context.eval.test.ts @@ -0,0 +1,67 @@ +import { + permissionCheckMock, + permissionCheckMockFns, +} from '@sim/testing/mocks/permission-check.mock' +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock, providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { workspaceFileSecretProvenanceMock } from '@sim/testing/mocks/workspace-file-secret-provenance.mock' +import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' +import { AGENT_CONTEXT_SCENARIOS } from '@/evals/agent-context/scenarios' +import { runExecutorScenario } from '@/evals/agent-tool-use/executor-harness' +import { writeEvalReport } from '@/evals/agent-tool-use/report' +import type { AgentToolUseResult } from '@/evals/agent-tool-use/types' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) +vi.mock('@/ee/access-control/utils/permission-check', () => permissionCheckMock) +vi.mock( + '@/lib/uploads/contexts/workspace/workspace-file-secret-provenance', + () => workspaceFileSecretProvenanceMock +) +vi.mock('@/lib/memory/agent-turn-session', () => ({ + openAgentTurnSession: vi.fn(async () => undefined), +})) +vi.mock('@/lib/internal/mcp/discover-tools', () => ({ + discoverMcpServerToolsAsExecutor: vi.fn(async () => []), +})) +vi.mock('@/lib/internal/custom-tools/read-available-by-id-or-title', () => ({ + readAvailableCustomToolByIdOrTitleAsExecutor: vi.fn(async () => undefined), +})) +vi.mock('@/executor/utils/http', () => ({ + buildAuthHeaders: vi.fn(async () => ({ 'Content-Type': 'application/json' })), + buildAPIUrl: vi.fn((path: string) => path), + extractAPIErrorMessage: vi.fn(async () => 'request failed'), +})) +vi.mock('@/lib/execution/cancellation', () => ({ + subscribeToExecutionCancellation: vi.fn(async () => () => {}), + isExecutionCancelled: vi.fn(async () => false), +})) + +const results: AgentToolUseResult[] = [] + +beforeEach(() => { + permissionCheckMockFns.mockValidateModelProvider.mockResolvedValue(undefined) + providersUtilsMockFns.mockGetProviderFromModel.mockReturnValue('mock-provider') +}) + +afterAll(() => { + const reportPath = process.env.EVAL_CONTEXT_REPORT_PATH + if (reportPath) writeEvalReport(results, reportPath) +}) + +describe('agent context eval suite', () => { + it.each(AGENT_CONTEXT_SCENARIOS)('$id: $name', async (scenario) => { + const result = await runExecutorScenario(scenario) + results.push(result) + + const failed = result.checks.filter((entry) => !entry.passed) + expect( + failed, + failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined + ).toEqual([]) + }) +}) diff --git a/apps/sim/evals/agent-context/scenarios.ts b/apps/sim/evals/agent-context/scenarios.ts new file mode 100644 index 00000000000..4edae71d927 --- /dev/null +++ b/apps/sim/evals/agent-context/scenarios.ts @@ -0,0 +1,71 @@ +import type { ExecutorScenario } from '@/evals/agent-tool-use/executor-harness' + +/** + * Agent context evals. + * + * These drive the real Agent block inside the `DAGExecutor` with conversation + * memory on. The memory read is stubbed per conversation id, so the provider + * request shows exactly what the handler assembled: prior history, then the new + * user prompt, with the system prompt preserved. A handler that passed the + * wrong conversation id, dropped history, or reordered the prompt fails. + * + * Windowing inside the memory service (`sliding_window`, token budgets) is + * covered by its own unit tests; this suite covers the assembly the model sees. + */ +export const AGENT_CONTEXT_SCENARIOS: ExecutorScenario[] = [ + { + id: 'context-remembers-prior-turns', + name: 'includes prior conversation memory before the new user prompt', + category: 'retrieval', + description: + 'Memory holds two prior turns. The provider request must contain both, in order, followed by the new user prompt, with the system prompt present.', + workflowInput: {}, + agent: { + model: 'gpt-4o', + systemPrompt: 'You are a helpful assistant.', + userPrompt: 'What is my name?', + memory: { + conversationId: 'conv-ada', + history: [ + { role: 'user', content: 'My name is Ada.' }, + { role: 'assistant', content: 'Nice to meet you, Ada.' }, + ], + }, + }, + providerResponse: { + content: 'Your name is Ada.', + tokens: { input: 30, output: 5, total: 35 }, + }, + expect: { + succeeds: true, + finalContent: 'Ada', + }, + }, + { + id: 'context-isolates-conversations', + name: 'reads the conversation named by the block, not another', + category: 'retrieval', + description: + 'The memory stub only returns history for the block conversation id; any other id yields a placeholder. A handler that passed the wrong id would surface the placeholder and fail.', + workflowInput: {}, + agent: { + model: 'gpt-4o', + userPrompt: 'What did we decide?', + memory: { + conversationId: 'conv-b', + history: [ + { role: 'user', content: 'We decided to ship on Friday.' }, + { role: 'assistant', content: 'Shipping Friday.' }, + ], + }, + }, + providerResponse: { + content: 'You decided to ship on Friday.', + tokens: { input: 25, output: 6, total: 31 }, + }, + expect: { + succeeds: true, + finalContent: 'Friday', + }, + }, +] diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts index 52bc329a0f3..aa073254833 100644 --- a/apps/sim/evals/agent-tool-use/executor-harness.ts +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -3,7 +3,9 @@ import { createSerializedWorkflow, } from '@sim/testing/factories/serialized-block.factory' import { providersMockFns } from '@sim/testing/mocks/providers.mock' +import { vi } from 'vitest' import { DAGExecutor } from '@/executor/execution/executor' +import { memoryService } from '@/executor/handlers/agent/memory' import type { SerializedBlock, SerializedWorkflow } from '@/serializer/types' import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' import type { @@ -58,6 +60,14 @@ export interface ExecutorScenario { retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number } /** Ordered models the Agent handler tries after the primary fails. */ fallbackModels?: Array<{ model: string }> + /** + * Turns on conversation memory. `history` is what the mocked memory read + * returns for `conversationId`, so a wrong id surfaces as a failed check. + */ + memory?: { + conversationId: string + history: Array<{ role: 'user' | 'assistant' | 'system'; content: string }> + } } /** One entry per model call; the last entry serves any extra/retry calls. */ providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] @@ -73,6 +83,13 @@ export interface ExecutorScenario { } } +function lastMatchingIndex(contents: string[], needle: string): number { + for (let index = contents.length - 1; index >= 0; index--) { + if (contents[index].includes(needle)) return index + } + return -1 +} + function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { const start: SerializedBlock = createSerializedBlock({ id: 'start', @@ -95,6 +112,9 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { ? { temperature: scenario.agent.temperature } : {}), ...(scenario.agent.fallbackModels ? { fallbackModels: scenario.agent.fallbackModels } : {}), + ...(scenario.agent.memory + ? { memoryType: 'conversation', conversationId: scenario.agent.memory.conversationId } + : {}), } if (scenario.agent.retry) agent.retry = scenario.agent.retry @@ -133,6 +153,17 @@ export async function runExecutorScenario( } ) + const fetchedConversationIds: unknown[] = [] + if (scenario.agent.memory) { + const memory = scenario.agent.memory + vi.spyOn(memoryService, 'fetchMemoryMessages').mockImplementation(async (_ctx, inputs) => { + fetchedConversationIds.push(inputs.conversationId) + return inputs.conversationId === memory.conversationId + ? memory.history.map((message) => ({ ...message })) + : [{ role: 'user', content: '__WRONG_CONVERSATION__' }] + }) + } + const executor = new DAGExecutor({ workflow: buildWorkflow(scenario), workflowInput: scenario.workflowInput, @@ -212,6 +243,59 @@ export async function runExecutorScenario( }) } + if (scenario.agent.memory) { + const memory = scenario.agent.memory + const requestMessages = ((requests[0] as { messages?: unknown[] } | undefined)?.messages ?? + []) as Array<{ role?: string; content?: unknown }> + const contents = requestMessages.map((message) => + typeof message.content === 'string' ? message.content : '' + ) + + const missingHistory = memory.history.filter( + (message) => !contents.some((content) => content.includes(message.content)) + ) + checks.push({ + name: 'memory-history-in-request', + passed: missingHistory.length === 0, + detail: + missingHistory.length === 0 + ? `all ${memory.history.length} history messages reached the provider` + : `missing [${missingHistory.map((message) => message.content).join(', ')}]`, + }) + + const lastHistoryIndex = + memory.history.length === 0 + ? -1 + : Math.max(...memory.history.map((message) => lastMatchingIndex(contents, message.content))) + const promptIndex = scenario.agent.userPrompt + ? lastMatchingIndex(contents, scenario.agent.userPrompt) + : -1 + checks.push({ + name: 'memory-before-user-prompt', + passed: promptIndex >= 0 && promptIndex > lastHistoryIndex, + detail: `history ends at ${lastHistoryIndex}, user prompt at ${promptIndex}`, + }) + + if (scenario.agent.systemPrompt) { + const systemPrompt = scenario.agent.systemPrompt + checks.push({ + name: 'system-prompt-in-request', + passed: requestMessages.some( + (message) => message.role === 'system' && String(message.content).includes(systemPrompt) + ), + detail: 'configured system prompt reached the provider', + }) + } + + checks.push({ + name: 'conversation-id', + passed: + fetchedConversationIds.length > 0 && + fetchedConversationIds.every((id) => id === memory.conversationId), + detail: `expected ${memory.conversationId}, got [${fetchedConversationIds.join(', ')}]`, + }) + } + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } return { diff --git a/apps/sim/package.json b/apps/sim/package.json index b9542513782..1aed9726ed8 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -26,6 +26,7 @@ "test:coverage": "vitest run --coverage", "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", "test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use", + "test:evals:context": "EVAL_CONTEXT_REPORT_PATH=test-results/evals/agent-context.json vitest run evals/agent-context", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .",