Skip to content

Commit c12342a

Browse files
committed
feat(evals): add agent context eval suite
Drive the Agent block through the executor with conversation memory on. The memory read is stubbed per conversation id, so the provider request shows what the handler assembled: prior history, then the new prompt, system prompt preserved, correct conversation id. A wrong id surfaces as missing history and fails (checked locally). - agent-context/scenarios.ts: two context scenarios - executor-harness.ts: memory seam + assembly/isolation checks - test:evals:context script; README documents the suite
1 parent c93426b commit c12342a

5 files changed

Lines changed: 242 additions & 0 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -126,6 +126,25 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`:
126126

127127
Both suites write one report, so executor rows appear alongside loop rows.
128128

129+
## Context evals
130+
131+
[`agent-context/`](./agent-context/) drives the Agent block through the executor
132+
with conversation memory on. The memory read is stubbed per conversation id, so
133+
the provider request shows exactly what the handler assembled: prior history,
134+
then the new user prompt, with the system prompt preserved, and the conversation
135+
id must match. Windowing inside the memory service (`sliding_window`, token
136+
budgets) is covered by its unit tests; this suite covers the assembly the model
137+
sees.
138+
139+
```sh
140+
cd apps/sim
141+
bun run test:evals:context # writes test-results/evals/agent-context.{json,md}
142+
```
143+
144+
Scenarios live in [`agent-context/scenarios.ts`](./agent-context/scenarios.ts)
145+
and reuse the executor harness, so a case is the same shape as an executor case
146+
plus `agent.memory`.
147+
129148
## Report shape
130149

131150
`report.json` is machine-readable for dashboards and trend tracking; `report.md`
Lines changed: 67 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,67 @@
1+
import {
2+
permissionCheckMock,
3+
permissionCheckMockFns,
4+
} from '@sim/testing/mocks/permission-check.mock'
5+
import { providersMock } from '@sim/testing/mocks/providers.mock'
6+
import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock'
7+
import { providersUtilsMock, providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock'
8+
import { toolsMock } from '@sim/testing/mocks/tools.mock'
9+
import { workspaceFileSecretProvenanceMock } from '@sim/testing/mocks/workspace-file-secret-provenance.mock'
10+
import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest'
11+
import { AGENT_CONTEXT_SCENARIOS } from '@/evals/agent-context/scenarios'
12+
import { runExecutorScenario } from '@/evals/agent-tool-use/executor-harness'
13+
import { writeEvalReport } from '@/evals/agent-tool-use/report'
14+
import type { AgentToolUseResult } from '@/evals/agent-tool-use/types'
15+
16+
vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock)
17+
vi.mock('@/tools', () => toolsMock)
18+
vi.mock('@/providers/utils', () => providersUtilsMock)
19+
vi.mock('@/providers', () => providersMock)
20+
vi.mock('@/ee/access-control/utils/permission-check', () => permissionCheckMock)
21+
vi.mock(
22+
'@/lib/uploads/contexts/workspace/workspace-file-secret-provenance',
23+
() => workspaceFileSecretProvenanceMock
24+
)
25+
vi.mock('@/lib/memory/agent-turn-session', () => ({
26+
openAgentTurnSession: vi.fn(async () => undefined),
27+
}))
28+
vi.mock('@/lib/internal/mcp/discover-tools', () => ({
29+
discoverMcpServerToolsAsExecutor: vi.fn(async () => []),
30+
}))
31+
vi.mock('@/lib/internal/custom-tools/read-available-by-id-or-title', () => ({
32+
readAvailableCustomToolByIdOrTitleAsExecutor: vi.fn(async () => undefined),
33+
}))
34+
vi.mock('@/executor/utils/http', () => ({
35+
buildAuthHeaders: vi.fn(async () => ({ 'Content-Type': 'application/json' })),
36+
buildAPIUrl: vi.fn((path: string) => path),
37+
extractAPIErrorMessage: vi.fn(async () => 'request failed'),
38+
}))
39+
vi.mock('@/lib/execution/cancellation', () => ({
40+
subscribeToExecutionCancellation: vi.fn(async () => () => {}),
41+
isExecutionCancelled: vi.fn(async () => false),
42+
}))
43+
44+
const results: AgentToolUseResult[] = []
45+
46+
beforeEach(() => {
47+
permissionCheckMockFns.mockValidateModelProvider.mockResolvedValue(undefined)
48+
providersUtilsMockFns.mockGetProviderFromModel.mockReturnValue('mock-provider')
49+
})
50+
51+
afterAll(() => {
52+
const reportPath = process.env.EVAL_CONTEXT_REPORT_PATH
53+
if (reportPath) writeEvalReport(results, reportPath)
54+
})
55+
56+
describe('agent context eval suite', () => {
57+
it.each(AGENT_CONTEXT_SCENARIOS)('$id: $name', async (scenario) => {
58+
const result = await runExecutorScenario(scenario)
59+
results.push(result)
60+
61+
const failed = result.checks.filter((entry) => !entry.passed)
62+
expect(
63+
failed,
64+
failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined
65+
).toEqual([])
66+
})
67+
})
Lines changed: 71 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,71 @@
1+
import type { ExecutorScenario } from '@/evals/agent-tool-use/executor-harness'
2+
3+
/**
4+
* Agent context evals.
5+
*
6+
* These drive the real Agent block inside the `DAGExecutor` with conversation
7+
* memory on. The memory read is stubbed per conversation id, so the provider
8+
* request shows exactly what the handler assembled: prior history, then the new
9+
* user prompt, with the system prompt preserved. A handler that passed the
10+
* wrong conversation id, dropped history, or reordered the prompt fails.
11+
*
12+
* Windowing inside the memory service (`sliding_window`, token budgets) is
13+
* covered by its own unit tests; this suite covers the assembly the model sees.
14+
*/
15+
export const AGENT_CONTEXT_SCENARIOS: ExecutorScenario[] = [
16+
{
17+
id: 'context-remembers-prior-turns',
18+
name: 'includes prior conversation memory before the new user prompt',
19+
category: 'retrieval',
20+
description:
21+
'Memory holds two prior turns. The provider request must contain both, in order, followed by the new user prompt, with the system prompt present.',
22+
workflowInput: {},
23+
agent: {
24+
model: 'gpt-4o',
25+
systemPrompt: 'You are a helpful assistant.',
26+
userPrompt: 'What is my name?',
27+
memory: {
28+
conversationId: 'conv-ada',
29+
history: [
30+
{ role: 'user', content: 'My name is Ada.' },
31+
{ role: 'assistant', content: 'Nice to meet you, Ada.' },
32+
],
33+
},
34+
},
35+
providerResponse: {
36+
content: 'Your name is Ada.',
37+
tokens: { input: 30, output: 5, total: 35 },
38+
},
39+
expect: {
40+
succeeds: true,
41+
finalContent: 'Ada',
42+
},
43+
},
44+
{
45+
id: 'context-isolates-conversations',
46+
name: 'reads the conversation named by the block, not another',
47+
category: 'retrieval',
48+
description:
49+
'The memory stub only returns history for the block conversation id; any other id yields a placeholder. A handler that passed the wrong id would surface the placeholder and fail.',
50+
workflowInput: {},
51+
agent: {
52+
model: 'gpt-4o',
53+
userPrompt: 'What did we decide?',
54+
memory: {
55+
conversationId: 'conv-b',
56+
history: [
57+
{ role: 'user', content: 'We decided to ship on Friday.' },
58+
{ role: 'assistant', content: 'Shipping Friday.' },
59+
],
60+
},
61+
},
62+
providerResponse: {
63+
content: 'You decided to ship on Friday.',
64+
tokens: { input: 25, output: 6, total: 31 },
65+
},
66+
expect: {
67+
succeeds: true,
68+
finalContent: 'Friday',
69+
},
70+
},
71+
]

‎apps/sim/evals/agent-tool-use/executor-harness.ts‎

Lines changed: 84 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3,7 +3,9 @@ import {
33
createSerializedWorkflow,
44
} from '@sim/testing/factories/serialized-block.factory'
55
import { providersMockFns } from '@sim/testing/mocks/providers.mock'
6+
import { vi } from 'vitest'
67
import { DAGExecutor } from '@/executor/execution/executor'
8+
import { memoryService } from '@/executor/handlers/agent/memory'
79
import type { SerializedBlock, SerializedWorkflow } from '@/serializer/types'
810
import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness'
911
import type {
@@ -58,6 +60,14 @@ export interface ExecutorScenario {
5860
retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number }
5961
/** Ordered models the Agent handler tries after the primary fails. */
6062
fallbackModels?: Array<{ model: string }>
63+
/**
64+
* Turns on conversation memory. `history` is what the mocked memory read
65+
* returns for `conversationId`, so a wrong id surfaces as a failed check.
66+
*/
67+
memory?: {
68+
conversationId: string
69+
history: Array<{ role: 'user' | 'assistant' | 'system'; content: string }>
70+
}
6171
}
6272
/** One entry per model call; the last entry serves any extra/retry calls. */
6373
providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[]
@@ -73,6 +83,13 @@ export interface ExecutorScenario {
7383
}
7484
}
7585

86+
function lastMatchingIndex(contents: string[], needle: string): number {
87+
for (let index = contents.length - 1; index >= 0; index--) {
88+
if (contents[index].includes(needle)) return index
89+
}
90+
return -1
91+
}
92+
7693
function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
7794
const start: SerializedBlock = createSerializedBlock({
7895
id: 'start',
@@ -95,6 +112,9 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
95112
? { temperature: scenario.agent.temperature }
96113
: {}),
97114
...(scenario.agent.fallbackModels ? { fallbackModels: scenario.agent.fallbackModels } : {}),
115+
...(scenario.agent.memory
116+
? { memoryType: 'conversation', conversationId: scenario.agent.memory.conversationId }
117+
: {}),
98118
}
99119
if (scenario.agent.retry) agent.retry = scenario.agent.retry
100120

@@ -133,6 +153,17 @@ export async function runExecutorScenario(
133153
}
134154
)
135155

156+
const fetchedConversationIds: unknown[] = []
157+
if (scenario.agent.memory) {
158+
const memory = scenario.agent.memory
159+
vi.spyOn(memoryService, 'fetchMemoryMessages').mockImplementation(async (_ctx, inputs) => {
160+
fetchedConversationIds.push(inputs.conversationId)
161+
return inputs.conversationId === memory.conversationId
162+
? memory.history.map((message) => ({ ...message }))
163+
: [{ role: 'user', content: '__WRONG_CONVERSATION__' }]
164+
})
165+
}
166+
136167
const executor = new DAGExecutor({
137168
workflow: buildWorkflow(scenario),
138169
workflowInput: scenario.workflowInput,
@@ -212,6 +243,59 @@ export async function runExecutorScenario(
212243
})
213244
}
214245

246+
if (scenario.agent.memory) {
247+
const memory = scenario.agent.memory
248+
const requestMessages = ((requests[0] as { messages?: unknown[] } | undefined)?.messages ??
249+
[]) as Array<{ role?: string; content?: unknown }>
250+
const contents = requestMessages.map((message) =>
251+
typeof message.content === 'string' ? message.content : ''
252+
)
253+
254+
const missingHistory = memory.history.filter(
255+
(message) => !contents.some((content) => content.includes(message.content))
256+
)
257+
checks.push({
258+
name: 'memory-history-in-request',
259+
passed: missingHistory.length === 0,
260+
detail:
261+
missingHistory.length === 0
262+
? `all ${memory.history.length} history messages reached the provider`
263+
: `missing [${missingHistory.map((message) => message.content).join(', ')}]`,
264+
})
265+
266+
const lastHistoryIndex =
267+
memory.history.length === 0
268+
? -1
269+
: Math.max(...memory.history.map((message) => lastMatchingIndex(contents, message.content)))
270+
const promptIndex = scenario.agent.userPrompt
271+
? lastMatchingIndex(contents, scenario.agent.userPrompt)
272+
: -1
273+
checks.push({
274+
name: 'memory-before-user-prompt',
275+
passed: promptIndex >= 0 && promptIndex > lastHistoryIndex,
276+
detail: `history ends at ${lastHistoryIndex}, user prompt at ${promptIndex}`,
277+
})
278+
279+
if (scenario.agent.systemPrompt) {
280+
const systemPrompt = scenario.agent.systemPrompt
281+
checks.push({
282+
name: 'system-prompt-in-request',
283+
passed: requestMessages.some(
284+
(message) => message.role === 'system' && String(message.content).includes(systemPrompt)
285+
),
286+
detail: 'configured system prompt reached the provider',
287+
})
288+
}
289+
290+
checks.push({
291+
name: 'conversation-id',
292+
passed:
293+
fetchedConversationIds.length > 0 &&
294+
fetchedConversationIds.every((id) => id === memory.conversationId),
295+
detail: `expected ${memory.conversationId}, got [${fetchedConversationIds.join(', ')}]`,
296+
})
297+
}
298+
215299
const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number }
216300

217301
return {

‎apps/sim/package.json‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -26,6 +26,7 @@
2626
"test:coverage": "vitest run --coverage",
2727
"test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use",
2828
"test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use",
29+
"test:evals:context": "EVAL_CONTEXT_REPORT_PATH=test-results/evals/agent-context.json vitest run evals/agent-context",
2930
"email:dev": "email dev --dir components/emails",
3031
"type-check": "tsc --noEmit",
3132
"lint": "biome check --write --unsafe .",

0 commit comments

Comments
 (0)