Skip to content

Commit 2023d07

Browse files
committed
feat(evals): judge the remaining open-ended scenarios
Add rubrics to single-tool-lookup, select-correct-tool, multi-step-planning, parallel-independent-tools, recovers-from-tool-error, and near-duplicate-names. In live mode each drops its brittle finalContent substring/regex (via liveExpect) so the rubric decides phrasing and grounding, while requiredTools and tool sequences still guard behavior. Scripted CI keeps the deterministic checks.
1 parent 74faa92 commit 2023d07

1 file changed

Lines changed: 59 additions & 3 deletions

File tree

‎apps/sim/evals/agent-tool-use/scenarios.ts‎

Lines changed: 59 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -191,6 +191,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
191191
successfulToolCalls: 1,
192192
erroredToolCalls: 0,
193193
},
194+
judge: {
195+
criteria: [
196+
{ id: 'grounding', description: 'the rate limit it states matches the retrieved snippet' },
197+
{ id: 'completeness', description: 'answers the user request' },
198+
],
199+
minScore: 0.7,
200+
},
201+
liveExpect: { finalContent: undefined },
194202
},
195203
{
196204
id: 'select-correct-tool',
@@ -219,6 +227,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
219227
finalContent: '17°C',
220228
maxIterations: 3,
221229
},
230+
judge: {
231+
criteria: [
232+
{ id: 'grounding', description: 'the conditions it states match the weather tool result' },
233+
{ id: 'completeness', description: 'answers the user request' },
234+
],
235+
minScore: 0.7,
236+
},
237+
liveExpect: { finalContent: undefined },
222238
},
223239
{
224240
id: 'multi-step-planning',
@@ -257,6 +273,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
257273
maxIterations: 4,
258274
successfulToolCalls: 2,
259275
},
276+
judge: {
277+
criteria: [
278+
{ id: 'grounding', description: "summarizes the file's actual contents" },
279+
{ id: 'completeness', description: 'answers the user request' },
280+
],
281+
minScore: 0.7,
282+
},
283+
liveExpect: { finalContent: undefined },
260284
},
261285
{
262286
id: 'uses-retrieved-value',
@@ -332,8 +356,19 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
332356
maxIterations: 3,
333357
successfulToolCalls: 2,
334358
},
359+
judge: {
360+
criteria: [
361+
{ id: 'completeness', description: 'reports both the weather and the news' },
362+
{ id: 'grounding', description: 'the values it states match the tool results' },
363+
],
364+
minScore: 0.7,
365+
},
335366
/** The two tools are independent; a real model may emit them in either order. */
336-
liveExpect: { toolCallSequence: undefined, requiredTools: ['get_weather', 'get_news'] },
367+
liveExpect: {
368+
toolCallSequence: undefined,
369+
requiredTools: ['get_weather', 'get_news'],
370+
finalContent: undefined,
371+
},
337372
},
338373
{
339374
id: 'recovers-from-tool-error',
@@ -373,8 +408,22 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
373408
successfulToolCalls: 1,
374409
erroredToolCalls: 1,
375410
},
411+
judge: {
412+
criteria: [
413+
{ id: 'grounding', description: 'states the exchange rate returned by the tool' },
414+
{
415+
id: 'recovery',
416+
description: 'makes clear the first attempt failed and the retry succeeded',
417+
},
418+
],
419+
minScore: 0.7,
420+
},
376421
/** A live model decides its own retry count; only the grounded answer is asserted. */
377-
liveExpect: { toolCallSequence: undefined, requiredTools: ['flaky_api'] },
422+
liveExpect: {
423+
toolCallSequence: undefined,
424+
requiredTools: ['flaky_api'],
425+
finalContent: undefined,
426+
},
378427
},
379428
{
380429
id: 'recovers-from-unknown-tool',
@@ -621,7 +670,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
621670
finalContent: /Europe\/Berlin/,
622671
maxIterations: 3,
623672
},
673+
judge: {
674+
criteria: [
675+
{ id: 'grounding', description: 'states the timezone returned by the settings tool' },
676+
{ id: 'completeness', description: 'answers the user request' },
677+
],
678+
minScore: 0.7,
679+
},
624680
/** Over-calling the profile tool is inefficiency, not wrong-tool selection. */
625-
liveExpect: { forbiddenTools: undefined },
681+
liveExpect: { forbiddenTools: undefined, finalContent: undefined },
626682
},
627683
]

0 commit comments

Comments
 (0)