@@ -191,6 +191,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
191191 successfulToolCalls : 1 ,
192192 erroredToolCalls : 0 ,
193193 } ,
194+ judge : {
195+ criteria : [
196+ { id : 'grounding' , description : 'the rate limit it states matches the retrieved snippet' } ,
197+ { id : 'completeness' , description : 'answers the user request' } ,
198+ ] ,
199+ minScore : 0.7 ,
200+ } ,
201+ liveExpect : { finalContent : undefined } ,
194202 } ,
195203 {
196204 id : 'select-correct-tool' ,
@@ -219,6 +227,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
219227 finalContent : '17°C' ,
220228 maxIterations : 3 ,
221229 } ,
230+ judge : {
231+ criteria : [
232+ { id : 'grounding' , description : 'the conditions it states match the weather tool result' } ,
233+ { id : 'completeness' , description : 'answers the user request' } ,
234+ ] ,
235+ minScore : 0.7 ,
236+ } ,
237+ liveExpect : { finalContent : undefined } ,
222238 } ,
223239 {
224240 id : 'multi-step-planning' ,
@@ -257,6 +273,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
257273 maxIterations : 4 ,
258274 successfulToolCalls : 2 ,
259275 } ,
276+ judge : {
277+ criteria : [
278+ { id : 'grounding' , description : "summarizes the file's actual contents" } ,
279+ { id : 'completeness' , description : 'answers the user request' } ,
280+ ] ,
281+ minScore : 0.7 ,
282+ } ,
283+ liveExpect : { finalContent : undefined } ,
260284 } ,
261285 {
262286 id : 'uses-retrieved-value' ,
@@ -332,8 +356,19 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
332356 maxIterations : 3 ,
333357 successfulToolCalls : 2 ,
334358 } ,
359+ judge : {
360+ criteria : [
361+ { id : 'completeness' , description : 'reports both the weather and the news' } ,
362+ { id : 'grounding' , description : 'the values it states match the tool results' } ,
363+ ] ,
364+ minScore : 0.7 ,
365+ } ,
335366 /** The two tools are independent; a real model may emit them in either order. */
336- liveExpect : { toolCallSequence : undefined , requiredTools : [ 'get_weather' , 'get_news' ] } ,
367+ liveExpect : {
368+ toolCallSequence : undefined ,
369+ requiredTools : [ 'get_weather' , 'get_news' ] ,
370+ finalContent : undefined ,
371+ } ,
337372 } ,
338373 {
339374 id : 'recovers-from-tool-error' ,
@@ -373,8 +408,22 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
373408 successfulToolCalls : 1 ,
374409 erroredToolCalls : 1 ,
375410 } ,
411+ judge : {
412+ criteria : [
413+ { id : 'grounding' , description : 'states the exchange rate returned by the tool' } ,
414+ {
415+ id : 'recovery' ,
416+ description : 'makes clear the first attempt failed and the retry succeeded' ,
417+ } ,
418+ ] ,
419+ minScore : 0.7 ,
420+ } ,
376421 /** A live model decides its own retry count; only the grounded answer is asserted. */
377- liveExpect : { toolCallSequence : undefined , requiredTools : [ 'flaky_api' ] } ,
422+ liveExpect : {
423+ toolCallSequence : undefined ,
424+ requiredTools : [ 'flaky_api' ] ,
425+ finalContent : undefined ,
426+ } ,
378427 } ,
379428 {
380429 id : 'recovers-from-unknown-tool' ,
@@ -621,7 +670,14 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
621670 finalContent : / E u r o p e \/ B e r l i n / ,
622671 maxIterations : 3 ,
623672 } ,
673+ judge : {
674+ criteria : [
675+ { id : 'grounding' , description : 'states the timezone returned by the settings tool' } ,
676+ { id : 'completeness' , description : 'answers the user request' } ,
677+ ] ,
678+ minScore : 0.7 ,
679+ } ,
624680 /** Over-calling the profile tool is inefficiency, not wrong-tool selection. */
625- liveExpect : { forbiddenTools : undefined } ,
681+ liveExpect : { forbiddenTools : undefined , finalContent : undefined } ,
626682 } ,
627683]
0 commit comments