Skip to content

Commit 3d8de59

Browse files
committed
feat(evals): judge open-ended agent answers
Wire the LLM judge into scenarios. scenarios gain an optional rubric, the live suite runs it (EVAL_JUDGE_MODEL, default the task model), and the deterministic checks stay. The judge scores grounding/completeness on the cases where a substring was never an honest measure: empty-result honesty, retrieved-value grounding, and the long chain. - JudgeCriterion/JudgeRubric move to types.ts so scenario data carries a rubric without pulling the loop graph; judge.ts re-exports them. - Rubrics on uses-retrieved-value, empty-result-no-hallucination, long-chain.
1 parent f619479 commit 3d8de59

4 files changed

Lines changed: 61 additions & 14 deletions

File tree

‎apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -40,6 +40,7 @@ const RECORD = process.env.EVAL_RECORD === '1'
4040
const TRIALS = Number(process.env.EVAL_TRIALS ?? '3')
4141
const MIN_PASS_RATE = Number(process.env.EVAL_MIN_PASS_RATE ?? '0')
4242
const MODEL = process.env.EVAL_MODEL ?? 'deepseek-chat'
43+
const JUDGE_MODEL = process.env.EVAL_JUDGE_MODEL ?? MODEL
4344
const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '180000')
4445
const FIXTURES_DIR = fileURLToPath(new URL('./fixtures', import.meta.url))
4546

@@ -61,6 +62,7 @@ describe.skipIf(!LIVE)('agent tool-use eval suite (live DeepSeek)', () => {
6162
'$id: $name',
6263
async (scenario) => {
6364
const base = createDeepSeekLiveCompletion(MODEL)
65+
const judgeCompletion = scenario.judge ? createDeepSeekLiveCompletion(JUDGE_MODEL) : undefined
6466
const results: AgentToolUseResult[] = []
6567

6668
for (let trial = 0; trial < TRIALS; trial++) {
@@ -76,6 +78,15 @@ describe.skipIf(!LIVE)('agent tool-use eval suite (live DeepSeek)', () => {
7678
mode: 'live',
7779
model: MODEL,
7880
providerName: 'DeepSeek',
81+
...(scenario.judge && judgeCompletion
82+
? {
83+
judge: {
84+
completion: judgeCompletion,
85+
model: JUDGE_MODEL,
86+
rubric: scenario.judge,
87+
},
88+
}
89+
: {}),
7990
})
8091
)
8192

‎apps/sim/evals/agent-tool-use/judge.ts‎

Lines changed: 3 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,9 @@
11
import type { ChatCompletionChunk } from 'openai/resources/chat/completions'
2+
import type { JudgeCriterion, JudgeRubric } from '@/evals/agent-tool-use/types'
23
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
34

5+
export type { JudgeCriterion, JudgeRubric }
6+
47
/**
58
* LLM-as-judge scoring for open-ended answers.
69
*
@@ -11,20 +14,6 @@ import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/str
1114
* recorded transcript can replay it deterministically in CI.
1215
*/
1316

14-
/** One rubric criterion the judge scores from 0 to 1. */
15-
export interface JudgeCriterion {
16-
id: string
17-
description: string
18-
/** Relative weight in the weighted score. Default 1. */
19-
weight?: number
20-
}
21-
22-
export interface JudgeRubric {
23-
criteria: JudgeCriterion[]
24-
/** Weighted score at or above this passes. Default 0.5. */
25-
minScore?: number
26-
}
27-
2817
export interface JudgeInput {
2918
completion: OpenAICompatCreateCompletion
3019
model: string

‎apps/sim/evals/agent-tool-use/scenarios.ts‎

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -289,6 +289,16 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
289289
},
290290
/** A live model may answer with the user-facing order number and the grounded status. */
291291
liveExpect: { finalContent: /shipped/i },
292+
judge: {
293+
criteria: [
294+
{
295+
id: 'grounding',
296+
description: 'every claim about the order matches the retrieved fields',
297+
},
298+
{ id: 'completeness', description: 'tells the user the order status' },
299+
],
300+
minScore: 0.7,
301+
},
292302
},
293303
{
294304
id: 'parallel-independent-tools',
@@ -502,6 +512,16 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
502512
/could ?n'?o?t find|cannot find|can'?t find|didn'?t find|wasn'?t able|not able to find|unable to find|no results|no matching|no documentation|no information|not covered|returned empty|came back empty|found nothing|no .*policy/i,
503513
maxIterations: 3,
504514
},
515+
judge: {
516+
criteria: [
517+
{
518+
id: 'grounding',
519+
description: 'does not assert a refund policy the evidence does not contain',
520+
},
521+
{ id: 'honesty', description: 'makes clear it could not find the policy' },
522+
],
523+
minScore: 0.7,
524+
},
505525
},
506526
{
507527
id: 'long-chain-dependency',
@@ -566,6 +586,13 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [
566586
requiredTools: ['list_orders', 'get_order', 'get_shipping'],
567587
successfulToolCalls: 3,
568588
},
589+
judge: {
590+
criteria: [
591+
{ id: 'grounding', description: 'uses the shipping data returned by the tools' },
592+
{ id: 'completeness', description: 'answers where the order is' },
593+
],
594+
minScore: 0.7,
595+
},
569596
},
570597
{
571598
id: 'near-duplicate-names',

‎apps/sim/evals/agent-tool-use/types.ts‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -68,6 +68,21 @@ export interface AgentToolUseExpectations {
6868
completesWithoutError?: boolean
6969
}
7070

71+
/** One rubric criterion an LLM judge scores from 0 to 1. */
72+
export interface JudgeCriterion {
73+
id: string
74+
description: string
75+
/** Relative weight in the weighted score. Default 1. */
76+
weight?: number
77+
}
78+
79+
/** A rubric the LLM judge scores an answer against. */
80+
export interface JudgeRubric {
81+
criteria: JudgeCriterion[]
82+
/** Weighted score at or above this passes. Default 0.5. */
83+
minScore?: number
84+
}
85+
7186
/** A single agent behavior case. */
7287
export interface AgentToolUseScenario {
7388
id: string
@@ -84,6 +99,11 @@ export interface AgentToolUseScenario {
8499
* meaningful once a real model chooses the calls.
85100
*/
86101
liveExpect?: Partial<AgentToolUseExpectations>
102+
/**
103+
* Optional LLM-as-judge rubric. Deterministic checks still run; the judge adds
104+
* a `judge` check in runs that supply a judge model (the live suite does).
105+
*/
106+
judge?: JudgeRubric
87107
/**
88108
* True when the case only makes sense with a scripted model (e.g. it requires
89109
* the model to emit malformed JSON on demand). Excluded from live runs.

0 commit comments

Comments
 (0)