Skip to content

Commit 74faa92

Browse files
committed
feat(evals): record judge identity and refuse cross-identity comparisons
A judge score is only comparable when the evaluator is the same. Every verdict now carries { model, rubricDigest, parserVersion, temperature }, rubricDigest canonicalizes the rubric, and compareJudgeIdentities reports which fields differ so a delta across a changed evaluator is insufficient evidence, not improvement. - judge.ts: JudgeIdentity/JudgeScore/JudgeVerdict split, rubricDigest, compare - judge.test.ts: digest stability/change + comparison guard + identity in verdict - harness.ts: the judge check shows the judge model and rubric digest
1 parent 3d8de59 commit 74faa92

3 files changed

Lines changed: 118 additions & 5 deletions

File tree

‎apps/sim/evals/agent-tool-use/harness.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -494,7 +494,7 @@ export async function runScenario(
494494
checks.push({
495495
name: 'judge',
496496
passed: verdict.passed,
497-
detail: `weighted ${verdict.weightedScore.toFixed(2)}; ${verdict.rationale}`,
497+
detail: `weighted ${verdict.weightedScore.toFixed(2)} [judge ${verdict.identity.model} ${verdict.identity.rubricDigest}]; ${verdict.rationale}`,
498498
})
499499
} catch (error) {
500500
checks.push({ name: 'judge', passed: false, detail: `judge failed: ${String(error)}` })

‎apps/sim/evals/agent-tool-use/judge.test.ts‎

Lines changed: 59 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,11 @@
11
import type { ChatCompletionChunk } from 'openai/resources/chat/completions'
22
import { describe, expect, it } from 'vitest'
3-
import { judgeAnswer, parseJudgeVerdict } from '@/evals/agent-tool-use/judge'
3+
import {
4+
compareJudgeIdentities,
5+
judgeAnswer,
6+
parseJudgeVerdict,
7+
rubricDigest,
8+
} from '@/evals/agent-tool-use/judge'
49
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
510

611
function completionReturning(text: string): OpenAICompatCreateCompletion {
@@ -96,3 +101,56 @@ describe('parseJudgeVerdict', () => {
96101
expect(() => parseJudgeVerdict('I think it is fine.', rubric)).toThrow('JSON object')
97102
})
98103
})
104+
105+
describe('judge identity', () => {
106+
it('records the model, parser version, and a digest of the rubric', async () => {
107+
const verdict = await judgeAnswer({
108+
completion: completionReturning(
109+
'{"scores":{"grounding":1,"completeness":1},"rationale":"good"}'
110+
),
111+
model: 'deepseek-chat',
112+
userMessage: 'q',
113+
answer: 'a',
114+
rubric,
115+
})
116+
expect(verdict.identity).toMatchObject({ model: 'deepseek-chat', parserVersion: '1' })
117+
expect(verdict.identity.rubricDigest).toBe(rubricDigest(rubric))
118+
})
119+
120+
it('digests the rubric independently of key order', () => {
121+
const reordered = {
122+
minScore: 0.7,
123+
criteria: [
124+
{ description: 'every claim is supported by the evidence', id: 'grounding' },
125+
{ description: 'answers the user request', id: 'completeness' },
126+
],
127+
}
128+
expect(rubricDigest(reordered)).toBe(rubricDigest(rubric))
129+
})
130+
131+
it('changes the digest when a criterion changes', () => {
132+
const changed = {
133+
criteria: [
134+
{ id: 'grounding', description: 'changed' },
135+
{ id: 'completeness', description: 'answers the user request' },
136+
],
137+
}
138+
expect(rubricDigest(changed)).not.toBe(rubricDigest(rubric))
139+
})
140+
141+
it('refuses comparison when a material identity field differs', () => {
142+
const base = { model: 'deepseek-chat', rubricDigest: 'sha256:abc', parserVersion: '1' }
143+
expect(compareJudgeIdentities(base, { ...base })).toEqual({
144+
comparable: true,
145+
differingFields: [],
146+
})
147+
expect(compareJudgeIdentities(base, { ...base, model: 'deepseek-reasoner' })).toEqual({
148+
comparable: false,
149+
differingFields: ['model'],
150+
})
151+
expect(compareJudgeIdentities(base, { ...base, parserVersion: '2' })).toEqual({
152+
comparable: false,
153+
differingFields: ['parserVersion'],
154+
})
155+
})
156+
})

‎apps/sim/evals/agent-tool-use/judge.ts‎

Lines changed: 58 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
1+
import { createHash } from 'node:crypto'
12
import type { ChatCompletionChunk } from 'openai/resources/chat/completions'
23
import type { JudgeCriterion, JudgeRubric } from '@/evals/agent-tool-use/types'
34
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
@@ -24,13 +25,58 @@ export interface JudgeInput {
2425
evidence?: string
2526
}
2627

27-
export interface JudgeVerdict {
28+
/** Parser/schema revision, bumped when the response contract changes. */
29+
export const JUDGE_PARSER_VERSION = '1'
30+
31+
/**
32+
* Who judged and how. Two scores are only comparable when this matches; a delta
33+
* across differing identity is not evidence of improvement.
34+
*/
35+
export interface JudgeIdentity {
36+
model: string
37+
rubricDigest: string
38+
parserVersion: string
39+
temperature?: number
40+
}
41+
42+
export interface JudgeScore {
2843
scores: Record<string, number>
2944
rationale: string
3045
weightedScore: number
3146
passed: boolean
3247
}
3348

49+
export interface JudgeVerdict extends JudgeScore {
50+
identity: JudgeIdentity
51+
}
52+
53+
/** Stable digest of the rubric's grading contract, independent of key order. */
54+
export function rubricDigest(rubric: JudgeRubric): string {
55+
const canonical = JSON.stringify(
56+
rubric.criteria.map((criterion) => ({
57+
id: criterion.id,
58+
description: criterion.description,
59+
weight: criterion.weight ?? 1,
60+
}))
61+
)
62+
return `sha256:${createHash('sha256').update(canonical).digest('hex').slice(0, 16)}`
63+
}
64+
65+
/** Whether two verdicts may be compared, and which identity fields differ. */
66+
export function compareJudgeIdentities(
67+
baseline: JudgeIdentity,
68+
candidate: JudgeIdentity
69+
): { comparable: boolean; differingFields: string[] } {
70+
const fields: Array<keyof JudgeIdentity> = [
71+
'model',
72+
'rubricDigest',
73+
'parserVersion',
74+
'temperature',
75+
]
76+
const differingFields = fields.filter((field) => baseline[field] !== candidate[field])
77+
return { comparable: differingFields.length === 0, differingFields }
78+
}
79+
3480
const JUDGE_SYSTEM_PROMPT =
3581
'You are a strict, literal evaluator of assistant answers. Score each criterion independently. ' +
3682
'Do not reward fluency or confidence; reward only what the answer actually establishes. ' +
@@ -86,7 +132,7 @@ function clampScore(value: unknown): number | undefined {
86132
}
87133

88134
/** Parses and validates a judge response against the rubric. */
89-
export function parseJudgeVerdict(raw: string, rubric: JudgeRubric): JudgeVerdict {
135+
export function parseJudgeVerdict(raw: string, rubric: JudgeRubric): JudgeScore {
90136
const parsed = JSON.parse(extractJsonObject(raw)) as {
91137
scores?: Record<string, unknown>
92138
rationale?: unknown
@@ -129,5 +175,14 @@ export async function judgeAnswer(input: JudgeInput): Promise<JudgeVerdict> {
129175
{ role: 'user', content: buildJudgePrompt(input) },
130176
],
131177
})
132-
return parseJudgeVerdict(await collectContent(iterable), input.rubric)
178+
const score = parseJudgeVerdict(await collectContent(iterable), input.rubric)
179+
return {
180+
...score,
181+
identity: {
182+
model: input.model,
183+
rubricDigest: rubricDigest(input.rubric),
184+
parserVersion: JUDGE_PARSER_VERSION,
185+
...(input.temperature !== undefined ? { temperature: input.temperature } : {}),
186+
},
187+
}
133188
}

0 commit comments

Comments
 (0)