Skip to content

Commit ba2faee

Browse files
committed
Grade benchmark mechanics with verified reference resolution
1 parent 9238bf2 commit ba2faee

17 files changed

Lines changed: 377 additions & 36 deletions

File tree

‎apps/sim/app/o/[organizationId]/benchmark/components/benchmark-detail.tsx‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -143,7 +143,7 @@ function BenchmarkEditor({
143143
<BenchmarkStep
144144
number={3}
145145
title='Fill in the blanks'
146-
description='A fresh reader fills the blanks by reading or searching the generated plan. It has no enterprise access or prior memory.'
146+
description='A fresh reader fills the blanks from the generated plan and may look up references already named there. Lookups cannot supply missing workflow rules. The reader never sees expected answers or the completed workspace.'
147147
pending={stage === 'reconstruct'}
148148
action={
149149
<Chip

‎apps/sim/app/o/[organizationId]/benchmark/components/benchmark-results.tsx‎

Lines changed: 36 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -43,6 +43,14 @@ export function BenchmarkResults({
4343
{review ? ' · Human reviewed' : ''}
4444
</span>
4545
)}
46+
{!result && answer?.evidenceError && (
47+
<div>
48+
<dt className='text-[var(--text-muted)]'>Evidence needs review</dt>
49+
<dd className='mt-1 whitespace-pre-wrap break-words text-[var(--text-body)]'>
50+
{answer.evidenceError}
51+
</dd>
52+
</div>
53+
)}
4654
</div>
4755
<dl className='flex flex-col gap-3 text-small'>
4856
{showGrade && (
@@ -54,7 +62,7 @@ export function BenchmarkResults({
5462
</div>
5563
)}
5664
<div>
57-
<dt className='text-[var(--text-muted)]'>Answer from the generated spec</dt>
65+
<dt className='text-[var(--text-muted)]'>Recovered answer</dt>
5866
<dd className='mt-1 whitespace-pre-wrap break-words text-[var(--text-body)]'>
5967
{answer?.answer || 'No answer returned'}
6068
</dd>
@@ -67,10 +75,37 @@ export function BenchmarkResults({
6775
{answer?.support || 'No supporting passage'}
6876
</dd>
6977
</div>
78+
{answer?.sources?.map((source, index) => (
79+
<div key={`${source.citationId}-${index}`}>
80+
<dt className='text-[var(--text-muted)]'>
81+
Resolved reference ·{' '}
82+
{source.url ? (
83+
<a
84+
href={source.url}
85+
target='_blank'
86+
rel='noopener noreferrer'
87+
className='underline'
88+
>
89+
{source.title || source.citationId}
90+
</a>
91+
) : (
92+
source.title || source.citationId
93+
)}
94+
</dt>
95+
<dd className='mt-1 whitespace-pre-wrap break-words text-[var(--text-body)]'>
96+
{source.quote}
97+
</dd>
98+
</div>
99+
))}
70100
{result && (
71101
<div>
72102
<dt className='text-[var(--text-muted)]'>
73103
AI assessment
104+
{result.basis === 'missing'
105+
? ' · Missing from plan'
106+
: result.basis === 'reference'
107+
? ' · Reference resolved'
108+
: ''}
74109
{review ? ` · ${result.correct ? 'Recovered' : 'Not recovered'}` : ''}
75110
</dt>
76111
<dd className='mt-1 whitespace-pre-wrap break-words text-[var(--text-body)]'>

‎apps/sim/app/o/[organizationId]/benchmark/components/benchmark-run-comparison.tsx‎

Lines changed: 6 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -30,6 +30,11 @@ function RunSnapshot({
3030
<p className='mt-1 text-[var(--text-muted)] text-small tabular-nums'>
3131
{Math.round((run.correct / run.total) * 100)}% · {run.correct} of {run.total} recovered
3232
</p>
33+
<p className='mt-1 text-[var(--text-muted)] text-small'>
34+
{run.artifacts.recoveryMode === 'references'
35+
? 'Spec with reference resolution'
36+
: 'Spec only'}
37+
</p>
3338
{run.reviewedCount > 0 && (
3439
<p className='mt-1 text-[var(--text-muted)] text-small'>
3540
{run.reviewedCount} human overrides · AI score{' '}
@@ -108,7 +113,7 @@ export function BenchmarkRunComparison({
108113
<p role='status' className='text-[var(--text-body)] text-small'>
109114
{comparable
110115
? `${delta > 0 ? '+' : ''}${Number(delta.toFixed(1))} percentage points vs baseline · ${improved} details improved · ${regressed} regressed`
111-
: 'These runs used different task briefs or reference details. Their scores are not directly comparable.'}
116+
: 'These runs used different inputs or recovery methods. Their scores are not directly comparable.'}
112117
</p>
113118
)}
114119
{(run.reviewedCount > 0 || (baseline?.reviewedCount ?? 0) > 0) && (

‎apps/sim/lib/benchmarks/application/run-stage.ts‎

Lines changed: 27 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -16,6 +16,7 @@ import {
1616
} from '@/lib/benchmarks/artifacts'
1717
import { getBenchmarkMothershipUrl } from '@/lib/benchmarks/config'
1818
import { gradeReconstruction, validateReconstruction } from '@/lib/benchmarks/evaluation'
19+
import { verifyRecoveryEvidence } from '@/lib/benchmarks/evidence'
1920
import {
2021
distillationMessages,
2122
gradingMessages,
@@ -35,6 +36,7 @@ import {
3536
benchmarkBriefSchema,
3637
benchmarkGradeSchema,
3738
benchmarkReconstructionSchema,
39+
benchmarkSourceSchema,
3840
benchmarkSpecSchema,
3941
} from '@/lib/benchmarks/types'
4042
import { executeBenchmarkJson, executeBenchmarkPlan } from '@/lib/benchmarks/worker'
@@ -58,9 +60,19 @@ const redactionSchema = z
5860
})
5961
.strict()
6062
const reconstructionSchema = z
61-
.object({ answers: z.array(benchmarkReconstructionSchema).min(1) })
63+
.object({
64+
answers: z
65+
.array(
66+
benchmarkReconstructionSchema.pick({ id: true, answer: true, support: true }).extend({
67+
sources: z.array(benchmarkSourceSchema.pick({ citationId: true, quote: true })),
68+
})
69+
)
70+
.min(1),
71+
})
72+
.strict()
73+
const gradingSchema = z
74+
.object({ judgments: z.array(benchmarkGradeSchema.required({ basis: true })).min(1) })
6275
.strict()
63-
const gradingSchema = z.object({ judgments: z.array(benchmarkGradeSchema).min(1) }).strict()
6476

6577
interface RunBenchmarkStageInput {
6678
organizationId: string
@@ -97,7 +109,7 @@ async function performStage(
97109
const artifacts = benchmark.artifacts
98110
switch (stage) {
99111
case 'distill': {
100-
const result = await executeBenchmarkJson({
112+
const { data: result } = await executeBenchmarkJson({
101113
principal,
102114
benchmark,
103115
signal,
@@ -108,7 +120,7 @@ async function performStage(
108120
return { artifacts: applyBenchmarkPatch(artifacts, result), plannerChatId: null }
109121
}
110122
case 'redact': {
111-
const result = await executeBenchmarkJson({
123+
const { data: result } = await executeBenchmarkJson({
112124
principal,
113125
benchmark,
114126
signal,
@@ -135,19 +147,26 @@ async function performStage(
135147
}
136148
}
137149
case 'reconstruct': {
138-
const result = await executeBenchmarkJson({
150+
const { data: result, toolCalls } = await executeBenchmarkJson({
139151
principal,
140152
benchmark,
141153
signal,
142154
schema: reconstructionSchema,
143155
messages: reconstructionMessages(artifacts.redactedSpec),
144-
profile: { stage: 'reconstruct', spec: artifacts.generatedSpec! },
156+
profile: { stage: 'resolve', spec: artifacts.generatedSpec! },
145157
})
146158
validateReconstruction(artifacts.blanks, result.answers)
147-
return { artifacts: { ...artifacts, reconstruction: result.answers, grade: null } }
159+
const reconstruction = verifyRecoveryEvidence(
160+
artifacts.generatedSpec!,
161+
result.answers,
162+
toolCalls
163+
)
164+
return {
165+
artifacts: { ...artifacts, recoveryMode: 'references', reconstruction, grade: null },
166+
}
148167
}
149168
case 'grade': {
150-
const result = await executeBenchmarkJson({
169+
const { data: result } = await executeBenchmarkJson({
151170
principal,
152171
benchmark,
153172
signal,

‎apps/sim/lib/benchmarks/evaluation.test.ts‎

Lines changed: 38 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -43,6 +43,19 @@ describe('benchmark reconstruction grading', () => {
4343
expect(result.every(({ correct }) => !correct)).toBe(true)
4444
})
4545

46+
it('keeps an answer available for review without crediting unverified source evidence', () => {
47+
const result = gradeReconstruction({
48+
blanks,
49+
reconstruction: reconstructed.map((answer) => ({
50+
...answer,
51+
evidenceError: 'Source quote was not retrieved.',
52+
})),
53+
generatedSpec: 'Support owns the escalation.',
54+
judgments: blanks.map(({ id }) => ({ id, correct: true, reason: 'Equivalent answer.' })),
55+
})
56+
expect(result[0]).toMatchObject({ correct: false, reason: 'Source quote was not retrieved.' })
57+
})
58+
4659
it('rejects an incomplete judge response instead of silently passing ungraded answers', () => {
4760
expect(() =>
4861
gradeReconstruction({
@@ -53,4 +66,29 @@ describe('benchmark reconstruction grading', () => {
5366
})
5467
).toThrow()
5568
})
69+
70+
it('rejects a missing mechanic even when an external source supplies the expected answer', () => {
71+
const spec = 'Use the Sim repository for issue intake.'
72+
const result = gradeReconstruction({
73+
blanks: [{ id: 'handoff', answer: 'Engineering accepts the case' }],
74+
reconstruction: [
75+
{
76+
id: 'handoff',
77+
answer: 'Engineering accepts the case',
78+
support: spec,
79+
sources: [{ citationId: 'repo', quote: 'Engineering accepts the case' }],
80+
},
81+
],
82+
generatedSpec: spec,
83+
judgments: [
84+
{
85+
id: 'handoff',
86+
correct: true,
87+
basis: 'missing',
88+
reason: 'The repository document describes the handoff, but the plan omits it.',
89+
},
90+
],
91+
})
92+
expect(result[0].correct).toBe(false)
93+
})
5694
})

‎apps/sim/lib/benchmarks/evaluation.ts‎

Lines changed: 10 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -25,7 +25,7 @@ export function validateReconstruction(
2525
requireMatchingIds(blanks, values)
2626
}
2727

28-
/** Correct guesses earn credit only when the reader quotes the submitted plan as evidence. */
28+
/** Source lookups may resolve references, but cannot supply mechanics absent from the plan. */
2929
export function gradeReconstruction(input: {
3030
blanks: BenchmarkArtifacts['blanks']
3131
reconstruction: Reconstruction
@@ -39,6 +39,7 @@ export function gradeReconstruction(input: {
3939
return input.blanks.map(({ id }) => {
4040
const answer = answers.get(id)!
4141
const judgment = judgments.get(id)!
42+
if (answer.evidenceError) return { ...judgment, correct: false, reason: answer.evidenceError }
4243
if (
4344
!answer.answer.trim() ||
4445
!answer.support.trim() ||
@@ -50,6 +51,14 @@ export function gradeReconstruction(input: {
5051
reason: 'No exact supporting passage was provided from the generated spec.',
5152
}
5253
}
54+
if (judgment.basis === 'missing') return { ...judgment, correct: false }
55+
if (judgment.basis === 'reference' && !answer.sources?.length) {
56+
return {
57+
...judgment,
58+
correct: false,
59+
reason: 'No verified source evidence was provided for this reference.',
60+
}
61+
}
5362
return judgment
5463
})
5564
}
Lines changed: 95 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,95 @@
1+
import { describe, expect, it } from 'vitest'
2+
import { verifyRecoveryEvidence } from '@/lib/benchmarks/evidence'
3+
import type { ToolCallSummary } from '@/lib/mothership/request/types'
4+
5+
describe('benchmark source provenance', () => {
6+
const support = 'Use the Sim repository for issue intake.'
7+
const answer = {
8+
id: 'repository',
9+
answer: 'simstudioai/sim',
10+
support,
11+
sources: [{ citationId: 'repo', quote: 'simstudioai/sim' }],
12+
}
13+
const call: ToolCallSummary = {
14+
id: 'lookup',
15+
name: 'search_workspace',
16+
status: 'success',
17+
params: { specPassage: support, query: 'Sim repository' },
18+
result: {
19+
success: true,
20+
data: {
21+
results: [
22+
{
23+
citationId: 'repo',
24+
content: 'Repository simstudioai/sim',
25+
citationUrl: 'https://github.com/simstudioai/sim',
26+
documentName: 'Sim',
27+
},
28+
],
29+
},
30+
},
31+
}
32+
33+
it('attaches the actual source URL only to quotations retrieved for the same spec passage', () => {
34+
const [verified] = verifyRecoveryEvidence(support, [answer], [call])
35+
expect(verified.sources).toEqual([
36+
{ ...answer.sources[0], url: 'https://github.com/simstudioai/sim', title: 'Sim' },
37+
])
38+
})
39+
40+
it.each([
41+
{ ...call, status: 'error' as const },
42+
{ ...call, name: 'sim_cli' },
43+
{
44+
...call,
45+
params: { specPassage: 'Use the Mothership repository.', query: 'Mothership repository' },
46+
},
47+
{
48+
...call,
49+
result: {
50+
success: true,
51+
data: { results: [{ citationId: 'different', content: 'simstudioai/sim' }] },
52+
},
53+
},
54+
{
55+
...call,
56+
result: {
57+
success: true,
58+
data: { results: [{ citationId: 'repo', content: 'simstudioai/mothership' }] },
59+
},
60+
},
61+
])(
62+
'preserves the answer for review but rejects failed, unrelated, or fabricated evidence',
63+
(toolCall) => {
64+
const [verified] = verifyRecoveryEvidence(support, [answer], [toolCall])
65+
expect(verified.answer).toBe(answer.answer)
66+
expect(verified.sources).toEqual([])
67+
expect(verified.evidenceError).toContain('repo')
68+
}
69+
)
70+
71+
it('does not turn an unsafe retrieved URL into a clickable source', () => {
72+
const [verified] = verifyRecoveryEvidence(
73+
support,
74+
[answer],
75+
[
76+
{
77+
...call,
78+
result: {
79+
success: true,
80+
data: {
81+
results: [
82+
{
83+
citationId: 'repo',
84+
content: 'simstudioai/sim',
85+
citationUrl: 'javascript:alert(1)',
86+
},
87+
],
88+
},
89+
},
90+
},
91+
]
92+
)
93+
expect(verified.sources?.[0].url).toBeUndefined()
94+
})
95+
})

0 commit comments

Comments
 (0)