Skip to content

Commit 4590af8

Browse files
committed
feat(evals): add LLM-as-judge scoring
Substring checks measure phrasing, not correctness. judgeAnswer scores an answer against a weighted rubric with a judge model and returns structured scores; runScenario gains an optional judge that adds a judge check. The judge transport is an injectable OpenAI-compatible completion, so a recorded transcript can replay it deterministically. - judge.ts: rubric, prompt, JSON parsing/clamping, verdict - judge.test.ts: parsing/weighting/clamping (key-free) - judge.live.test.ts: grounded answer outscores an invented one (opt-in) - test:evals:judge script; README documents it
1 parent c93426b commit 4590af8

6 files changed

Lines changed: 342 additions & 0 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -75,6 +75,22 @@ example, an exact retry count). The report is at
7575
`test-results/evals/agent-tool-use-live.{json,md}` with pass rates, average
7676
iterations, latency, and the failed check names.
7777

78+
### LLM-as-judge
79+
80+
Substring and regex checks measure phrasing, not correctness. `judge.ts` scores
81+
an answer against a rubric with a judge model and returns structured numbers:
82+
83+
```sh
84+
cd apps/sim
85+
DEEPSEEK_API_KEY=... bun run test:evals:judge
86+
```
87+
88+
`judgeAnswer` takes the judge transport, the user request, the answer, optional
89+
tool evidence, and a rubric of weighted criteria, and returns a verdict. Pass a
90+
`judge` option to `runScenario` to add a `judge` check alongside the
91+
deterministic ones. The judge transport is an ordinary OpenAI-compatible
92+
completion, so a recorded transcript can replay it deterministically in CI.
93+
7894
## Add a case
7995

8096
1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add

‎apps/sim/evals/agent-tool-use/harness.ts‎

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,7 @@ import type { StreamingToolLoopComplete } from '@/providers/streaming-tool-loop-
1515
import { adaptOpenAIChatToolSchema } from '@/providers/tool-schema-adapter'
1616
import type { ProviderToolConfig, TimeSegment } from '@/providers/types'
1717
import type { ToolResponse } from '@/tools/types'
18+
import { type JudgeRubric, judgeAnswer } from './judge'
1819
import type {
1920
AgentToolUseExpectations,
2021
AgentToolUseResult,
@@ -63,6 +64,12 @@ export interface RunScenarioOptions {
6364
model?: string
6465
/** Provider label used in loop diagnostics. */
6566
providerName?: string
67+
/** Optional LLM-as-judge scorer for the answer; adds a `judge` check. */
68+
judge?: {
69+
completion: OpenAICompatCreateCompletion
70+
model: string
71+
rubric: JudgeRubric
72+
}
6673
}
6774

6875
/** Raw call counts as a value the loop never reads; scenarios only assert on it. */
@@ -390,6 +397,26 @@ export async function runScenario(
390397
const finalContent = completed?.content ?? ''
391398
const iterations = completed?.iterations ?? 0
392399
const checks = scoreExpectations(expected, toolCalls, finalContent, iterations, streamError, mode)
400+
401+
if (options.judge) {
402+
try {
403+
const verdict = await judgeAnswer({
404+
completion: options.judge.completion,
405+
model: options.judge.model,
406+
rubric: options.judge.rubric,
407+
userMessage: scenario.userMessage,
408+
answer: finalContent,
409+
evidence: JSON.stringify(invocations),
410+
})
411+
checks.push({
412+
name: 'judge',
413+
passed: verdict.passed,
414+
detail: `weighted ${verdict.weightedScore.toFixed(2)}; ${verdict.rationale}`,
415+
})
416+
} catch (error) {
417+
checks.push({ name: 'judge', passed: false, detail: `judge failed: ${String(error)}` })
418+
}
419+
}
393420
const successful = toolCalls.filter((call) => call.success).length
394421

395422
return {
Lines changed: 56 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,56 @@
1+
import { describe, expect, it } from 'vitest'
2+
import { judgeAnswer } from '@/evals/agent-tool-use/judge'
3+
import { createDeepSeekLiveCompletion } from '@/evals/agent-tool-use/live'
4+
5+
/**
6+
* Validate the judge prompt against a real model. Opt-in:
7+
*
8+
* EVAL_LIVE=1 DEEPSEEK_API_KEY=... \
9+
* bun run --cwd apps/sim test --mode live evals/agent-tool-use/judge.live.test.ts
10+
*
11+
* The grounded answer must outscore the invented one, which proves the rubric
12+
* distinguishes claims the evidence supports from claims it does not.
13+
*/
14+
const LIVE = process.env.EVAL_LIVE === '1' && Boolean(process.env.DEEPSEEK_API_KEY)
15+
const MODEL = process.env.EVAL_MODEL ?? 'deepseek-chat'
16+
const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '120000')
17+
18+
const rubric = {
19+
criteria: [
20+
{ id: 'grounding', description: 'every factual claim is supported by the evidence' },
21+
{ id: 'completeness', description: 'answers the user request' },
22+
],
23+
minScore: 0.5,
24+
}
25+
26+
describe.skipIf(!LIVE)('llm judge (live)', () => {
27+
it(
28+
'scores a grounded answer above an invented one',
29+
async () => {
30+
const completion = createDeepSeekLiveCompletion(MODEL)
31+
const evidence = 'The API rate limit is 100 requests per minute.'
32+
33+
const grounded = await judgeAnswer({
34+
completion,
35+
model: MODEL,
36+
userMessage: 'What is the API rate limit?',
37+
answer: 'The API rate limit is 100 requests per minute.',
38+
evidence,
39+
rubric,
40+
})
41+
const invented = await judgeAnswer({
42+
completion,
43+
model: MODEL,
44+
userMessage: 'What is the API rate limit?',
45+
answer: 'The API rate limit is 10,000 requests per second on every plan.',
46+
evidence,
47+
rubric,
48+
})
49+
50+
expect(grounded.weightedScore).toBeGreaterThan(invented.weightedScore)
51+
expect(grounded.passed).toBe(true)
52+
expect(invented.passed).toBe(false)
53+
},
54+
TIMEOUT_MS
55+
)
56+
})
Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
import type { ChatCompletionChunk } from 'openai/resources/chat/completions'
2+
import { describe, expect, it } from 'vitest'
3+
import { judgeAnswer, parseJudgeVerdict } from '@/evals/agent-tool-use/judge'
4+
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
5+
6+
function completionReturning(text: string): OpenAICompatCreateCompletion {
7+
return async () =>
8+
(async function* () {
9+
yield {
10+
id: 'chunk',
11+
object: 'chat.completion.chunk',
12+
created: 0,
13+
model: 'judge',
14+
choices: [{ index: 0, delta: { content: text }, finish_reason: 'stop', logprobs: null }],
15+
} satisfies ChatCompletionChunk
16+
})()
17+
}
18+
19+
const rubric = {
20+
criteria: [
21+
{ id: 'grounding', description: 'every claim is supported by the evidence' },
22+
{ id: 'completeness', description: 'answers the user request' },
23+
],
24+
minScore: 0.7,
25+
}
26+
27+
describe('judgeAnswer', () => {
28+
it('parses scores and passes above the threshold', async () => {
29+
const verdict = await judgeAnswer({
30+
completion: completionReturning(
31+
'{"scores":{"grounding":1,"completeness":0.8},"rationale":"grounded"}'
32+
),
33+
model: 'judge',
34+
userMessage: 'q',
35+
answer: 'a',
36+
rubric,
37+
})
38+
expect(verdict.scores).toEqual({ grounding: 1, completeness: 0.8 })
39+
expect(verdict.weightedScore).toBeCloseTo(0.9)
40+
expect(verdict.passed).toBe(true)
41+
})
42+
43+
it('fails below the threshold', async () => {
44+
const verdict = await judgeAnswer({
45+
completion: completionReturning(
46+
'{"scores":{"grounding":0.2,"completeness":0.4},"rationale":"weak"}'
47+
),
48+
model: 'judge',
49+
userMessage: 'q',
50+
answer: 'a',
51+
rubric,
52+
})
53+
expect(verdict.passed).toBe(false)
54+
})
55+
})
56+
57+
describe('parseJudgeVerdict', () => {
58+
it('unwraps fenced JSON', () => {
59+
const verdict = parseJudgeVerdict(
60+
'```json\n{"scores":{"grounding":0.5,"completeness":0.5},"rationale":"ok"}\n```',
61+
rubric
62+
)
63+
expect(verdict.weightedScore).toBe(0.5)
64+
expect(verdict.rationale).toBe('ok')
65+
})
66+
67+
it('clamps out-of-range scores', () => {
68+
const verdict = parseJudgeVerdict(
69+
'{"scores":{"grounding":2,"completeness":-1},"rationale":"x"}',
70+
rubric
71+
)
72+
expect(verdict.scores).toEqual({ grounding: 1, completeness: 0 })
73+
})
74+
75+
it('applies criterion weights', () => {
76+
const verdict = parseJudgeVerdict(
77+
'{"scores":{"grounding":1,"completeness":0},"rationale":"x"}',
78+
{
79+
criteria: [
80+
{ id: 'grounding', description: 'g', weight: 3 },
81+
{ id: 'completeness', description: 'c', weight: 1 },
82+
],
83+
minScore: 0.5,
84+
}
85+
)
86+
expect(verdict.weightedScore).toBeCloseTo(0.75)
87+
})
88+
89+
it('rejects a response missing a criterion', () => {
90+
expect(() => parseJudgeVerdict('{"scores":{"grounding":1},"rationale":"x"}', rubric)).toThrow(
91+
'completeness'
92+
)
93+
})
94+
95+
it('rejects a response that is not a JSON object', () => {
96+
expect(() => parseJudgeVerdict('I think it is fine.', rubric)).toThrow('JSON object')
97+
})
98+
})
Lines changed: 144 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,144 @@
1+
import type { ChatCompletionChunk } from 'openai/resources/chat/completions'
2+
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
3+
4+
/**
5+
* LLM-as-judge scoring for open-ended answers.
6+
*
7+
* Substring and regex checks measure phrasing, not correctness — they kept
8+
* failing on valid paraphrases. A judge model scores an answer against a rubric
9+
* (grounding, completeness, …) and returns structured numbers, so the eval can
10+
* assert behavior instead of wording. The judge transport is injectable, so a
11+
* recorded transcript can replay it deterministically in CI.
12+
*/
13+
14+
/** One rubric criterion the judge scores from 0 to 1. */
15+
export interface JudgeCriterion {
16+
id: string
17+
description: string
18+
/** Relative weight in the weighted score. Default 1. */
19+
weight?: number
20+
}
21+
22+
export interface JudgeRubric {
23+
criteria: JudgeCriterion[]
24+
/** Weighted score at or above this passes. Default 0.5. */
25+
minScore?: number
26+
}
27+
28+
export interface JudgeInput {
29+
completion: OpenAICompatCreateCompletion
30+
model: string
31+
userMessage: string
32+
answer: string
33+
rubric: JudgeRubric
34+
/** Tool results or other material the answer should be grounded in. */
35+
evidence?: string
36+
}
37+
38+
export interface JudgeVerdict {
39+
scores: Record<string, number>
40+
rationale: string
41+
weightedScore: number
42+
passed: boolean
43+
}
44+
45+
const JUDGE_SYSTEM_PROMPT =
46+
'You are a strict, literal evaluator of assistant answers. Score each criterion independently. ' +
47+
'Do not reward fluency or confidence; reward only what the answer actually establishes. ' +
48+
'Return ONLY a JSON object of the form {"scores":{"<criterion>":<number 0..1>},"rationale":"<one sentence>"} ' +
49+
'with no markdown fences and no extra text.'
50+
51+
function buildJudgePrompt(input: JudgeInput): string {
52+
const criteria = input.rubric.criteria
53+
.map((criterion) => `- ${criterion.id}: ${criterion.description}`)
54+
.join('\n')
55+
return [
56+
'User request:',
57+
input.userMessage,
58+
'',
59+
'Assistant answer:',
60+
input.answer || '(empty)',
61+
'',
62+
'Evidence available to the assistant (tool results):',
63+
input.evidence?.trim() || '(none)',
64+
'',
65+
'Criteria (score each from 0 to 1):',
66+
criteria,
67+
].join('\n')
68+
}
69+
70+
async function collectContent(iterable: AsyncIterable<ChatCompletionChunk>): Promise<string> {
71+
let content = ''
72+
for await (const chunk of iterable) {
73+
const delta = chunk.choices?.[0]?.delta?.content
74+
if (typeof delta === 'string') content += delta
75+
}
76+
return content
77+
}
78+
79+
/** Strips markdown fences and returns the outermost JSON object as a string. */
80+
function extractJsonObject(raw: string): string {
81+
const trimmed = raw
82+
.trim()
83+
.replace(/^```(?:json)?/i, '')
84+
.replace(/```$/, '')
85+
.trim()
86+
const start = trimmed.indexOf('{')
87+
const end = trimmed.lastIndexOf('}')
88+
if (start === -1 || end === -1 || end < start) {
89+
throw new Error('Judge did not return a JSON object')
90+
}
91+
return trimmed.slice(start, end + 1)
92+
}
93+
94+
function clampScore(value: unknown): number | undefined {
95+
if (typeof value !== 'number' || !Number.isFinite(value)) return undefined
96+
return Math.min(1, Math.max(0, value))
97+
}
98+
99+
/** Parses and validates a judge response against the rubric. */
100+
export function parseJudgeVerdict(raw: string, rubric: JudgeRubric): JudgeVerdict {
101+
const parsed = JSON.parse(extractJsonObject(raw)) as {
102+
scores?: Record<string, unknown>
103+
rationale?: unknown
104+
}
105+
const rawScores = parsed.scores
106+
if (!rawScores || typeof rawScores !== 'object') {
107+
throw new Error('Judge response is missing "scores"')
108+
}
109+
110+
const scores: Record<string, number> = {}
111+
let weighted = 0
112+
let totalWeight = 0
113+
for (const criterion of rubric.criteria) {
114+
const score = clampScore(rawScores[criterion.id])
115+
if (score === undefined) {
116+
throw new Error(`Judge response is missing a score for "${criterion.id}"`)
117+
}
118+
const weight = criterion.weight ?? 1
119+
scores[criterion.id] = score
120+
weighted += score * weight
121+
totalWeight += weight
122+
}
123+
124+
const weightedScore = totalWeight === 0 ? 0 : weighted / totalWeight
125+
return {
126+
scores,
127+
rationale: typeof parsed.rationale === 'string' ? parsed.rationale : '',
128+
weightedScore,
129+
passed: weightedScore >= (rubric.minScore ?? 0.5),
130+
}
131+
}
132+
133+
/** Runs the judge model and returns a validated verdict. */
134+
export async function judgeAnswer(input: JudgeInput): Promise<JudgeVerdict> {
135+
const iterable = await input.completion({
136+
model: input.model,
137+
stream: true,
138+
messages: [
139+
{ role: 'system', content: JUDGE_SYSTEM_PROMPT },
140+
{ role: 'user', content: buildJudgePrompt(input) },
141+
],
142+
})
143+
return parseJudgeVerdict(await collectContent(iterable), input.rubric)
144+
}

‎apps/sim/package.json‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -26,6 +26,7 @@
2626
"test:coverage": "vitest run --coverage",
2727
"test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use",
2828
"test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use",
29+
"test:evals:judge": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use/judge.live.test.ts",
2930
"email:dev": "email dev --dir components/emails",
3031
"type-check": "tsc --noEmit",
3132
"lint": "biome check --write --unsafe .",

0 commit comments

Comments
 (0)