|
| 1 | +import type { ChatCompletionChunk } from 'openai/resources/chat/completions' |
| 2 | +import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop' |
| 3 | + |
| 4 | +/** |
| 5 | + * LLM-as-judge scoring for open-ended answers. |
| 6 | + * |
| 7 | + * Substring and regex checks measure phrasing, not correctness — they kept |
| 8 | + * failing on valid paraphrases. A judge model scores an answer against a rubric |
| 9 | + * (grounding, completeness, …) and returns structured numbers, so the eval can |
| 10 | + * assert behavior instead of wording. The judge transport is injectable, so a |
| 11 | + * recorded transcript can replay it deterministically in CI. |
| 12 | + */ |
| 13 | + |
| 14 | +/** One rubric criterion the judge scores from 0 to 1. */ |
| 15 | +export interface JudgeCriterion { |
| 16 | + id: string |
| 17 | + description: string |
| 18 | + /** Relative weight in the weighted score. Default 1. */ |
| 19 | + weight?: number |
| 20 | +} |
| 21 | + |
| 22 | +export interface JudgeRubric { |
| 23 | + criteria: JudgeCriterion[] |
| 24 | + /** Weighted score at or above this passes. Default 0.5. */ |
| 25 | + minScore?: number |
| 26 | +} |
| 27 | + |
| 28 | +export interface JudgeInput { |
| 29 | + completion: OpenAICompatCreateCompletion |
| 30 | + model: string |
| 31 | + userMessage: string |
| 32 | + answer: string |
| 33 | + rubric: JudgeRubric |
| 34 | + /** Tool results or other material the answer should be grounded in. */ |
| 35 | + evidence?: string |
| 36 | +} |
| 37 | + |
| 38 | +export interface JudgeVerdict { |
| 39 | + scores: Record<string, number> |
| 40 | + rationale: string |
| 41 | + weightedScore: number |
| 42 | + passed: boolean |
| 43 | +} |
| 44 | + |
| 45 | +const JUDGE_SYSTEM_PROMPT = |
| 46 | + 'You are a strict, literal evaluator of assistant answers. Score each criterion independently. ' + |
| 47 | + 'Do not reward fluency or confidence; reward only what the answer actually establishes. ' + |
| 48 | + 'Return ONLY a JSON object of the form {"scores":{"<criterion>":<number 0..1>},"rationale":"<one sentence>"} ' + |
| 49 | + 'with no markdown fences and no extra text.' |
| 50 | + |
| 51 | +function buildJudgePrompt(input: JudgeInput): string { |
| 52 | + const criteria = input.rubric.criteria |
| 53 | + .map((criterion) => `- ${criterion.id}: ${criterion.description}`) |
| 54 | + .join('\n') |
| 55 | + return [ |
| 56 | + 'User request:', |
| 57 | + input.userMessage, |
| 58 | + '', |
| 59 | + 'Assistant answer:', |
| 60 | + input.answer || '(empty)', |
| 61 | + '', |
| 62 | + 'Evidence available to the assistant (tool results):', |
| 63 | + input.evidence?.trim() || '(none)', |
| 64 | + '', |
| 65 | + 'Criteria (score each from 0 to 1):', |
| 66 | + criteria, |
| 67 | + ].join('\n') |
| 68 | +} |
| 69 | + |
| 70 | +async function collectContent(iterable: AsyncIterable<ChatCompletionChunk>): Promise<string> { |
| 71 | + let content = '' |
| 72 | + for await (const chunk of iterable) { |
| 73 | + const delta = chunk.choices?.[0]?.delta?.content |
| 74 | + if (typeof delta === 'string') content += delta |
| 75 | + } |
| 76 | + return content |
| 77 | +} |
| 78 | + |
| 79 | +/** Strips markdown fences and returns the outermost JSON object as a string. */ |
| 80 | +function extractJsonObject(raw: string): string { |
| 81 | + const trimmed = raw |
| 82 | + .trim() |
| 83 | + .replace(/^```(?:json)?/i, '') |
| 84 | + .replace(/```$/, '') |
| 85 | + .trim() |
| 86 | + const start = trimmed.indexOf('{') |
| 87 | + const end = trimmed.lastIndexOf('}') |
| 88 | + if (start === -1 || end === -1 || end < start) { |
| 89 | + throw new Error('Judge did not return a JSON object') |
| 90 | + } |
| 91 | + return trimmed.slice(start, end + 1) |
| 92 | +} |
| 93 | + |
| 94 | +function clampScore(value: unknown): number | undefined { |
| 95 | + if (typeof value !== 'number' || !Number.isFinite(value)) return undefined |
| 96 | + return Math.min(1, Math.max(0, value)) |
| 97 | +} |
| 98 | + |
| 99 | +/** Parses and validates a judge response against the rubric. */ |
| 100 | +export function parseJudgeVerdict(raw: string, rubric: JudgeRubric): JudgeVerdict { |
| 101 | + const parsed = JSON.parse(extractJsonObject(raw)) as { |
| 102 | + scores?: Record<string, unknown> |
| 103 | + rationale?: unknown |
| 104 | + } |
| 105 | + const rawScores = parsed.scores |
| 106 | + if (!rawScores || typeof rawScores !== 'object') { |
| 107 | + throw new Error('Judge response is missing "scores"') |
| 108 | + } |
| 109 | + |
| 110 | + const scores: Record<string, number> = {} |
| 111 | + let weighted = 0 |
| 112 | + let totalWeight = 0 |
| 113 | + for (const criterion of rubric.criteria) { |
| 114 | + const score = clampScore(rawScores[criterion.id]) |
| 115 | + if (score === undefined) { |
| 116 | + throw new Error(`Judge response is missing a score for "${criterion.id}"`) |
| 117 | + } |
| 118 | + const weight = criterion.weight ?? 1 |
| 119 | + scores[criterion.id] = score |
| 120 | + weighted += score * weight |
| 121 | + totalWeight += weight |
| 122 | + } |
| 123 | + |
| 124 | + const weightedScore = totalWeight === 0 ? 0 : weighted / totalWeight |
| 125 | + return { |
| 126 | + scores, |
| 127 | + rationale: typeof parsed.rationale === 'string' ? parsed.rationale : '', |
| 128 | + weightedScore, |
| 129 | + passed: weightedScore >= (rubric.minScore ?? 0.5), |
| 130 | + } |
| 131 | +} |
| 132 | + |
| 133 | +/** Runs the judge model and returns a validated verdict. */ |
| 134 | +export async function judgeAnswer(input: JudgeInput): Promise<JudgeVerdict> { |
| 135 | + const iterable = await input.completion({ |
| 136 | + model: input.model, |
| 137 | + stream: true, |
| 138 | + messages: [ |
| 139 | + { role: 'system', content: JUDGE_SYSTEM_PROMPT }, |
| 140 | + { role: 'user', content: buildJudgePrompt(input) }, |
| 141 | + ], |
| 142 | + }) |
| 143 | + return parseJudgeVerdict(await collectContent(iterable), input.rubric) |
| 144 | +} |
0 commit comments