Skip to content

Commit ffa7e73

Browse files
committed
test(eval): add ContextBench harness core
1 parent 62d3110 commit ffa7e73

19 files changed

Lines changed: 9873 additions & 0 deletions

‎scripts/contextbench-retrieval-gate.mjs‎

Lines changed: 1249 additions & 0 deletions
Large diffs are not rendered by default.

‎scripts/contextbench-runner.mjs‎

Lines changed: 3586 additions & 0 deletions
Large diffs are not rendered by default.

‎src/eval/contextbench-answer.ts‎

Lines changed: 229 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,229 @@
1+
import type {
2+
ContextBenchEvidenceReference,
3+
ContextBenchStructuredAnswer,
4+
JsonSchemaDefinition,
5+
JsonValue
6+
} from './contextbench-types.js';
7+
8+
export interface StructuredAnswerParseResult {
9+
status: 'valid' | 'invalid_schema';
10+
answer: ContextBenchStructuredAnswer | null;
11+
errors: string[];
12+
}
13+
14+
export interface SchemaBoundDiagnostics {
15+
missingRequiredFacts?: string[];
16+
contradictoryFacts?: string[];
17+
missingEvidenceFiles?: string[];
18+
unsupportedEvidenceFiles?: string[];
19+
}
20+
21+
export interface AnswerClassification {
22+
unsupportedClaim: boolean;
23+
falseReady: boolean;
24+
reasons: string[];
25+
}
26+
27+
const confidenceValues = new Set(['low', 'medium', 'high']);
28+
29+
const evidenceReferenceFields = new Set(['file', 'lineRange', 'reason']);
30+
const lineRangeFields = new Set(['start', 'end']);
31+
32+
export const CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS = [
33+
'answer',
34+
'confidence',
35+
'evidence',
36+
'filesReferenced',
37+
'symbolsReferenced',
38+
'unsupportedClaims',
39+
'readyToEdit'
40+
] as const;
41+
42+
export const CONTEXTBENCH_STRUCTURED_ANSWER_JSON_SCHEMA = {
43+
type: 'object',
44+
additionalProperties: false,
45+
required: [...CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS],
46+
properties: {
47+
answer: { type: ['object', 'array', 'string', 'number', 'boolean', 'null'] },
48+
confidence: { type: 'string', enum: ['low', 'medium', 'high'] },
49+
evidence: {
50+
type: 'array',
51+
items: {
52+
type: 'object',
53+
additionalProperties: false,
54+
required: ['file', 'lineRange', 'reason'],
55+
properties: {
56+
file: { type: 'string', minLength: 1 },
57+
lineRange: {
58+
type: 'object',
59+
additionalProperties: false,
60+
required: ['start', 'end'],
61+
properties: {
62+
start: { type: 'integer', minimum: 1 },
63+
end: { type: 'integer', minimum: 1 }
64+
}
65+
},
66+
reason: { type: 'string', minLength: 1 }
67+
}
68+
}
69+
},
70+
filesReferenced: { type: 'array', items: { type: 'string' } },
71+
symbolsReferenced: { type: 'array', items: { type: 'string' } },
72+
unsupportedClaims: { type: 'array', items: { type: 'string' } },
73+
readyToEdit: { type: 'boolean' }
74+
}
75+
} satisfies JsonSchemaDefinition;
76+
77+
function isRecord(value: unknown): value is Record<string, unknown> {
78+
return value !== null && typeof value === 'object' && !Array.isArray(value);
79+
}
80+
81+
function isStringArray(value: unknown): value is string[] {
82+
return Array.isArray(value) && value.every((entry) => typeof entry === 'string');
83+
}
84+
85+
function findAdditionalFields(
86+
value: Record<string, unknown>,
87+
allowedFields: ReadonlySet<string>,
88+
prefix: string
89+
): string[] {
90+
return Object.keys(value)
91+
.filter((field) => !allowedFields.has(field))
92+
.map((field) => `additional_${prefix}_${field}`);
93+
}
94+
95+
function isJsonValue(value: unknown): value is JsonValue {
96+
if (value === null) return true;
97+
if (typeof value === 'string' || typeof value === 'number' || typeof value === 'boolean')
98+
return true;
99+
if (Array.isArray(value)) return value.every(isJsonValue);
100+
if (!isRecord(value)) return false;
101+
return Object.values(value).every(isJsonValue);
102+
}
103+
104+
export function isValidEvidenceReference(value: unknown): value is ContextBenchEvidenceReference {
105+
if (!isRecord(value)) return false;
106+
if (findAdditionalFields(value, evidenceReferenceFields, 'evidence_field').length > 0) return false;
107+
const lineRange = value.lineRange;
108+
if (!isRecord(lineRange)) return false;
109+
if (findAdditionalFields(lineRange, lineRangeFields, 'line_range_field').length > 0)
110+
return false;
111+
const start = lineRange.start;
112+
const end = lineRange.end;
113+
return (
114+
typeof value.file === 'string' &&
115+
value.file.trim().length > 0 &&
116+
typeof value.reason === 'string' &&
117+
value.reason.trim().length > 0 &&
118+
Number.isInteger(start) &&
119+
Number.isInteger(end) &&
120+
typeof start === 'number' &&
121+
typeof end === 'number' &&
122+
start > 0 &&
123+
end >= start
124+
);
125+
}
126+
127+
function validateStructuredAnswer(value: unknown): StructuredAnswerParseResult {
128+
const errors: string[] = [];
129+
if (!isRecord(value)) {
130+
return { status: 'invalid_schema', answer: null, errors: ['answer_root_not_object'] };
131+
}
132+
133+
for (const field of CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS) {
134+
if (!(field in value)) errors.push(`missing_${field}`);
135+
}
136+
errors.push(
137+
...findAdditionalFields(value, new Set(CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS), 'root_field')
138+
);
139+
140+
if (!isJsonValue(value.answer)) errors.push('answer_not_json_value');
141+
if (typeof value.confidence !== 'string' || !confidenceValues.has(value.confidence))
142+
errors.push('invalid_confidence');
143+
if (!Array.isArray(value.evidence)) errors.push('evidence_not_array');
144+
if (!isStringArray(value.filesReferenced)) errors.push('files_referenced_not_string_array');
145+
if (!isStringArray(value.symbolsReferenced)) errors.push('symbols_referenced_not_string_array');
146+
if (!isStringArray(value.unsupportedClaims)) errors.push('unsupported_claims_not_string_array');
147+
if (typeof value.readyToEdit !== 'boolean') errors.push('ready_to_edit_not_boolean');
148+
149+
const evidence = Array.isArray(value.evidence) ? value.evidence : [];
150+
for (const entry of evidence) {
151+
if (!isRecord(entry)) continue;
152+
errors.push(...findAdditionalFields(entry, evidenceReferenceFields, 'evidence_field'));
153+
if (isRecord(entry.lineRange)) {
154+
errors.push(...findAdditionalFields(entry.lineRange, lineRangeFields, 'line_range_field'));
155+
}
156+
}
157+
const malformedEvidence = evidence.some((entry) => !isValidEvidenceReference(entry));
158+
if (malformedEvidence) errors.push('malformed_evidence_reference');
159+
160+
if (errors.length > 0) return { status: 'invalid_schema', answer: null, errors };
161+
162+
return {
163+
status: 'valid',
164+
answer: {
165+
answer: value.answer as JsonValue,
166+
confidence: value.confidence as ContextBenchStructuredAnswer['confidence'],
167+
evidence: evidence as ContextBenchEvidenceReference[],
168+
filesReferenced: value.filesReferenced as string[],
169+
symbolsReferenced: value.symbolsReferenced as string[],
170+
unsupportedClaims: value.unsupportedClaims as string[],
171+
readyToEdit: value.readyToEdit as boolean
172+
},
173+
errors: []
174+
};
175+
}
176+
177+
export function parseStructuredAnswer(raw: string): StructuredAnswerParseResult {
178+
const trimmed = raw.trim();
179+
if (trimmed.length === 0)
180+
return { status: 'invalid_schema', answer: null, errors: ['missing_json'] };
181+
try {
182+
return validateStructuredAnswer(JSON.parse(trimmed) as unknown);
183+
} catch {
184+
return { status: 'invalid_schema', answer: null, errors: ['invalid_json'] };
185+
}
186+
}
187+
188+
export function classifyStructuredAnswer(
189+
answer: ContextBenchStructuredAnswer,
190+
diagnostics: SchemaBoundDiagnostics = {}
191+
): AnswerClassification {
192+
const reasons: string[] = [];
193+
const malformedEvidence = answer.evidence.some((entry) => !isValidEvidenceReference(entry));
194+
if (answer.unsupportedClaims.length > 0) reasons.push('model_reported_unsupported_claims');
195+
if ((diagnostics.unsupportedEvidenceFiles?.length ?? 0) > 0)
196+
reasons.push('unsupported_evidence_files');
197+
if ((diagnostics.missingRequiredFacts?.length ?? 0) > 0) reasons.push('missing_required_facts');
198+
if ((diagnostics.contradictoryFacts?.length ?? 0) > 0) reasons.push('contradictory_facts');
199+
if ((diagnostics.missingEvidenceFiles?.length ?? 0) > 0) reasons.push('missing_evidence_files');
200+
201+
const unsupportedClaim = reasons.length > 0;
202+
if (answer.readyToEdit && answer.confidence === 'low') reasons.push('ready_with_low_confidence');
203+
if (answer.readyToEdit && answer.evidence.length === 0) reasons.push('ready_without_evidence');
204+
if (answer.readyToEdit && malformedEvidence) reasons.push('ready_with_malformed_evidence');
205+
206+
const falseReady =
207+
answer.readyToEdit &&
208+
(unsupportedClaim ||
209+
answer.confidence === 'low' ||
210+
answer.evidence.length === 0 ||
211+
malformedEvidence);
212+
return { unsupportedClaim, falseReady, reasons: [...new Set(reasons)] };
213+
}
214+
215+
export function evaluateSchemaBoundDiagnostics(
216+
answer: ContextBenchStructuredAnswer,
217+
expected: { requiredFacts?: string[]; requiredEvidenceFiles?: string[] }
218+
): SchemaBoundDiagnostics {
219+
const answerText = JSON.stringify(answer.answer).toLowerCase();
220+
const citedFiles = new Set(answer.evidence.map((entry) => entry.file));
221+
return {
222+
missingRequiredFacts: (expected.requiredFacts ?? []).filter(
223+
(fact) => !answerText.includes(fact.toLowerCase())
224+
),
225+
missingEvidenceFiles: (expected.requiredEvidenceFiles ?? []).filter(
226+
(file) => !citedFiles.has(file)
227+
)
228+
};
229+
}

0 commit comments

Comments
 (0)