Skip to content

Commit b2fa208

Browse files
committed
fix(format): format ContextBench harness sources
1 parent ffa7e73 commit b2fa208

2 files changed

Lines changed: 46 additions & 15 deletions

File tree

‎src/eval/contextbench-answer.ts‎

Lines changed: 8 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -103,11 +103,11 @@ function isJsonValue(value: unknown): value is JsonValue {
103103

104104
export function isValidEvidenceReference(value: unknown): value is ContextBenchEvidenceReference {
105105
if (!isRecord(value)) return false;
106-
if (findAdditionalFields(value, evidenceReferenceFields, 'evidence_field').length > 0) return false;
106+
if (findAdditionalFields(value, evidenceReferenceFields, 'evidence_field').length > 0)
107+
return false;
107108
const lineRange = value.lineRange;
108109
if (!isRecord(lineRange)) return false;
109-
if (findAdditionalFields(lineRange, lineRangeFields, 'line_range_field').length > 0)
110-
return false;
110+
if (findAdditionalFields(lineRange, lineRangeFields, 'line_range_field').length > 0) return false;
111111
const start = lineRange.start;
112112
const end = lineRange.end;
113113
return (
@@ -134,7 +134,11 @@ function validateStructuredAnswer(value: unknown): StructuredAnswerParseResult {
134134
if (!(field in value)) errors.push(`missing_${field}`);
135135
}
136136
errors.push(
137-
...findAdditionalFields(value, new Set(CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS), 'root_field')
137+
...findAdditionalFields(
138+
value,
139+
new Set(CONTEXTBENCH_STRUCTURED_ANSWER_REQUIRED_FIELDS),
140+
'root_field'
141+
)
138142
);
139143

140144
if (!isJsonValue(value.answer)) errors.push('answer_not_json_value');

‎src/eval/contextbench-evidence-gate.ts‎

Lines changed: 38 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -187,8 +187,15 @@ function hasOfficialEvaluatorProof(
187187
);
188188
}
189189

190-
function hasDiagnosticFallback(row: ContextBenchRunManifestRow, score: ContextBenchScoreEvidence | undefined): boolean {
191-
return row.scoring.claimBearing === false || Boolean(row.scoring.fallbackReason) || score?.mode === 'diagnostic_fallback';
190+
function hasDiagnosticFallback(
191+
row: ContextBenchRunManifestRow,
192+
score: ContextBenchScoreEvidence | undefined
193+
): boolean {
194+
return (
195+
row.scoring.claimBearing === false ||
196+
Boolean(row.scoring.fallbackReason) ||
197+
score?.mode === 'diagnostic_fallback'
198+
);
192199
}
193200

194201
function hasLaneIsolationProof(
@@ -198,7 +205,8 @@ function hasLaneIsolationProof(
198205
): boolean {
199206
if (!isolation?.proven) return false;
200207
if (!policy) return false;
201-
if (!isolation.sourceKind || ['not_captured', 'env_override'].includes(isolation.sourceKind)) return false;
208+
if (!isolation.sourceKind || ['not_captured', 'env_override'].includes(isolation.sourceKind))
209+
return false;
202210
if (policy.laneId !== row.lane_id) return false;
203211
if (isolation.laneId !== row.lane_id) return false;
204212
if (isolation.expectedContextTool !== policy.expectedContextTool) return false;
@@ -219,7 +227,8 @@ function hasRunnerProvenance(
219227
rawTrace: ContextBenchRawTraceEvidence | undefined,
220228
expectedRunnerHash: string | undefined
221229
): boolean {
222-
if (!rawTrace?.executor || !rawTrace.model || !rawTrace.runnerHash || !expectedRunnerHash) return false;
230+
if (!rawTrace?.executor || !rawTrace.model || !rawTrace.runnerHash || !expectedRunnerHash)
231+
return false;
223232
return (
224233
rawTrace.executor === row.taskExecution.executor &&
225234
rawTrace.model === row.taskExecution.model &&
@@ -228,7 +237,9 @@ function hasRunnerProvenance(
228237
);
229238
}
230239

231-
function rowKey(row: Pick<ContextBenchRunManifestRow, 'lane_id' | 'task_id' | 'repeat_index'>): string {
240+
function rowKey(
241+
row: Pick<ContextBenchRunManifestRow, 'lane_id' | 'task_id' | 'repeat_index'>
242+
): string {
232243
return `${row.lane_id}\u0000${row.task_id}\u0000${row.repeat_index}`;
233244
}
234245

@@ -252,7 +263,11 @@ export function evaluateContextBenchEvidenceGate(
252263
});
253264
}
254265

255-
if (input.expectedTotalRows <= 0 || input.requiredLaneIds.length === 0 || input.requiredTaskIds.length === 0) {
266+
if (
267+
input.expectedTotalRows <= 0 ||
268+
input.requiredLaneIds.length === 0 ||
269+
input.requiredTaskIds.length === 0
270+
) {
256271
failures.push({
257272
code: 'denominator_contract_missing',
258273
message: 'Claim validation requires a frozen denominator contract.'
@@ -289,7 +304,11 @@ export function evaluateContextBenchEvidenceGate(
289304
}
290305
if (row.protocol_hash !== input.expectedProtocolHash) {
291306
failures.push(
292-
makeFailure(row, 'protocol_hash_mismatch', 'Row protocol hash does not match the frozen protocol hash.')
307+
makeFailure(
308+
row,
309+
'protocol_hash_mismatch',
310+
'Row protocol hash does not match the frozen protocol hash.'
311+
)
293312
);
294313
}
295314
if (row.task_manifest_hash !== input.expectedTaskManifestHash) {
@@ -351,7 +370,9 @@ export function evaluateContextBenchEvidenceGate(
351370

352371
const artifacts = input.artifactsByRunId[row.run_id];
353372
if (row.status !== 'completed') {
354-
failures.push(makeFailure(row, 'non_completed_status', 'Claim-bearing runs must complete.'));
373+
failures.push(
374+
makeFailure(row, 'non_completed_status', 'Claim-bearing runs must complete.')
375+
);
355376
}
356377

357378
if (
@@ -377,11 +398,15 @@ export function evaluateContextBenchEvidenceGate(
377398
);
378399
}
379400

380-
if (!hasLaneIsolationProof(row, artifacts?.laneIsolation, input.lanePoliciesById[row.lane_id])) {
401+
if (
402+
!hasLaneIsolationProof(row, artifacts?.laneIsolation, input.lanePoliciesById[row.lane_id])
403+
) {
381404
failures.push(
382405
makeFailure(
383406
row,
384-
artifacts?.laneIsolation?.violations?.length ? 'lane_isolation_violation' : 'lane_isolation_missing',
407+
artifacts?.laneIsolation?.violations?.length
408+
? 'lane_isolation_violation'
409+
: 'lane_isolation_missing',
385410
'Lane isolation must be proven by explicit allowed/observed tool evidence.'
386411
)
387412
);
@@ -410,7 +435,9 @@ export function evaluateContextBenchEvidenceGate(
410435
}
411436
}
412437

413-
const blockingFailures = failures.filter((failure) => failure.code !== 'artifact_verification_missing');
438+
const blockingFailures = failures.filter(
439+
(failure) => failure.code !== 'artifact_verification_missing'
440+
);
414441
const shapePass = blockingFailures.length === 0;
415442
const claimPass = failures.length === 0;
416443
return {

0 commit comments

Comments
 (0)