Repository navigation
Expand file tree
/
Copy pathexternal-eval.mjs
More file actions
701 lines (655 loc) · 30.9 KB
/
Copy pathexternal-eval.mjs
File metadata and controls
701 lines (655 loc) · 30.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
#!/usr/bin/env node
/**
* External evaluation suite.
*
* Runs the full Drift loop against real open-source Next.js repos and asserts the
* behaviours the falsification test pinned down: every repo onboards, the learned
* contract names the repo's real data layer, an injected direct-DB route is caught
* with correct file:line evidence, and a properly layered route is not flagged.
*
* Output is diffed against scripts/external-eval-baseline.json, so a clean run prints
* one line and a regression prints only what changed.
*
* node scripts/external-eval.mjs # verify against baseline
* node scripts/external-eval.mjs --update # rewrite the baseline
* node scripts/external-eval.mjs --only dub,taxonomy
*
* Repos are expected at $DRIFT_EVAL_REPOS (default ~/drift-falsification/repos).
*/
import { execFileSync } from "node:child_process";
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import { homedir, tmpdir } from "node:os";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { envelopeBudgetBand, mergeBaselineRows, PACKET_ENVELOPE_BUDGET, repoVerdict, unsafeBaselineMoves, updateGate } from "./external-eval-predicate.mjs";
import { EVAL_REPOS } from "./eval-repos.mjs";
import { importOf } from "./data-layer-import.mjs";
import { contaminationAllowed, contaminationRefusal } from "./worktree-contamination.mjs";
import { startCountsFrom } from "./external-eval-start.mjs";
const HERE = dirname(fileURLToPath(import.meta.url));
const REPO_ROOT = resolve(HERE, "..");
const CLI = join(REPO_ROOT, "packages/cli/dist/main.js");
const ENGINE = join(REPO_ROOT, "target/release/drift-engine");
const BASELINE = join(HERE, "external-eval-baseline.json");
const REPOS_DIR = process.env.DRIFT_EVAL_REPOS || join(homedir(), "drift-falsification/repos");
// O-3: the repo table is shared with the evasion matrix - see scripts/eval-repos.mjs
// for the per-repo documentation (data layers, expectedExitCode provenance, transitional
// notes). Extracted so the two harnesses can never disagree about what is under test.
const REPOS = EVAL_REPOS;
// O-1: the exit-code assertion fails closed (a missing expectation is a FAIL), so refuse
// to even start with a suite repo that lacks one - the failure should name the config gap,
// not surface as a per-repo FAIL an hour into a run.
for (const cfg of REPOS) {
if (!Number.isInteger(cfg.expectedExitCode)) {
console.error(`REPOS entry "${cfg.name}" has no integer expectedExitCode; every suite repo must record one.`);
process.exit(1);
}
}
const args = process.argv.slice(2);
const UPDATE = args.includes("--update");
// O-4: each explicitly accepted unsafe baseline move, as "<repo>:<field>". Repeatable.
const ACCEPTED_REGRESSIONS = new Set(
args.flatMap((arg, index) => (arg === "--accept-regression" && args[index + 1] ? [args[index + 1]] : []))
);
const flag = (name) => {
const index = args.indexOf(name);
return index >= 0 ? args[index + 1] : undefined;
};
/**
* T06: evaluate a repo that is not in the suite, for triaging user reports without
* committing to the baseline. Results are printed but never compared or written.
*
* node scripts/external-eval.mjs --repo-path <dir> --data-module <spec> \
* --data-symbol <sym> --route-dir <dir> [--clean-module <spec>] [--declare <specs>] \
* [--expect-exit-code <n>]
*
* The exit-code assertion fails closed: without --expect-exit-code the ad-hoc result is
* FAIL with `exit_code_expectation_recorded` named, so the printed record still shows
* every measured field for triage.
*/
const AD_HOC = flag("--repo-path");
const onlyIndex = args.indexOf("--only");
const only =
onlyIndex >= 0 && args[onlyIndex + 1]
? args[onlyIndex + 1].split(",").map((s) => s.trim()).filter(Boolean)
: null;
const env = { ...process.env, DRIFT_ENGINE_BIN: ENGINE };
delete env.DRIFT_ALLOW_TYPESCRIPT_ENGINE_FALLBACK;
function git(cwd, ...a) {
return execFileSync("git", a, { cwd, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"] });
}
function drift(cwd, extraEnv, ...a) {
try {
const stdout = execFileSync(process.execPath, [CLI, ...a], {
cwd,
encoding: "utf8",
env: { ...env, ...extraEnv },
stdio: ["ignore", "pipe", "pipe"],
maxBuffer: 256 * 1024 * 1024
});
return { ok: true, stdout, stderr: "", code: 0 };
} catch (error) {
return {
ok: false,
stdout: error.stdout?.toString() ?? "",
stderr: error.stderr?.toString() ?? "",
code: error.status ?? 1
};
}
}
/**
* Re-check with only the injected violation present.
*
* Returns whether the finding's enforcement_result equals the contract's enforcement_mode, or
* null when there is nothing to compare. Isolated because the negative controls in the main check
* deliberately import non-existent modules, which suppresses blocking for the whole check.
*/
function enforcementInIsolation(root, dbEnv, repoId, cfg, enforcementMode) {
resetTree(root);
const badPath = writeRoute(root, cfg.routeDir, "drift-eval-enforce", cfg.dataModule, cfg.dataSymbol, cfg);
git(root, "add", "-A");
const run = drift(root, dbEnv, "check", "--diff", "HEAD", "--scope", "changed-hunks", "--repo", repoId, "--json");
let verdict = null;
try {
const payload = JSON.parse(run.stdout);
const finding = (payload.findings ?? []).find((entry) =>
(entry.evidence_refs ?? []).some((ref) => ref.file_path === badPath)
);
verdict =
finding && enforcementMode ? finding.enforcement_result === enforcementMode : null;
} catch {
verdict = null;
}
resetTree(root);
return verdict;
}
function ensureDir(dir) {
mkdirSync(dir, { recursive: true });
return join(dir, "route.ts");
}
/**
* Write an injected route.
*
* `cfg` decides whether the import is named or default: an injection that would not compile in the
* repo it is injected into cannot measure anything about that repo (see data-layer-import.mjs).
*/
function writeRoute(root, relDir, sub, module, symbol, cfg) {
const dir = join(root, relDir, sub);
mkdirSync(dir, { recursive: true });
writeFileSync(
join(dir, "route.ts"),
`import { NextResponse } from "next/server";\n` +
`${importOf(cfg, module, symbol)}\n\n` +
`export async function GET() {\n` +
` return NextResponse.json({ ok: String(${symbol}) });\n` +
`}\n`
);
return `${relDir}/${sub}/route.ts`;
}
/**
* Return the repo to a pristine checkout of HEAD.
*
* `git clean` alone is not enough: it does not remove files that are staged, and the
* injection step stages the routes it writes so they appear in `git diff HEAD`. A
* clean-only reset therefore leaked injected routes between runs, which then got
* committed as part of the base tree and made the injection look undetected. The hard
* reset drops the index as well.
*/
function resetTree(root) {
try {
git(root, "reset", "-q", "--hard", "HEAD");
git(root, "clean", "-qfd");
} catch {
/* best effort */
}
}
function evaluateRepo(cfg) {
return evaluateRepoAt(REPOS_DIR, cfg);
}
function evaluateRepoAt(reposDir, cfg) {
const root = join(reposDir, cfg.name);
const result = { repo: cfg.name };
if (!existsSync(root)) {
return { ...result, status: "MISSING_REPO" };
}
// EW-7 (DET-2): refuse a contaminated worktree rather than measuring it.
//
// Before the reset, deliberately - the reset is what destroyed the evidence that made the cal.com
// findings flap unfalsifiable. A number from a repo another process was editing is not a slightly
// wrong number, it is a number about a different repo, and publishing it looks like measurement.
const contamination = contaminationRefusal(root, cfg.name);
if (contamination.refused && !contaminationAllowed(process.argv)) {
console.error(` REFUSED ${cfg.name}: ${contamination.reason}`);
return {
...result,
status: "CONTAMINATED_WORKTREE",
contaminated_files: contamination.entries.map((entry) => entry.path),
head: contamination.head
};
}
resetTree(root);
// Fresh Drift state per repo: onboarding is part of what we measure.
const home = mkdtempSync(join(tmpdir(), "drift-eval-home-"));
const repoEnv = { HOME: home, DRIFT_HOME: home };
// T02: for a repo whose data layer the whitelist cannot see, first prove the F4 gap is
// actually exercised - inference alone must find no data-access convention, and A6
// discovery must name the wrapper anyway. Without this the repo is in the suite but
// tests nothing it was added for.
if (cfg.whitelistIndependent) {
const probeHome = mkdtempSync(join(tmpdir(), "drift-eval-probe-"));
const probe = drift(
root,
{ HOME: probeHome, DRIFT_HOME: probeHome },
"start",
"--repo-root",
".",
"--accept-defaults",
"--json"
);
try {
const payload = JSON.parse(probe.stdout);
const discovery = payload.data_layer_discovery;
result.inference_alone_found_data_layer = discovery === undefined;
result.discovery_named_data_layer = Boolean(
discovery?.suggestions?.some((s) => s.filePath === cfg.expectDiscoveryWrapper)
);
result.discovery_reason = discovery?.reason ?? null;
} catch {
result.inference_alone_found_data_layer = null;
result.discovery_named_data_layer = false;
result.discovery_reason = "probe_unparseable";
}
rmSync(probeHome, { recursive: true, force: true });
resetTree(root);
}
const started = Date.now();
// BB-8: `--json`, because every count below used to be regexed out of `start`'s human output and one
// of them died silently when BB-3 rewrote the sentence.
//
// The old reader was `/Baselined (\d+) existing violation/`. BB-3's disclosure says
// "397 existing violations baselined - ...", reversed, so the regex matched nothing, `?? 0` supplied
// a plausible zero, and the baseline was then updated 397->0 on every repo under a "new fields"
// rationale. The product was baselining correctly the whole time; only the measurement was dead, and
// with it the ability of this suite to see a baselining regression - the decision-C behaviour T121
// exists to protect.
//
// Sentences are UX and will change again. `start --json` is schema-locked (BetaStartResponseSchema),
// so it is what a gate should read.
const startArgs = ["start", "--repo-root", ".", "--accept-defaults", "--json"];
if (cfg.declaredDataModules) {
startArgs.push("--data-modules", cfg.declaredDataModules);
}
const start = drift(root, repoEnv, ...startArgs);
result.onboard_seconds = Number(((Date.now() - started) / 1000).toFixed(1));
result.onboarded = start.ok;
if (!start.ok) {
result.status = "ONBOARD_FAILED";
result.error = (start.stderr || start.stdout).trim().split("\n")[0]?.slice(0, 300);
rmSync(home, { recursive: true, force: true });
return result;
}
let startPayload;
try {
startPayload = JSON.parse(start.stdout);
} catch (error) {
// A start that exited 0 but produced unparseable JSON is a broken measurement, not a zero.
result.status = "ONBOARD_FAILED";
result.error = `start --json emitted unparseable output: ${error.message}`.slice(0, 300);
rmSync(home, { recursive: true, force: true });
return result;
}
const counts = startCountsFrom(startPayload);
const repoId = counts.repo_id;
result.repo_id = repoId;
result.files = counts.files;
result.facts = counts.facts;
result.candidates = counts.candidates;
result.baselined = counts.baselined;
const dbEnv = { ...repoEnv, DRIFT_DB: join(home, ".drift/repos", repoId ?? "", "drift.sqlite") };
const contractRun = drift(root, dbEnv, "contract", "show", "--repo", repoId, "--json");
let forbidden = [];
// CV-5: the id of the convention these controls are ABOUT. Every false-positive assertion below is a
// statement about the data-access kind, and once a repo can accept more than one convention, "any
// finding on the control route" stops meaning "a data-access false positive". An auth finding on a
// route that genuinely calls no wrapper is TRUE, and scoring it as a false positive would make the
// harness fail for the product being right.
let dataConventionId = null;
try {
const conventions = JSON.parse(contractRun.stdout).contract.conventions;
const dataConvention = conventions.find((c) => c.kind === "api_route_no_direct_data_access");
forbidden = dataConvention?.matcher?.forbidden_imports ?? [];
dataConventionId = dataConvention?.id ?? null;
result.enforcement_mode = dataConvention?.enforcement_mode ?? null;
} catch {
result.enforcement_mode = null;
}
result.forbidden_imports = [...forbidden].sort();
result.contract_names_real_data_layer = cfg.expectForbidden.every((want) =>
forbidden.includes(want)
);
// T14: pin the exact set where it is known. expectForbidden alone only proves the real data
// layer is present, so it would not notice over-matching creeping back - cal.com carried four
// wrong entries out of six while passing that check.
result.forbidden_imports_exact_match =
cfg.expectForbiddenExact === undefined
? null
: JSON.stringify([...cfg.expectForbiddenExact].sort()) === JSON.stringify(result.forbidden_imports);
// Injection and clean control land in the same diff. No commit is needed: the tree is
// already a pristine HEAD checkout, and `git diff HEAD` includes staged new files.
// Committing here would permanently mutate the eval repos.
const badPath = writeRoute(root, cfg.routeDir, "drift-eval-bad", cfg.dataModule, cfg.dataSymbol, cfg);
const cleanPath = writeRoute(
root,
cfg.routeDir,
"drift-eval-clean",
cfg.cleanModule,
cfg.cleanSymbol,
// The clean control imports a *service* module, whose export style is its own; the data layer's
// import kind does not apply, so this stays the named form the clean modules actually use.
{ ...cfg, dataImportKind: "named" }
);
// T03 negative controls. The suite proved detection but never restraint: a rule that
// flagged everything would satisfy every other assertion here. Each of these is a route
// that must NOT be flagged, and each pins a specific fix against real repos rather than
// fixtures.
//
// type-only - A3/F5: `import type` from the data module is erased at compile time and
// creates no runtime dependency.
// lookalike - B3: `<dataModule>-legacy` must not match a forbidden `<dataModule>`,
// which bare substring matching got wrong.
// subpath - B3 the other way: a *genuine* subpath of the data module must still be
// caught, so the boundary fix did not overshoot into a false negative.
const typeOnlyPath = `${cfg.routeDir}/drift-eval-typeonly/route.ts`;
writeFileSync(
ensureDir(join(root, cfg.routeDir, "drift-eval-typeonly")),
`import { NextResponse } from "next/server";\n` +
`import type { ${cfg.dataSymbol} } from "${cfg.dataModule}";\n\n` +
`export async function GET() {\n` +
` const value: ${cfg.dataSymbol} | undefined = undefined;\n` +
` return NextResponse.json({ ok: value === undefined });\n` +
`}\n`
);
const lookalikePath = writeRoute(
root,
cfg.routeDir,
"drift-eval-lookalike",
`${cfg.dataModule}-legacy`,
cfg.dataSymbol,
cfg
);
const subpathPath = writeRoute(
root,
cfg.routeDir,
"drift-eval-subpath",
`${cfg.dataModule}/internal`,
cfg.dataSymbol,
cfg
);
git(root, "add", "-A");
const check = drift(
root,
dbEnv,
"check",
"--diff",
"HEAD",
"--scope",
"changed-files",
"--repo",
repoId,
"--json"
);
// O-1: the exit code is asserted against the per-repo expectation, not merely recorded.
// See the expectedExitCode note on REPOS for why several repos legitimately expect 3.
result.check_exit_code = check.code;
result.expected_exit_code = cfg.expectedExitCode ?? null;
try {
const payload = JSON.parse(check.stdout);
const findings = payload.findings ?? [];
result.check_status = payload.check?.status ?? null;
result.engine_source = payload.check?.fallback_status?.engine_source ?? null;
result.fallback_used = payload.check?.fallback_status?.fallback_used ?? null;
result.can_block = payload.check?.capability_completeness?.can_block ?? null;
result.findings_count = findings.length;
result.blocking_count = payload.summary?.blocking_count ?? 0;
const hasPath = (finding, path) =>
(finding.evidence_refs ?? []).some((ref) => ref.file_path === path);
// CV-5: attribution, not just location. A finding counts toward these controls only if it came from
// the data-access convention they are about. `dataConventionId === null` keeps the old behaviour for
// a repo whose contract could not be read, rather than silently passing everything.
const fromDataAccess = (finding) =>
dataConventionId === null || finding.convention_id === dataConventionId;
const dataFindings = findings.filter(fromDataAccess);
const onBad = dataFindings.filter((finding) => hasPath(finding, badPath));
const onClean = dataFindings.filter((finding) => hasPath(finding, cleanPath));
const evidence = onBad[0]?.evidence_refs?.[0];
result.injection_caught = onBad.length > 0;
// The violating import is always line 2 of the generated route.
result.injection_evidence_correct =
evidence?.start_line === 2 && evidence?.import_source === cfg.dataModule;
// O-1 attribution: the evidence the finding leads with must name the injected route
// itself. `injection_caught` only proves SOME evidence ref touches the route; a finding
// attributed to an intermediate barrel would still satisfy it (the papermark artifact).
result.injected_route = badPath;
result.injection_evidence_file = evidence?.file_path ?? null;
// BB-5: the exemplar integrity invariant, measured on every suite repo rather than only in
// fixtures. An exemplar that itself violates the convention it exemplifies is what produced the
// trial-B1 defection ("the preflight's claim doesn't hold up against the actual codebase"), so
// this is asserted, not recorded: a violation is a product defect, not a baseline movement.
const violatingPaths = new Set(
findings.flatMap((finding) => (finding.evidence_refs ?? []).map((ref) => ref.file_path))
);
const emittedExemplars = findings.flatMap((finding) =>
(finding.conforming_examples ?? []).map((example) => example.file_path)
);
result.exemplars_emitted = emittedExemplars.length;
result.exemplar_integrity = emittedExemplars.every((path) => !violatingPaths.has(path));
result.exemplar_violators = emittedExemplars.filter((path) => violatingPaths.has(path)).sort();
// BB-6: the guidance view's byte budget, asserted on every suite repo rather than in a fixture.
// cal.com is the worst case (~2,500 parser gaps) and is exactly the repo a synthetic test would
// fail to represent. EW-8's lesson: only byte assertions force real fixes.
const prepared = drift(root, dbEnv, "prepare", "add an endpoint that lists items", "--repo", repoId, "--json");
if (prepared.ok) {
try {
const packet = JSON.parse(prepared.stdout);
result.guidance_bytes = Buffer.byteLength(JSON.stringify(packet.guidance ?? null), "utf8");
result.guidance_within_budget = packet.guidance ? result.guidance_bytes <= 32768 : false;
// A threshold, not a recorded byte count: the full envelope's size moves by tens of bytes
// between runs (ids and timestamps), so baselining the exact number would make the suite flap
// on noise. 500 KB is the BB-6 requirement; the exact figure is not the thing under test.
// W7: record the measurement, not just the verdict. `packet_within_envelope_budget` is a
// threshold with nothing behind it, so when it flipped there was no way to see what had
// grown without reproducing the whole run by hand. Volatile, for the reason the comment
// above gives - the number moves by tens of bytes between runs - but present.
result.packet_bytes = Buffer.byteLength(prepared.stdout, "utf8");
result.packet_largest_sections = Object.entries(packet)
.map(([key, value]) => [key, Buffer.byteLength(JSON.stringify(value) ?? "null", "utf8")])
.sort((left, right) => right[1] - left[1])
.slice(0, 3)
.map(([key, bytes]) => `${key}:${bytes}`);
result.packet_within_envelope_budget = result.packet_bytes < PACKET_ENVELOPE_BUDGET;
// W0: the band IS compared (it is deliberately absent from VOLATILE). The boolean above
// only moves once the budget is already blown; this moves at 60%, while there is still
// room to act. See envelopeBudgetBand for why a band rather than the raw byte count.
result.packet_budget_band = envelopeBudgetBand(result.packet_bytes);
} catch (error) {
result.guidance_within_budget = false;
result.guidance_parse_error = String(error.message).slice(0, 200);
}
} else {
result.guidance_within_budget = false;
result.guidance_error = (prepared.stderr || "prepare failed").trim().split("\n").slice(-1)[0];
}
result.injection_diff_status = onBad[0]?.diff_status ?? null;
result.injection_enforcement = onBad[0]?.enforcement_result ?? null;
result.injection_finding_status = onBad[0]?.status ?? null;
// Enforcement is measured in a SEPARATE check containing only the injected violation.
//
// It cannot be measured in the check above, and that is by construction rather than by
// accident: the lookalike and subpath probes must import modules that do not exist in order
// to be negative controls, and an unresolvable import on a route in scope legitimately makes
// capability coverage incomplete. The engine then declines to block anything - correctly, and
// the contract enforces it (`blocking findings require complete capability coverage`).
//
// So the previously recorded mismatch on taxonomy, cal.com, papermark and midday was the
// harness suppressing the very thing it was trying to observe. Product behaviour is right.
result.enforcement_matches_mode = enforcementInIsolation(root, dbEnv, repoId, cfg, result.enforcement_mode);
result.clean_control_false_positive = onClean.length > 0;
result.fp_type_only_import = dataFindings.some((f) => hasPath(f, typeOnlyPath));
result.fp_lookalike_module = dataFindings.some((f) => hasPath(f, lookalikePath));
result.catches_genuine_subpath = dataFindings.some((f) => hasPath(f, subpathPath));
} catch {
result.check_status = "UNPARSEABLE";
}
resetTree(root);
rmSync(home, { recursive: true, force: true });
const verdict = repoVerdict(result, cfg);
result.status = verdict.status;
if (verdict.failures.length) {
result.failed_assertions = verdict.failures;
}
return result;
}
// Counts move as upstream repos change; compared for reporting only, not for regressions.
/**
* T04: onboarding time is compared against a generous ceiling rather than exactly, so a
* real regression fails while ordinary upstream growth does not. The ceiling comes from the
* baseline (3x, floor 30s) because absolute numbers differ per machine.
*/
function performanceCeiling(baselineSeconds) {
return Math.max(30, Math.round((baselineSeconds ?? 0) * 3));
}
// Fields excluded from the baseline diff because they are environment-dependent rather than product
// behaviour: wall-clock timing, a repo id derived from an absolute path, and the raw scan counts that
// move with any engine extraction change.
//
// BB-8: `baselined` was in this set, and that is the second half of why the dead cell went unnoticed.
// Being volatile, its 397->0 collapse was never printed as a "changed vs baseline" line - so the update
// that recorded the corpse looked like it touched only the three new exemplar fields. It is a
// deterministic product output on a pinned repo (the whole point of the decision-C behaviour T121
// protects), so it belongs in the gate, not in the noise.
const VOLATILE = new Set([
"onboard_seconds",
"repo_id",
"files",
"facts",
"candidates",
// W7: the packet's measured size and its top sections. Recorded so a budget flip can be
// diagnosed from the baseline instead of by reproducing the run, and volatile for the same
// reason `packet_within_envelope_budget` is a threshold rather than a pinned number: ids and
// timestamps move it by tens of bytes between runs. The gated field is still the boolean.
"packet_bytes",
"packet_largest_sections"
]);
function diffResult(before, after) {
const changes = [];
for (const key of new Set([...Object.keys(before ?? {}), ...Object.keys(after)])) {
if (key === "repo" || VOLATILE.has(key)) continue;
const a = JSON.stringify(before?.[key]);
const b = JSON.stringify(after[key]);
if (a !== b) changes.push(`${key}: ${a ?? "(absent)"} -> ${b}`);
}
return changes;
}
/**
* Disk preflight. Each repo's evaluation builds a fresh Drift state in a temp HOME, and for a
* large repo that reaches ~1 GB. Exhausting the disk mid-run does not fail cleanly: it
* produces false test failures and a raw "database or disk is full" from SQLite, so results
* become untrustworthy rather than merely incomplete. Refuse up front instead.
*/
function freeSpaceGb() {
try {
const out = execFileSync("df", ["-k", REPOS_DIR], { encoding: "utf8" }).trim().split("\n").pop();
const available = Number(out.split(/\s+/)[3]);
return Number.isFinite(available) ? available / 1024 / 1024 : Infinity;
} catch {
return Infinity;
}
}
const MIN_FREE_GB = 5;
const free = freeSpaceGb();
if (free < MIN_FREE_GB) {
console.error(
`Refusing to run: ${free.toFixed(1)} GB free, need ${MIN_FREE_GB} GB.\n` +
`Each repo builds a fresh Drift state (~1 GB for the largest). Free space with:\n` +
` rm -rf ~/.drift /tmp/drift-eval-*`
);
process.exit(3);
}
if (!existsSync(CLI)) {
console.error(`Missing CLI build at ${CLI}. Run: pnpm build`);
process.exit(1);
}
if (!existsSync(ENGINE)) {
console.error(`Missing engine at ${ENGINE}. Run: cargo build --release -p drift-engine`);
process.exit(1);
}
if (AD_HOC) {
const { basename, dirname: parentOf } = await import("node:path");
const adHocCfg = {
name: basename(AD_HOC),
routeDir: flag("--route-dir") ?? "app/api",
dataModule: flag("--data-module") ?? "",
dataSymbol: flag("--data-symbol") ?? "db",
cleanModule: flag("--clean-module") ?? "next/server",
cleanSymbol: "NextResponse",
expectForbidden: flag("--data-module") ? [flag("--data-module")] : [],
...(flag("--declare") ? { declaredDataModules: flag("--declare") } : {}),
...(flag("--expect-exit-code") !== undefined
? { expectedExitCode: Number(flag("--expect-exit-code")) }
: {})
};
if (!adHocCfg.dataModule) {
console.error("--repo-path requires --data-module");
process.exit(1);
}
// evaluateRepo resolves the repo under REPOS_DIR, so point that at the parent.
process.env.DRIFT_EVAL_REPOS = parentOf(AD_HOC);
const adHocResult = evaluateRepoAt(parentOf(AD_HOC), adHocCfg);
console.log(JSON.stringify(adHocResult, null, 2));
process.exit(adHocResult.status === "PASS" ? 0 : 1);
}
const results = REPOS.filter((cfg) => !only || only.includes(cfg.name)).map((cfg) => {
const r = evaluateRepo(cfg);
console.log(
` ${r.status === "PASS" ? "ok " : "FAIL"} ${r.repo.padEnd(11)}` +
` onboard=${r.onboarded ? "y" : "n"}` +
` contract=${r.contract_names_real_data_layer ? "y" : "n"}` +
` injected=${r.injection_caught ? "y" : "n"}` +
` evidence=${r.injection_evidence_correct ? "y" : "n"}` +
` cleanFP=${r.clean_control_false_positive ? "YES" : "no"}` +
` neg=${r.fp_type_only_import === false && r.fp_lookalike_module === false ? "ok" : "FP"}` +
` subpath=${r.catches_genuine_subpath ? "y" : "n"}` +
// BB-5: `n/N` - clean exemplars emitted out of exemplars emitted. A visible 0 total is the
// signal that the feature stopped running, which a boolean pass would hide.
` exemplars=${(r.exemplars_emitted ?? 0) - (r.exemplar_violators?.length ?? 0)}/${r.exemplars_emitted ?? 0}` +
` guidance=${r.guidance_bytes === undefined ? "?" : `${Math.round(r.guidance_bytes / 1024)}k`}` +
(r.discovery_named_data_layer !== undefined
? ` f4gap=${r.inference_alone_found_data_layer === false && r.discovery_named_data_layer ? "y" : "n"}`
: "") +
` (${r.onboard_seconds ?? "?"}s)` +
(r.error ? `\n ${r.error}` : "") +
(r.failed_assertions ? `\n failed: ${r.failed_assertions.join(", ")}` : "")
);
return r;
});
const baseline = existsSync(BASELINE) ? JSON.parse(readFileSync(BASELINE, "utf8")) : [];
const byName = new Map(baseline.map((r) => [r.repo, r]));
if (UPDATE) {
// O-4: --update is not a rubber stamp.
// (1) --only merges into the existing baseline; it used to TRUNCATE it to the
// filtered repos, silently destroying every other row (verified live: 7 -> 1).
// (2) a safety-relevant field moving in the unsafe direction (enforcement
// block -> none/warn, blocking_count > 0 -> 0, caught -> uncaught, exit 2/3 -> 0,
// fail/refused -> pass) is refused with no write unless each move is named via
// --accept-regression <repo>:<field>.
// (3) a FAILING verdict is refused with no write and a nonzero exit - it used to
// print "baseline updated - 0/1 passing" and exit 0.
const gate = updateGate({
results,
baselineByRepo: byName,
acceptedRegressions: ACCEPTED_REGRESSIONS,
unsafeMovesFor: unsafeBaselineMoves
});
if (!gate.ok) {
console.error("\nrefusing to update baseline:");
for (const refusal of gate.refusals) console.error(` ${refusal}`);
process.exit(1);
}
const merged = mergeBaselineRows(baseline, results, REPOS.map((cfg) => cfg.name));
writeFileSync(BASELINE, `${JSON.stringify(merged, null, 2)}\n`);
console.log(
`\nbaseline updated - ${results.length}/${results.length} passing` +
(only ? ` (merged ${results.length} repo(s) into the ${merged.length}-row baseline)` : "")
);
process.exit(0);
}
const changed = [];
for (const after of results) {
const before = byName.get(after.repo);
const ceiling = performanceCeiling(before?.onboard_seconds);
if (before && after.onboard_seconds > ceiling) {
changed.push(
` ${after.repo}:`,
` onboard_seconds: ${before.onboard_seconds} -> ${after.onboard_seconds} (exceeds ${ceiling}s ceiling)`
);
after.status = "FAIL";
}
const changes = diffResult(before, after);
if (changes.length) {
changed.push(` ${after.repo}:`);
for (const line of changes) changed.push(` ${line}`);
}
}
const failing = results.filter((r) => r.status !== "PASS");
if (!changed.length && !failing.length) {
console.log(`\nno change vs baseline - ${results.length}/${results.length} passing`);
process.exit(0);
}
if (changed.length) {
console.log("\nchanged vs baseline:");
console.log(changed.join("\n"));
}
if (failing.length) {
console.log(`\n${failing.length} repo(s) failing: ${failing.map((r) => r.repo).join(", ")}`);
}
process.exit(1);