tickmarkr 2.5.4 → 2.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/adapters/registry.js +6 -1
- package/dist/adapters/types.d.ts +3 -0
- package/dist/cli/commands/approve.d.ts +1 -0
- package/dist/cli/commands/approve.js +118 -9
- package/dist/cli/commands/doctor.d.ts +10 -0
- package/dist/cli/commands/doctor.js +44 -29
- package/dist/cli/commands/fleet.js +199 -33
- package/dist/cli/commands/init.js +196 -6
- package/dist/cli/commands/plan.js +20 -4
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/run.js +11 -2
- package/dist/cli/commands/verify.js +102 -81
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +3 -1
- package/dist/config/config.d.ts +25 -8
- package/dist/config/config.js +41 -29
- package/dist/config/fleet-overlay.d.ts +3 -9
- package/dist/config/fleet-overlay.js +114 -19
- package/dist/config/fleet-why.d.ts +7 -0
- package/dist/config/fleet-why.js +5 -0
- package/dist/drivers/index.d.ts +15 -1
- package/dist/drivers/index.js +39 -10
- package/dist/drivers/orca.d.ts +119 -10
- package/dist/drivers/orca.js +781 -110
- package/dist/gates/baseline.d.ts +17 -3
- package/dist/gates/baseline.js +63 -15
- package/dist/gates/cache.d.ts +101 -0
- package/dist/gates/cache.js +401 -0
- package/dist/gates/llm.d.ts +3 -0
- package/dist/gates/llm.js +11 -0
- package/dist/gates/review.d.ts +28 -3
- package/dist/gates/review.js +106 -14
- package/dist/gates/run-gates.d.ts +9 -1
- package/dist/gates/run-gates.js +407 -102
- package/dist/gates/test-manifest.d.ts +128 -0
- package/dist/gates/test-manifest.js +463 -0
- package/dist/gates/test-reporter.d.ts +4 -0
- package/dist/gates/test-reporter.js +57 -0
- package/dist/graph/graph.d.ts +2 -0
- package/dist/graph/graph.js +45 -1
- package/dist/route/preference.d.ts +22 -1
- package/dist/route/preference.js +123 -25
- package/dist/route/router.js +31 -6
- package/dist/run/consult.js +5 -4
- package/dist/run/daemon.d.ts +15 -0
- package/dist/run/daemon.js +638 -167
- package/dist/run/execution-budget.d.ts +25 -0
- package/dist/run/execution-budget.js +142 -0
- package/dist/run/git.d.ts +50 -1
- package/dist/run/git.js +131 -12
- package/dist/run/journal.d.ts +16 -2
- package/dist/run/journal.js +115 -25
- package/dist/run/lease.d.ts +58 -0
- package/dist/run/lease.js +310 -0
- package/dist/run/merge.d.ts +2 -0
- package/dist/run/merge.js +91 -3
- package/dist/run/operator-state.d.ts +11 -0
- package/dist/run/operator-state.js +17 -3
- package/dist/run/recovery.d.ts +8 -0
- package/dist/run/recovery.js +25 -0
- package/dist/run/repair-selection.d.ts +12 -0
- package/dist/run/repair-selection.js +56 -0
- package/dist/run/stall.d.ts +6 -1
- package/dist/run/stall.js +60 -3
- package/dist/tui/cockpit/board.d.ts +96 -0
- package/dist/tui/cockpit/board.js +346 -0
- package/dist/tui/cockpit/decision-actions.js +2 -0
- package/dist/tui/cockpit/layout.d.ts +5 -1
- package/dist/tui/cockpit/layout.js +8 -3
- package/dist/tui/cockpit/live-runtime.js +71 -21
- package/dist/tui/cockpit/run-view.d.ts +7 -5
- package/dist/tui/cockpit/run-view.js +12 -11
- package/dist/tui/ink/fleet-app.d.ts +49 -27
- package/dist/tui/ink/fleet-app.js +229 -38
- package/package.json +2 -2
- package/skills/tickmarkr-overseer/SKILL.md +173 -113
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { mkdtempSync, readFileSync, rmSync } from "node:fs";
|
|
1
|
+
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
2
2
|
import { loadavg, tmpdir } from "node:os";
|
|
3
3
|
import { join, posix } from "node:path";
|
|
4
4
|
import { channelKey, shq } from "../adapters/types.js";
|
|
@@ -6,15 +6,19 @@ import { TIER_RANK } from "../config/config.js";
|
|
|
6
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
7
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
8
8
|
import { acceptanceGate } from "./acceptance.js";
|
|
9
|
-
import { compareToBaseline } from "./baseline.js";
|
|
9
|
+
import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
|
|
10
10
|
import { evidenceGate } from "./evidence.js";
|
|
11
11
|
import { captureLlmOutput } from "./llm.js";
|
|
12
12
|
import { disallowedBy } from "../route/preference.js";
|
|
13
13
|
import { marginalCostRank } from "../route/router.js";
|
|
14
|
-
import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
14
|
+
import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
15
|
import { scopeGate } from "./scope.js";
|
|
16
|
-
import {
|
|
16
|
+
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
+
import { executionSignal } from "../run/execution-budget.js";
|
|
18
|
+
import { failureDisposition } from "../run/recovery.js";
|
|
19
|
+
import { preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
17
20
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
21
|
+
import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
18
22
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
19
23
|
let loadProvider = productionLoadProvider;
|
|
20
24
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
@@ -94,6 +98,23 @@ async function captureLlmDispatches(adapters, run) {
|
|
|
94
98
|
throw error;
|
|
95
99
|
}
|
|
96
100
|
}
|
|
101
|
+
/** Parse `git status --porcelain --untracked-files=all -z` output. A rename/copy carries its
|
|
102
|
+
* ORIGINAL path in a second NUL field immediately after the current one; skip it — dirt only
|
|
103
|
+
* cares about paths that exist in the worktree now. */
|
|
104
|
+
function parseStatusZ(stdout) {
|
|
105
|
+
const fields = stdout.split("\0");
|
|
106
|
+
const entries = [];
|
|
107
|
+
for (let i = 0; i < fields.length; i++) {
|
|
108
|
+
const rec = fields[i];
|
|
109
|
+
if (!rec)
|
|
110
|
+
continue;
|
|
111
|
+
const status = rec.slice(0, 2);
|
|
112
|
+
entries.push({ status, path: rec.slice(3) });
|
|
113
|
+
if (status.includes("R") || status.includes("C"))
|
|
114
|
+
i++; // consume the paired original-path field
|
|
115
|
+
}
|
|
116
|
+
return entries;
|
|
117
|
+
}
|
|
97
118
|
const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
|
|
98
119
|
// relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
|
|
99
120
|
const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
|
|
@@ -192,13 +213,68 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
192
213
|
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
193
214
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
194
215
|
}
|
|
216
|
+
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
217
|
+
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
|
|
218
|
+
const entry = baseline.commands.test;
|
|
219
|
+
const outcome = await evaluateManifestedTest(cmd, worktree, {
|
|
220
|
+
baselineDurations: entry?.fileDurations,
|
|
221
|
+
longestFile: entry?.longestFile,
|
|
222
|
+
overallCeilingMs: effectiveCeilingMs(entry),
|
|
223
|
+
artifactDir,
|
|
224
|
+
});
|
|
225
|
+
const reportPath = outcome.reportPath;
|
|
226
|
+
if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
|
|
227
|
+
&& outcome.meta?.retryable !== false) {
|
|
228
|
+
const waitedMs = await waitForCalmWindow(executionSignal());
|
|
229
|
+
if (!calmWindowReady())
|
|
230
|
+
return { gate: "test", pass: false, details: outcome.details,
|
|
231
|
+
meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
|
|
232
|
+
if (!retry.authorizeRetry("infra")) {
|
|
233
|
+
return { gate: "test", pass: false, details: outcome.details,
|
|
234
|
+
meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
|
|
235
|
+
}
|
|
236
|
+
const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
|
|
237
|
+
return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
238
|
+
}
|
|
239
|
+
return {
|
|
240
|
+
gate: "test",
|
|
241
|
+
pass: outcome.pass,
|
|
242
|
+
details: outcome.details,
|
|
243
|
+
meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
|
|
247
|
+
const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
|
|
248
|
+
/** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
|
|
249
|
+
* classification happens after persistence, too late to keep a scripted runner's non-verdict out.
|
|
250
|
+
* Keep named failures as work verdicts and preserve details for failure-policy fingerprinting. */
|
|
251
|
+
function classifySignalOnlyTest(g) {
|
|
252
|
+
if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
|
|
253
|
+
return;
|
|
254
|
+
const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
|
|
255
|
+
if (named || FAILURE_IDENTITY_RE.test(g.details))
|
|
256
|
+
return;
|
|
257
|
+
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
258
|
+
}
|
|
195
259
|
export async function runGates(task, ctx) {
|
|
196
260
|
const results = [];
|
|
261
|
+
let selectionDecision;
|
|
197
262
|
let commits = [];
|
|
263
|
+
const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
|
|
264
|
+
const verdictStore = getVerdictStore(stateDir);
|
|
265
|
+
// R41: the verification protocol and the EFFECTIVE npm lifecycle policy measured for THIS
|
|
266
|
+
// checkout — the policy its runner children receive (an explicit process export, else npm's own
|
|
267
|
+
// resolved config for this worktree, project npmrc included). Every result leaves through
|
|
268
|
+
// withTelemetry carrying it, so the daemon's gate-result row records what the gate measured under
|
|
269
|
+
// rather than a session-wide value resolved somewhere else.
|
|
270
|
+
const verification = verificationProtocol(process.env, ctx.worktree);
|
|
271
|
+
// VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
|
|
272
|
+
// the ledger names the reuse and the identity even where the gate-result row's details must stay
|
|
273
|
+
// the fresh verdict's (see formatReusedRow).
|
|
274
|
+
const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
|
|
198
275
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
199
276
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
200
277
|
const failed = () => results.some((r) => !r.pass);
|
|
201
|
-
const v185 = ctx.pipeline === "v185";
|
|
202
278
|
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
203
279
|
// round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
|
|
204
280
|
// exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
|
|
@@ -250,13 +326,17 @@ export async function runGates(task, ctx) {
|
|
|
250
326
|
// path that forgets to measure is visibly missing its telemetry rather than carrying a fabricated
|
|
251
327
|
// zero. The daemon lifts these off `meta` onto the gate-result row (src/run/daemon.ts).
|
|
252
328
|
const withTelemetry = (result) => {
|
|
329
|
+
if (result.gate === "test" && selectionDecision) {
|
|
330
|
+
result = { ...result, meta: { ...result.meta, selectionDecision } };
|
|
331
|
+
}
|
|
253
332
|
const span = spans.get(result.gate);
|
|
254
333
|
if (!span)
|
|
255
|
-
return result;
|
|
334
|
+
return { ...result, meta: { ...result.meta, verification } };
|
|
256
335
|
return {
|
|
257
336
|
...result,
|
|
258
337
|
meta: {
|
|
259
338
|
...result.meta,
|
|
339
|
+
verification,
|
|
260
340
|
...span,
|
|
261
341
|
...(result.gate === "test" && selectedDurationMs !== undefined ? { selectedDurationMs } : {}),
|
|
262
342
|
...(result.gate === "test" && fullDurationMs !== undefined ? { fullDurationMs } : {}),
|
|
@@ -297,7 +377,7 @@ export async function runGates(task, ctx) {
|
|
|
297
377
|
if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
|
|
298
378
|
const dirt = await dirtyWorktree();
|
|
299
379
|
if (dirt) {
|
|
300
|
-
const refusal = withTelemetry(dirtyRoundRefusal(last.gate, dirt));
|
|
380
|
+
const refusal = withTelemetry(await dirtyRoundRefusal(last.gate, dirt));
|
|
301
381
|
results[results.indexOf(last)] = refusal;
|
|
302
382
|
sorted[sorted.length - 1] = refusal;
|
|
303
383
|
await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
|
|
@@ -322,89 +402,236 @@ export async function runGates(task, ctx) {
|
|
|
322
402
|
* exempt: an untracked source file is uncommitted work by every reading git offers.
|
|
323
403
|
*/
|
|
324
404
|
const dirtyWorktree = async () => {
|
|
325
|
-
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
|
|
405
|
+
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain --untracked-files=all -z", ctx.worktree);
|
|
326
406
|
if (r.code !== 0)
|
|
327
|
-
return `git status failed (exit ${r.code}) — the worktree cannot be proven clean
|
|
328
|
-
const entries = r.stdout
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
return entries.length ? entries.join("\n") : undefined;
|
|
407
|
+
return { text: `git status failed (exit ${r.code}) — the worktree cannot be proven clean`, entries: [] };
|
|
408
|
+
const entries = parseStatusZ(r.stdout).filter((e) => !/^\.tickmarkr-[^/]*$/.test(e.path));
|
|
409
|
+
if (!entries.length)
|
|
410
|
+
return undefined;
|
|
411
|
+
return { text: entries.map((e) => `${e.status} ${e.path}`).join("\n"), entries };
|
|
333
412
|
};
|
|
334
413
|
const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
|
|
335
414
|
+ `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
|
|
336
415
|
+ `and never merged (and the committed diff would never be run)`;
|
|
337
416
|
// `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
|
|
338
|
-
const dirtyRefusal = (gate, dirt, left) =>
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
417
|
+
const dirtyRefusal = async (gate, dirt, left) => {
|
|
418
|
+
let preservedRef;
|
|
419
|
+
let preservationError;
|
|
420
|
+
try {
|
|
421
|
+
preservedRef = await preserveWorktree(ctx.worktree);
|
|
422
|
+
}
|
|
423
|
+
catch (error) {
|
|
424
|
+
// Never masks the refusal, but never pretends a snapshot exists either — surfaced below.
|
|
425
|
+
preservationError = error instanceof Error ? error.message : String(error);
|
|
426
|
+
}
|
|
427
|
+
const dirtyPaths = [];
|
|
428
|
+
let allUntracked = true;
|
|
429
|
+
let totalBytes = 0;
|
|
430
|
+
for (const { status, path } of dirt.entries) {
|
|
431
|
+
dirtyPaths.push(path);
|
|
432
|
+
if (status !== "??") {
|
|
433
|
+
allUntracked = false;
|
|
434
|
+
}
|
|
435
|
+
try {
|
|
436
|
+
const st = statSync(join(ctx.worktree, path));
|
|
437
|
+
if (st.isFile()) {
|
|
438
|
+
totalBytes += st.size;
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
catch {
|
|
442
|
+
// ignore deleted or unreadable
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
let allAbsentFromDiff = true;
|
|
446
|
+
if (ctx.baseRef) {
|
|
447
|
+
try {
|
|
448
|
+
const diffOut = await shGit(`git diff --name-only -z ${shq(ctx.baseRef)}..HEAD`, ctx.worktree);
|
|
449
|
+
if (diffOut.code === 0) {
|
|
450
|
+
// Paths in Git's newline output are C-quoted, which is not a lossless path encoding
|
|
451
|
+
// (and must never be parsed as JSON). `-z` lets this comparison retain arbitrary
|
|
452
|
+
// filenames exactly, including whitespace and non-ASCII bytes.
|
|
453
|
+
const touched = new Set(diffOut.stdout.split("\0").filter(Boolean));
|
|
454
|
+
for (const p of dirtyPaths) {
|
|
455
|
+
if (touched.has(p)) {
|
|
456
|
+
allAbsentFromDiff = false;
|
|
457
|
+
break;
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
else {
|
|
462
|
+
// An unreadable committed diff cannot prove the litter is unrelated to the worker.
|
|
463
|
+
allAbsentFromDiff = false;
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
catch {
|
|
467
|
+
allAbsentFromDiff = false;
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
const isInfra = Boolean(left && allUntracked && allAbsentFromDiff);
|
|
471
|
+
const primaryFile = dirtyPaths[0] ?? "";
|
|
472
|
+
const meta = {
|
|
473
|
+
dirtyWorktree: true,
|
|
474
|
+
ref: preservedRef,
|
|
475
|
+
preservedRef,
|
|
476
|
+
paths: dirtyPaths,
|
|
477
|
+
files: dirtyPaths,
|
|
478
|
+
path: primaryFile,
|
|
479
|
+
file: primaryFile,
|
|
480
|
+
bytes: totalBytes,
|
|
481
|
+
byteCount: totalBytes,
|
|
482
|
+
};
|
|
483
|
+
if (left) {
|
|
484
|
+
meta.dirtiedBy = gate;
|
|
485
|
+
meta.culprit = left;
|
|
486
|
+
meta.culpritCommand = left;
|
|
487
|
+
meta.command = left;
|
|
488
|
+
}
|
|
489
|
+
if (isInfra) {
|
|
490
|
+
meta.infra = true;
|
|
491
|
+
meta.classification = "infra";
|
|
492
|
+
}
|
|
493
|
+
if (preservationError) {
|
|
494
|
+
meta.preservationFailed = true;
|
|
495
|
+
meta.preservationError = preservationError;
|
|
496
|
+
// A refusal without a durable snapshot must never enter the ordinary repair/escalation
|
|
497
|
+
// path: the daemon recognizes infra rows as terminal parks for this attempt. This also
|
|
498
|
+
// overrides a chargeable entry/tracked-dirt classification, because retrying could lose
|
|
499
|
+
// the only remaining copy of the worktree state.
|
|
500
|
+
meta.infra = true;
|
|
501
|
+
meta.classification = "infra";
|
|
502
|
+
meta.retryable = false;
|
|
503
|
+
meta.recoveryBlocked = "dirty worktree preservation failed; no recovery ref exists";
|
|
504
|
+
}
|
|
505
|
+
return {
|
|
506
|
+
gate,
|
|
507
|
+
pass: false,
|
|
508
|
+
details: DIRTY_WHY
|
|
509
|
+
+ (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
|
|
510
|
+
+ dirt.text
|
|
511
|
+
+ (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
|
|
512
|
+
meta,
|
|
513
|
+
};
|
|
514
|
+
};
|
|
346
515
|
// The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
|
|
347
516
|
// the last cleanliness check, and naming a culprit this function cannot identify would be a worse
|
|
348
517
|
// record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
|
|
349
|
-
const dirtyRoundRefusal = (gate, dirt) =>
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
518
|
+
const dirtyRoundRefusal = async (gate, dirt) => {
|
|
519
|
+
let preservedRef;
|
|
520
|
+
let preservationError;
|
|
521
|
+
try {
|
|
522
|
+
preservedRef = await preserveWorktree(ctx.worktree);
|
|
523
|
+
}
|
|
524
|
+
catch (error) {
|
|
525
|
+
preservationError = error instanceof Error ? error.message : String(error);
|
|
526
|
+
}
|
|
527
|
+
const dirtyPaths = [];
|
|
528
|
+
let totalBytes = 0;
|
|
529
|
+
for (const { path } of dirt.entries) {
|
|
530
|
+
dirtyPaths.push(path);
|
|
531
|
+
try {
|
|
532
|
+
const st = statSync(join(ctx.worktree, path));
|
|
533
|
+
if (st.isFile()) {
|
|
534
|
+
totalBytes += st.size;
|
|
535
|
+
}
|
|
536
|
+
}
|
|
537
|
+
catch { }
|
|
538
|
+
}
|
|
539
|
+
const primaryFile = dirtyPaths[0] ?? "";
|
|
540
|
+
return {
|
|
541
|
+
gate,
|
|
542
|
+
pass: false,
|
|
543
|
+
details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
|
|
544
|
+
+ `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
|
|
545
|
+
+ `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
|
|
546
|
+
+ `Uncommitted at round end:\n${dirt.text}`
|
|
547
|
+
+ (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
|
|
548
|
+
meta: {
|
|
549
|
+
dirtyWorktree: true,
|
|
550
|
+
dirtyAtRoundEnd: true,
|
|
551
|
+
ref: preservedRef,
|
|
552
|
+
preservedRef,
|
|
553
|
+
paths: dirtyPaths,
|
|
554
|
+
files: dirtyPaths,
|
|
555
|
+
path: primaryFile,
|
|
556
|
+
file: primaryFile,
|
|
557
|
+
bytes: totalBytes,
|
|
558
|
+
byteCount: totalBytes,
|
|
559
|
+
...(preservationError ? {
|
|
560
|
+
preservationFailed: true,
|
|
561
|
+
preservationError,
|
|
562
|
+
infra: true,
|
|
563
|
+
classification: "infra",
|
|
564
|
+
retryable: false,
|
|
565
|
+
recoveryBlocked: "dirty worktree preservation failed; no recovery ref exists",
|
|
566
|
+
} : {}),
|
|
567
|
+
},
|
|
568
|
+
};
|
|
569
|
+
};
|
|
358
570
|
// shell tools vs the shared baseline
|
|
571
|
+
const retryOptions = (identity) => ctx.authorizeInfraRetry
|
|
572
|
+
? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
|
|
573
|
+
: {};
|
|
359
574
|
const runBattery = async (commands, selected, gates = toolGates) => {
|
|
360
575
|
if (!gates.length)
|
|
361
576
|
return;
|
|
362
|
-
if (!v185) {
|
|
363
|
-
// ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
|
|
364
|
-
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
365
|
-
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
366
|
-
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
367
|
-
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
368
|
-
const finish = startMeasurement();
|
|
369
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
|
|
370
|
-
const batch = finish();
|
|
371
|
-
for (const g of gates)
|
|
372
|
-
addMeasurement(g, batch);
|
|
373
|
-
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
374
|
-
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
375
|
-
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
376
|
-
// reported as the red it is: the round already ends, and the command output is the better lead.
|
|
377
|
-
const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
|
|
378
|
-
const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
|
|
379
|
-
for (const r of toolResults) {
|
|
380
|
-
await emitStart(r.gate);
|
|
381
|
-
await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
|
|
382
|
-
}
|
|
383
|
-
return;
|
|
384
|
-
}
|
|
385
577
|
// T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
|
|
386
578
|
// any later tool before anyone reads its verdict.
|
|
387
579
|
for (const g of gates) {
|
|
388
580
|
await emitStart(g);
|
|
389
|
-
const
|
|
581
|
+
const cmd = commands[g];
|
|
582
|
+
let r;
|
|
583
|
+
let cached = false;
|
|
584
|
+
let identity;
|
|
585
|
+
if (cmd !== undefined) {
|
|
586
|
+
identity = await computeVerificationIdentity({
|
|
587
|
+
worktree: ctx.worktree,
|
|
588
|
+
gate: g,
|
|
589
|
+
scope: ctx.verificationScope,
|
|
590
|
+
command: cmd,
|
|
591
|
+
baseline: ctx.baseline,
|
|
592
|
+
selectedSet: g === "test" ? selected : undefined,
|
|
593
|
+
capacity: resolvedCapacity(),
|
|
594
|
+
});
|
|
595
|
+
const hit = verdictStore.get(identity);
|
|
596
|
+
if (hit)
|
|
597
|
+
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
598
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
599
|
+
r = formatReusedRow(hit, identity);
|
|
600
|
+
cached = true;
|
|
601
|
+
await noteReuse(g, r, identity);
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
if (!r) {
|
|
605
|
+
// VL-1: a detected vitest test command is judged by its own invocation-bound report — the
|
|
606
|
+
// stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
|
|
607
|
+
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
608
|
+
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
609
|
+
r = useManifest
|
|
610
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
|
|
611
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
612
|
+
}
|
|
390
613
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
614
|
if (g === "test" && selected)
|
|
392
|
-
selectedDurationMs = spans.get("test")
|
|
615
|
+
selectedDurationMs = spans.get("test")?.durationMs ?? 0;
|
|
393
616
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
394
617
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
395
618
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
396
619
|
// command, the last one included, and fail the gate whose command did it. (A red command needs
|
|
397
620
|
// no check: it already ends the round, and its own output is the truer verdict.)
|
|
398
|
-
if (r.pass && commands[g]) {
|
|
621
|
+
if (!cached && r.pass && commands[g]) {
|
|
399
622
|
const dirt = await dirtyWorktree();
|
|
400
623
|
if (dirt) {
|
|
401
|
-
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
624
|
+
await record(await dirtyRefusal(g, dirt, commands[g]));
|
|
402
625
|
return;
|
|
403
626
|
}
|
|
404
627
|
}
|
|
628
|
+
if (r)
|
|
629
|
+
classifySignalOnlyTest(r);
|
|
630
|
+
if (!cached && identity && r && !isInfraResult(r)) {
|
|
631
|
+
verdictStore.set(identity, { ...r, meta: { ...r.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
632
|
+
}
|
|
405
633
|
if (g === "test" && selected) {
|
|
406
634
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
407
|
-
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
408
635
|
if (!screened.pass)
|
|
409
636
|
await record(screened);
|
|
410
637
|
else {
|
|
@@ -593,24 +820,46 @@ export async function runGates(task, ctx) {
|
|
|
593
820
|
const rv = captured.value;
|
|
594
821
|
if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
|
|
595
822
|
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
|
|
596
|
-
if (
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
823
|
+
if (typeof rv.meta.reviewer === "string") {
|
|
824
|
+
const reviewer = rv.meta.reviewer;
|
|
825
|
+
const cause = String(rv.meta.cause);
|
|
826
|
+
// OBS-1025 add.2: the run-scoped tally; two no-verdicts retire the seat for the rest of the run.
|
|
827
|
+
const causes = rv.meta.noVerdict === true && ctx.reviewNoVerdicts
|
|
828
|
+
? [...(ctx.reviewNoVerdicts.get(reviewer) ?? []), cause] : undefined;
|
|
829
|
+
if (causes)
|
|
830
|
+
ctx.reviewNoVerdicts.set(reviewer, causes);
|
|
831
|
+
// OBS-1039: a seat below the byte floor at the beat is silent — demoted like a zero-byte seat.
|
|
832
|
+
const silent = rv.meta.seatAuthoredBytes === 0 || cause === "silent" || cause === "launch-never-started";
|
|
833
|
+
const twice = (causes?.length ?? 0) >= 2;
|
|
834
|
+
if ((silent || twice) && !ctx.demotedReviewers?.has(reviewer)) {
|
|
835
|
+
ctx.demotedReviewers?.add(reviewer);
|
|
836
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
|
|
837
|
+
payload: { reviewer, cause, seatAuthoredBytes: rv.meta.seatAuthoredBytes ?? 0, ...(twice ? { causes } : {}) }, result: rv });
|
|
838
|
+
}
|
|
601
839
|
}
|
|
602
840
|
}
|
|
603
841
|
return rv;
|
|
604
842
|
};
|
|
843
|
+
// OBS-1025 add.2: seats with two no-verdicts this run are out of the rotation for every pick below.
|
|
844
|
+
const retired = [...(ctx.reviewNoVerdicts ?? [])].filter(([, causes]) => causes.length >= 2).map(([seat]) => seat);
|
|
605
845
|
// RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
|
|
606
846
|
// it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
|
|
607
847
|
// retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
|
|
608
848
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
//
|
|
613
|
-
|
|
849
|
+
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
850
|
+
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
851
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
|
|
852
|
+
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
853
|
+
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
854
|
+
// verdict never enters results; an exhausted pool preserves its cause.
|
|
855
|
+
// OBS-1013 add.3: the re-route LOOPS while seats return no verdict — a closure mismatch or a silent
|
|
856
|
+
// seat is re-routed to a third seat of another vendor, and only an exhausted pool ends the round,
|
|
857
|
+
// as a terminal infra result that keeps the carried materials. Never a worker.
|
|
858
|
+
let retryPrior = [...priorReviewers];
|
|
859
|
+
const routes = [];
|
|
860
|
+
let hop = 0;
|
|
861
|
+
while ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
|
|
862
|
+
hop++;
|
|
614
863
|
const flaked = rv.meta.reviewer;
|
|
615
864
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
616
865
|
if (emptyOutput) {
|
|
@@ -621,29 +870,30 @@ export async function runGates(task, ctx) {
|
|
|
621
870
|
});
|
|
622
871
|
}
|
|
623
872
|
const retryVia = ctx.via
|
|
624
|
-
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) +
|
|
873
|
+
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
|
|
625
874
|
: undefined;
|
|
626
|
-
const priorExclusions = ctx.excludeReviewers ?? [];
|
|
627
875
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
628
876
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
629
877
|
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
630
878
|
// and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
|
|
631
|
-
|
|
879
|
+
retryPrior = [...retryPrior, flaked];
|
|
632
880
|
const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
|
|
633
|
-
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...
|
|
881
|
+
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
|
|
634
882
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
635
|
-
|
|
636
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia,
|
|
883
|
+
exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
884
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
|
|
637
885
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
638
886
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
639
887
|
const route = exclusion === "adapter"
|
|
640
888
|
? `different-adapter retry; excluded flaked adapter ${flakedAdapter}`
|
|
641
889
|
: `same-adapter fallback; excluded flaked channel ${flaked}`;
|
|
890
|
+
const produced = emptyOutput ? "EMPTY output" : rv.meta.cause === "closure-mismatch" ? "a verdict closing no carried fingerprint" : "no parseable verdict";
|
|
891
|
+
// `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep every
|
|
892
|
+
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
893
|
+
routes.push(`review re-route (${route}): ${flaked} produced ${produced}; replaced by ${retried}`);
|
|
642
894
|
rv = {
|
|
643
895
|
...second,
|
|
644
|
-
|
|
645
|
-
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
646
|
-
details: `review re-route (${route}): ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
|
|
896
|
+
details: `${routes.join("\n")}\n${second.details}`,
|
|
647
897
|
meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
|
|
648
898
|
};
|
|
649
899
|
}
|
|
@@ -652,8 +902,15 @@ export async function runGates(task, ctx) {
|
|
|
652
902
|
// floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
|
|
653
903
|
const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
|
|
654
904
|
rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
|
|
905
|
+
break;
|
|
655
906
|
}
|
|
656
907
|
}
|
|
908
|
+
if (rv.meta?.noVerdict === true) {
|
|
909
|
+
// Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
|
|
910
|
+
// row is not a passing review — and the sibling judge result is untouched beside it.
|
|
911
|
+
const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
|
|
912
|
+
rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
|
|
913
|
+
}
|
|
657
914
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
658
915
|
};
|
|
659
916
|
// v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
|
|
@@ -672,10 +929,10 @@ export async function runGates(task, ctx) {
|
|
|
672
929
|
if (entryDirt) {
|
|
673
930
|
addMeasurement(sequence[0], entryMeasurement);
|
|
674
931
|
await emitStart(sequence[0]);
|
|
675
|
-
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
932
|
+
await record(await dirtyRefusal(sequence[0], entryDirt));
|
|
676
933
|
return done();
|
|
677
934
|
}
|
|
678
|
-
if (
|
|
935
|
+
if (await screenBlocks())
|
|
679
936
|
return done();
|
|
680
937
|
await runBattery(ctx.commands);
|
|
681
938
|
if (failed())
|
|
@@ -692,13 +949,33 @@ export async function runGates(task, ctx) {
|
|
|
692
949
|
}
|
|
693
950
|
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
694
951
|
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
695
|
-
|
|
952
|
+
let selected = ctx.selectTests && enabled("test") && ctx.commands.test
|
|
696
953
|
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
697
954
|
: undefined;
|
|
955
|
+
let selectionReason = ctx.selectionReason ?? (selected ? "affected-tests" : "full-suite-fallback");
|
|
956
|
+
if (selected && ctx.requiredRepairTests?.length) {
|
|
957
|
+
const required = ctx.requiredRepairTests;
|
|
958
|
+
const safe = required.every((path) => posix.normalize(path) === path && !path.startsWith("../")
|
|
959
|
+
&& !path.startsWith("/") && TEST_FILE_RE.test(path) && existsSync(join(ctx.worktree, path)));
|
|
960
|
+
const listed = safe ? await shGit(`git ls-files -z -- ${required.map(shq).join(" ")}`, ctx.worktree) : undefined;
|
|
961
|
+
const tracked = new Set(listed?.stdout.split("\0").filter(Boolean));
|
|
962
|
+
if (!listed || listed.code !== 0 || required.some((path) => !tracked.has(path))) {
|
|
963
|
+
selected = undefined;
|
|
964
|
+
selectionReason = "required-repair-test-unavailable";
|
|
965
|
+
}
|
|
966
|
+
else
|
|
967
|
+
selected = [...new Set([...selected, ...required])].sort();
|
|
968
|
+
}
|
|
969
|
+
if (ctx.selectionReason)
|
|
970
|
+
selectionDecision = {
|
|
971
|
+
scope: selected ? "selected" : "full", reason: selected ? selectionReason
|
|
972
|
+
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
|
|
973
|
+
requiredFiles: [...(ctx.requiredRepairTests ?? [])],
|
|
974
|
+
};
|
|
698
975
|
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
|
|
699
976
|
if (failed())
|
|
700
977
|
return done();
|
|
701
|
-
if (
|
|
978
|
+
if (enabled("acceptance") || enabled("review")) {
|
|
702
979
|
// Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
|
|
703
980
|
// unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
|
|
704
981
|
// verdict: each gets the same commit and the same brief it always got, and neither promise is
|
|
@@ -719,25 +996,19 @@ export async function runGates(task, ctx) {
|
|
|
719
996
|
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
720
997
|
const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
|
|
721
998
|
const reviewed = reviewing?.then((outcome) => record(outcome));
|
|
722
|
-
|
|
999
|
+
if (executionSignal()) {
|
|
1000
|
+
// A cancelled sibling still owns a process until it unwinds; do not settle the task early.
|
|
1001
|
+
const settled = await Promise.allSettled([judged, reviewed]);
|
|
1002
|
+
const rejected = settled.find((result) => result.status === "rejected");
|
|
1003
|
+
if (rejected?.status === "rejected")
|
|
1004
|
+
throw rejected.reason;
|
|
1005
|
+
executionSignal()?.throwIfAborted();
|
|
1006
|
+
}
|
|
1007
|
+
else
|
|
1008
|
+
await Promise.all([judged, reviewed]);
|
|
723
1009
|
if (failed())
|
|
724
1010
|
return done();
|
|
725
1011
|
}
|
|
726
|
-
else if (!v185) {
|
|
727
|
-
// Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
|
|
728
|
-
if (enabled("acceptance")) {
|
|
729
|
-
await emitStart("acceptance");
|
|
730
|
-
const judged = await measure("acceptance", runAcceptance);
|
|
731
|
-
await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
|
|
732
|
-
if (failed())
|
|
733
|
-
return done();
|
|
734
|
-
}
|
|
735
|
-
if (enabled("review")) {
|
|
736
|
-
await emitStart("review");
|
|
737
|
-
await record(await measure("review", runReview));
|
|
738
|
-
}
|
|
739
|
-
return done();
|
|
740
|
-
}
|
|
741
1012
|
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
742
1013
|
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
743
1014
|
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
|
|
@@ -748,11 +1019,45 @@ export async function runGates(task, ctx) {
|
|
|
748
1019
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
749
1020
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
750
1021
|
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
1022
|
+
// VL-1: the merge-candidate's manifest is the FULL set — a full suite whose report lacks one
|
|
1023
|
+
// manifest file never reaches the pass branch below, so a selected-only green can never merge.
|
|
1024
|
+
let full;
|
|
1025
|
+
let cached = false;
|
|
1026
|
+
let identity;
|
|
1027
|
+
if (ctx.commands.test !== undefined) {
|
|
1028
|
+
identity = await computeVerificationIdentity({
|
|
1029
|
+
worktree: ctx.worktree,
|
|
1030
|
+
gate: "test",
|
|
1031
|
+
scope: ctx.verificationScope,
|
|
1032
|
+
command: ctx.commands.test,
|
|
1033
|
+
baseline: ctx.baseline,
|
|
1034
|
+
selectedSet: undefined,
|
|
1035
|
+
capacity: resolvedCapacity(),
|
|
1036
|
+
});
|
|
1037
|
+
const hit = verdictStore.get(identity);
|
|
1038
|
+
if (hit)
|
|
1039
|
+
classifySignalOnlyTest(hit);
|
|
1040
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
1041
|
+
full = formatReusedRow(hit, identity);
|
|
1042
|
+
cached = true;
|
|
1043
|
+
await noteReuse("test", full, identity);
|
|
1044
|
+
}
|
|
1045
|
+
}
|
|
1046
|
+
if (!full) {
|
|
1047
|
+
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
1048
|
+
full = fullUsesManifest
|
|
1049
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
|
|
1050
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
|
|
1051
|
+
}
|
|
1052
|
+
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
1053
|
+
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
1054
|
+
if (full)
|
|
1055
|
+
classifySignalOnlyTest(full);
|
|
1056
|
+
if (!cached && identity && full && !dirt && !isInfraResult(full)) {
|
|
1057
|
+
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
1058
|
+
}
|
|
754
1059
|
const merged = withTelemetry(dirt
|
|
755
|
-
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
1060
|
+
? await dirtyRefusal("test", dirt, ctx.commands.test)
|
|
756
1061
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
757
1062
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
758
1063
|
heldTest = undefined;
|