tickmarkr 2.5.5 → 2.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +69 -8
- package/dist/cli/commands/doctor.d.ts +10 -0
- package/dist/cli/commands/fleet.js +4 -0
- package/dist/cli/commands/plan.js +20 -4
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +108 -85
- package/dist/config/config.d.ts +11 -0
- package/dist/config/config.js +21 -12
- package/dist/config/fleet-overlay.js +55 -17
- package/dist/drivers/index.js +2 -1
- package/dist/drivers/orca.d.ts +21 -1
- package/dist/drivers/orca.js +209 -27
- package/dist/gates/baseline.d.ts +21 -5
- package/dist/gates/baseline.js +67 -17
- package/dist/gates/cache.d.ts +14 -13
- package/dist/gates/cache.js +17 -5
- package/dist/gates/llm.d.ts +3 -0
- package/dist/gates/llm.js +11 -0
- package/dist/gates/review.d.ts +28 -3
- package/dist/gates/review.js +118 -15
- package/dist/gates/run-gates.d.ts +10 -2
- package/dist/gates/run-gates.js +362 -108
- package/dist/gates/test-manifest.d.ts +33 -1
- package/dist/gates/test-manifest.js +132 -40
- package/dist/gates/test-reporter.js +20 -7
- package/dist/graph/graph.d.ts +4 -0
- package/dist/graph/graph.js +50 -1
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/consult.js +5 -4
- package/dist/run/daemon.d.ts +11 -0
- package/dist/run/daemon.js +663 -128
- package/dist/run/execution-budget.d.ts +25 -0
- package/dist/run/execution-budget.js +142 -0
- package/dist/run/git.d.ts +46 -1
- package/dist/run/git.js +149 -12
- package/dist/run/journal.d.ts +23 -5
- package/dist/run/journal.js +98 -25
- package/dist/run/lease.d.ts +44 -0
- package/dist/run/lease.js +226 -3
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/run/recovery.d.ts +8 -0
- package/dist/run/recovery.js +25 -0
- package/dist/run/repair-selection.d.ts +12 -0
- package/dist/run/repair-selection.js +56 -0
- package/dist/run/stall.d.ts +6 -1
- package/dist/run/stall.js +60 -3
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/dist/tui/ink/fleet-app.d.ts +10 -2
- package/dist/tui/ink/fleet-app.js +33 -15
- package/package.json +2 -2
- package/skills/tickmarkr-overseer/SKILL.md +55 -3
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
2
3
|
import { loadavg, tmpdir } from "node:os";
|
|
3
4
|
import { join, posix } from "node:path";
|
|
4
5
|
import { channelKey, shq } from "../adapters/types.js";
|
|
@@ -6,17 +7,19 @@ import { TIER_RANK } from "../config/config.js";
|
|
|
6
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
8
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
8
9
|
import { acceptanceGate } from "./acceptance.js";
|
|
9
|
-
import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
|
|
10
|
+
import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
|
|
10
11
|
import { evidenceGate } from "./evidence.js";
|
|
11
12
|
import { captureLlmOutput } from "./llm.js";
|
|
12
13
|
import { disallowedBy } from "../route/preference.js";
|
|
13
14
|
import { marginalCostRank } from "../route/router.js";
|
|
14
|
-
import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
|
+
import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
16
|
import { scopeGate } from "./scope.js";
|
|
16
17
|
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
-
import {
|
|
18
|
+
import { executionSignal } from "../run/execution-budget.js";
|
|
19
|
+
import { failureDisposition } from "../run/recovery.js";
|
|
20
|
+
import { preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
18
21
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
19
|
-
import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
22
|
+
import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
20
23
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
21
24
|
let loadProvider = productionLoadProvider;
|
|
22
25
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
@@ -96,6 +99,23 @@ async function captureLlmDispatches(adapters, run) {
|
|
|
96
99
|
throw error;
|
|
97
100
|
}
|
|
98
101
|
}
|
|
102
|
+
/** Parse `git status --porcelain --untracked-files=all -z` output. A rename/copy carries its
|
|
103
|
+
* ORIGINAL path in a second NUL field immediately after the current one; skip it — dirt only
|
|
104
|
+
* cares about paths that exist in the worktree now. */
|
|
105
|
+
function parseStatusZ(stdout) {
|
|
106
|
+
const fields = stdout.split("\0");
|
|
107
|
+
const entries = [];
|
|
108
|
+
for (let i = 0; i < fields.length; i++) {
|
|
109
|
+
const rec = fields[i];
|
|
110
|
+
if (!rec)
|
|
111
|
+
continue;
|
|
112
|
+
const status = rec.slice(0, 2);
|
|
113
|
+
entries.push({ status, path: rec.slice(3) });
|
|
114
|
+
if (status.includes("R") || status.includes("C"))
|
|
115
|
+
i++; // consume the paired original-path field
|
|
116
|
+
}
|
|
117
|
+
return entries;
|
|
118
|
+
}
|
|
99
119
|
const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
|
|
100
120
|
// relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
|
|
101
121
|
const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
|
|
@@ -195,7 +215,7 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
195
215
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
196
216
|
}
|
|
197
217
|
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
198
|
-
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
|
|
218
|
+
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
|
|
199
219
|
const entry = baseline.commands.test;
|
|
200
220
|
const outcome = await evaluateManifestedTest(cmd, worktree, {
|
|
201
221
|
baselineDurations: entry?.fileDurations,
|
|
@@ -204,6 +224,19 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
204
224
|
artifactDir,
|
|
205
225
|
});
|
|
206
226
|
const reportPath = outcome.reportPath;
|
|
227
|
+
if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
|
|
228
|
+
&& outcome.meta?.retryable !== false) {
|
|
229
|
+
const waitedMs = await waitForCalmWindow(executionSignal());
|
|
230
|
+
if (!calmWindowReady())
|
|
231
|
+
return { gate: "test", pass: false, details: outcome.details,
|
|
232
|
+
meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
|
|
233
|
+
if (!retry.authorizeRetry("infra")) {
|
|
234
|
+
return { gate: "test", pass: false, details: outcome.details,
|
|
235
|
+
meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
|
|
236
|
+
}
|
|
237
|
+
const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
|
|
238
|
+
return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
239
|
+
}
|
|
207
240
|
return {
|
|
208
241
|
gate: "test",
|
|
209
242
|
pass: outcome.pass,
|
|
@@ -226,17 +259,65 @@ function classifySignalOnlyTest(g) {
|
|
|
226
259
|
}
|
|
227
260
|
export async function runGates(task, ctx) {
|
|
228
261
|
const results = [];
|
|
262
|
+
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
263
|
+
// allocates a new invocation, including retries whose local spawn counter starts at one again.
|
|
264
|
+
let currentBuild;
|
|
265
|
+
let buildStarted = false;
|
|
266
|
+
let receiptNotes = Promise.resolve();
|
|
267
|
+
const beginBuild = () => {
|
|
268
|
+
buildStarted = false;
|
|
269
|
+
currentBuild = { ...(ctx.buildReceiptIdentity ?? {
|
|
270
|
+
runId: ctx.artifactDir ?? "standalone", taskId: task.id, attempt: 0, gateRound: 0,
|
|
271
|
+
}), invocation: randomUUID() };
|
|
272
|
+
return { ...currentBuild };
|
|
273
|
+
};
|
|
274
|
+
const buildReceipt = (receipt, reason) => {
|
|
275
|
+
const matches = currentBuild !== undefined && receipt.attribution !== undefined
|
|
276
|
+
&& Object.keys(currentBuild)
|
|
277
|
+
.every((key) => receipt.attribution[key] === currentBuild[key]);
|
|
278
|
+
const attributed = matches && (receipt.outcome === "started" || !receipt.confirmedStart || buildStarted);
|
|
279
|
+
if (attributed)
|
|
280
|
+
buildStarted = receipt.outcome === "started";
|
|
281
|
+
const { attribution, ...observation } = receipt;
|
|
282
|
+
const payload = attributed ? { ...receipt, gate: "build", ...(reason ? { reason, freshBuildRan: false } : {}) } : {
|
|
283
|
+
...observation, gate: "build", attributionStatus: "unattributed",
|
|
284
|
+
reportedAttribution: attribution,
|
|
285
|
+
};
|
|
286
|
+
// Shell observers are synchronous. Serialize asynchronous note sinks and drain before the
|
|
287
|
+
// verdict (also on cancellation), without letting an observer change execution or retry policy.
|
|
288
|
+
receiptNotes = receiptNotes.then(async () => {
|
|
289
|
+
await ctx.onGate?.({ phase: "note", gate: "build", name: "build-receipt", payload });
|
|
290
|
+
}).catch(() => { });
|
|
291
|
+
};
|
|
292
|
+
const noBuild = async (outcome, reason) => {
|
|
293
|
+
buildReceipt({ outcome, confirmedStart: false, attribution: beginBuild() }, reason);
|
|
294
|
+
await receiptNotes;
|
|
295
|
+
};
|
|
296
|
+
let selectionDecision;
|
|
229
297
|
let commits = [];
|
|
230
298
|
const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
|
|
231
299
|
const verdictStore = getVerdictStore(stateDir);
|
|
300
|
+
// R41: the verification protocol and the EFFECTIVE npm lifecycle policy measured for THIS
|
|
301
|
+
// checkout — the policy its runner children receive (an explicit process export, else npm's own
|
|
302
|
+
// resolved config for this worktree, project npmrc included). Every result leaves through
|
|
303
|
+
// withTelemetry carrying it, so the daemon's gate-result row records what the gate measured under
|
|
304
|
+
// rather than a session-wide value resolved somewhere else.
|
|
305
|
+
const verification = verificationProtocol(process.env, ctx.worktree);
|
|
232
306
|
// VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
|
|
233
307
|
// the ledger names the reuse and the identity even where the gate-result row's details must stay
|
|
234
308
|
// the fresh verdict's (see formatReusedRow).
|
|
235
309
|
const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
|
|
310
|
+
// OBS-1055: on a recheck a cached red is the answer the operator just said was wrongly given; it is
|
|
311
|
+
// journaled as discarded and the gate runs. Returns true when the hit must NOT be reused.
|
|
312
|
+
const discardCachedRed = async (gate, hit) => {
|
|
313
|
+
if (!ctx.recheck || hit.pass)
|
|
314
|
+
return false;
|
|
315
|
+
await ctx.onGate?.({ phase: "note", gate, name: "recheck-rerun", payload: { gate, reason: "cached-red-discarded" }, result: hit });
|
|
316
|
+
return true;
|
|
317
|
+
};
|
|
236
318
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
237
319
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
238
320
|
const failed = () => results.some((r) => !r.pass);
|
|
239
|
-
const v185 = ctx.pipeline === "v185";
|
|
240
321
|
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
241
322
|
// round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
|
|
242
323
|
// exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
|
|
@@ -288,13 +369,17 @@ export async function runGates(task, ctx) {
|
|
|
288
369
|
// path that forgets to measure is visibly missing its telemetry rather than carrying a fabricated
|
|
289
370
|
// zero. The daemon lifts these off `meta` onto the gate-result row (src/run/daemon.ts).
|
|
290
371
|
const withTelemetry = (result) => {
|
|
372
|
+
if (result.gate === "test" && selectionDecision) {
|
|
373
|
+
result = { ...result, meta: { ...result.meta, selectionDecision } };
|
|
374
|
+
}
|
|
291
375
|
const span = spans.get(result.gate);
|
|
292
376
|
if (!span)
|
|
293
|
-
return result;
|
|
377
|
+
return { ...result, meta: { ...result.meta, verification } };
|
|
294
378
|
return {
|
|
295
379
|
...result,
|
|
296
380
|
meta: {
|
|
297
381
|
...result.meta,
|
|
382
|
+
verification,
|
|
298
383
|
...span,
|
|
299
384
|
...(result.gate === "test" && selectedDurationMs !== undefined ? { selectedDurationMs } : {}),
|
|
300
385
|
...(result.gate === "test" && fullDurationMs !== undefined ? { fullDurationMs } : {}),
|
|
@@ -335,7 +420,7 @@ export async function runGates(task, ctx) {
|
|
|
335
420
|
if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
|
|
336
421
|
const dirt = await dirtyWorktree();
|
|
337
422
|
if (dirt) {
|
|
338
|
-
const refusal = withTelemetry(dirtyRoundRefusal(last.gate, dirt));
|
|
423
|
+
const refusal = withTelemetry(await dirtyRoundRefusal(last.gate, dirt));
|
|
339
424
|
results[results.indexOf(last)] = refusal;
|
|
340
425
|
sorted[sorted.length - 1] = refusal;
|
|
341
426
|
await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
|
|
@@ -360,66 +445,178 @@ export async function runGates(task, ctx) {
|
|
|
360
445
|
* exempt: an untracked source file is uncommitted work by every reading git offers.
|
|
361
446
|
*/
|
|
362
447
|
const dirtyWorktree = async () => {
|
|
363
|
-
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
|
|
448
|
+
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain --untracked-files=all -z", ctx.worktree);
|
|
364
449
|
if (r.code !== 0)
|
|
365
|
-
return `git status failed (exit ${r.code}) — the worktree cannot be proven clean
|
|
366
|
-
const entries = r.stdout
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
return entries.length ? entries.join("\n") : undefined;
|
|
450
|
+
return { text: `git status failed (exit ${r.code}) — the worktree cannot be proven clean`, entries: [] };
|
|
451
|
+
const entries = parseStatusZ(r.stdout).filter((e) => !/^\.tickmarkr-[^/]*$/.test(e.path));
|
|
452
|
+
if (!entries.length)
|
|
453
|
+
return undefined;
|
|
454
|
+
return { text: entries.map((e) => `${e.status} ${e.path}`).join("\n"), entries };
|
|
371
455
|
};
|
|
372
456
|
const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
|
|
373
457
|
+ `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
|
|
374
458
|
+ `and never merged (and the committed diff would never be run)`;
|
|
375
459
|
// `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
|
|
376
|
-
const dirtyRefusal = (gate, dirt, left) =>
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
460
|
+
const dirtyRefusal = async (gate, dirt, left) => {
|
|
461
|
+
let preservedRef;
|
|
462
|
+
let preservationError;
|
|
463
|
+
try {
|
|
464
|
+
preservedRef = await preserveWorktree(ctx.worktree);
|
|
465
|
+
}
|
|
466
|
+
catch (error) {
|
|
467
|
+
// Never masks the refusal, but never pretends a snapshot exists either — surfaced below.
|
|
468
|
+
preservationError = error instanceof Error ? error.message : String(error);
|
|
469
|
+
}
|
|
470
|
+
const dirtyPaths = [];
|
|
471
|
+
let allUntracked = true;
|
|
472
|
+
let totalBytes = 0;
|
|
473
|
+
for (const { status, path } of dirt.entries) {
|
|
474
|
+
dirtyPaths.push(path);
|
|
475
|
+
if (status !== "??") {
|
|
476
|
+
allUntracked = false;
|
|
477
|
+
}
|
|
478
|
+
try {
|
|
479
|
+
const st = statSync(join(ctx.worktree, path));
|
|
480
|
+
if (st.isFile()) {
|
|
481
|
+
totalBytes += st.size;
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
catch {
|
|
485
|
+
// ignore deleted or unreadable
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
let allAbsentFromDiff = true;
|
|
489
|
+
if (ctx.baseRef) {
|
|
490
|
+
try {
|
|
491
|
+
const diffOut = await shGit(`git diff --name-only -z ${shq(ctx.baseRef)}..HEAD`, ctx.worktree);
|
|
492
|
+
if (diffOut.code === 0) {
|
|
493
|
+
// Paths in Git's newline output are C-quoted, which is not a lossless path encoding
|
|
494
|
+
// (and must never be parsed as JSON). `-z` lets this comparison retain arbitrary
|
|
495
|
+
// filenames exactly, including whitespace and non-ASCII bytes.
|
|
496
|
+
const touched = new Set(diffOut.stdout.split("\0").filter(Boolean));
|
|
497
|
+
for (const p of dirtyPaths) {
|
|
498
|
+
if (touched.has(p)) {
|
|
499
|
+
allAbsentFromDiff = false;
|
|
500
|
+
break;
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
else {
|
|
505
|
+
// An unreadable committed diff cannot prove the litter is unrelated to the worker.
|
|
506
|
+
allAbsentFromDiff = false;
|
|
507
|
+
}
|
|
508
|
+
}
|
|
509
|
+
catch {
|
|
510
|
+
allAbsentFromDiff = false;
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
const isInfra = Boolean(left && allUntracked && allAbsentFromDiff);
|
|
514
|
+
const primaryFile = dirtyPaths[0] ?? "";
|
|
515
|
+
const meta = {
|
|
516
|
+
dirtyWorktree: true,
|
|
517
|
+
ref: preservedRef,
|
|
518
|
+
preservedRef,
|
|
519
|
+
paths: dirtyPaths,
|
|
520
|
+
files: dirtyPaths,
|
|
521
|
+
path: primaryFile,
|
|
522
|
+
file: primaryFile,
|
|
523
|
+
bytes: totalBytes,
|
|
524
|
+
byteCount: totalBytes,
|
|
525
|
+
};
|
|
526
|
+
if (left) {
|
|
527
|
+
meta.dirtiedBy = gate;
|
|
528
|
+
meta.culprit = left;
|
|
529
|
+
meta.culpritCommand = left;
|
|
530
|
+
meta.command = left;
|
|
531
|
+
}
|
|
532
|
+
if (isInfra) {
|
|
533
|
+
meta.infra = true;
|
|
534
|
+
meta.classification = "infra";
|
|
535
|
+
}
|
|
536
|
+
if (preservationError) {
|
|
537
|
+
meta.preservationFailed = true;
|
|
538
|
+
meta.preservationError = preservationError;
|
|
539
|
+
// A refusal without a durable snapshot must never enter the ordinary repair/escalation
|
|
540
|
+
// path: the daemon recognizes infra rows as terminal parks for this attempt. This also
|
|
541
|
+
// overrides a chargeable entry/tracked-dirt classification, because retrying could lose
|
|
542
|
+
// the only remaining copy of the worktree state.
|
|
543
|
+
meta.infra = true;
|
|
544
|
+
meta.classification = "infra";
|
|
545
|
+
meta.retryable = false;
|
|
546
|
+
meta.recoveryBlocked = "dirty worktree preservation failed; no recovery ref exists";
|
|
547
|
+
}
|
|
548
|
+
return {
|
|
549
|
+
gate,
|
|
550
|
+
pass: false,
|
|
551
|
+
details: DIRTY_WHY
|
|
552
|
+
+ (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
|
|
553
|
+
+ dirt.text
|
|
554
|
+
+ (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
|
|
555
|
+
meta,
|
|
556
|
+
};
|
|
557
|
+
};
|
|
384
558
|
// The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
|
|
385
559
|
// the last cleanliness check, and naming a culprit this function cannot identify would be a worse
|
|
386
560
|
// record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
|
|
387
|
-
const dirtyRoundRefusal = (gate, dirt) =>
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
561
|
+
const dirtyRoundRefusal = async (gate, dirt) => {
|
|
562
|
+
let preservedRef;
|
|
563
|
+
let preservationError;
|
|
564
|
+
try {
|
|
565
|
+
preservedRef = await preserveWorktree(ctx.worktree);
|
|
566
|
+
}
|
|
567
|
+
catch (error) {
|
|
568
|
+
preservationError = error instanceof Error ? error.message : String(error);
|
|
569
|
+
}
|
|
570
|
+
const dirtyPaths = [];
|
|
571
|
+
let totalBytes = 0;
|
|
572
|
+
for (const { path } of dirt.entries) {
|
|
573
|
+
dirtyPaths.push(path);
|
|
574
|
+
try {
|
|
575
|
+
const st = statSync(join(ctx.worktree, path));
|
|
576
|
+
if (st.isFile()) {
|
|
577
|
+
totalBytes += st.size;
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
catch { }
|
|
581
|
+
}
|
|
582
|
+
const primaryFile = dirtyPaths[0] ?? "";
|
|
583
|
+
return {
|
|
584
|
+
gate,
|
|
585
|
+
pass: false,
|
|
586
|
+
details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
|
|
587
|
+
+ `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
|
|
588
|
+
+ `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
|
|
589
|
+
+ `Uncommitted at round end:\n${dirt.text}`
|
|
590
|
+
+ (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
|
|
591
|
+
meta: {
|
|
592
|
+
dirtyWorktree: true,
|
|
593
|
+
dirtyAtRoundEnd: true,
|
|
594
|
+
ref: preservedRef,
|
|
595
|
+
preservedRef,
|
|
596
|
+
paths: dirtyPaths,
|
|
597
|
+
files: dirtyPaths,
|
|
598
|
+
path: primaryFile,
|
|
599
|
+
file: primaryFile,
|
|
600
|
+
bytes: totalBytes,
|
|
601
|
+
byteCount: totalBytes,
|
|
602
|
+
...(preservationError ? {
|
|
603
|
+
preservationFailed: true,
|
|
604
|
+
preservationError,
|
|
605
|
+
infra: true,
|
|
606
|
+
classification: "infra",
|
|
607
|
+
retryable: false,
|
|
608
|
+
recoveryBlocked: "dirty worktree preservation failed; no recovery ref exists",
|
|
609
|
+
} : {}),
|
|
610
|
+
},
|
|
611
|
+
};
|
|
612
|
+
};
|
|
396
613
|
// shell tools vs the shared baseline
|
|
614
|
+
const retryOptions = (identity) => ctx.authorizeInfraRetry
|
|
615
|
+
? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
|
|
616
|
+
: {};
|
|
397
617
|
const runBattery = async (commands, selected, gates = toolGates) => {
|
|
398
618
|
if (!gates.length)
|
|
399
619
|
return;
|
|
400
|
-
if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
|
|
401
|
-
// ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
|
|
402
|
-
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
403
|
-
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
404
|
-
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
405
|
-
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
406
|
-
const finish = startMeasurement();
|
|
407
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
|
|
408
|
-
const batch = finish();
|
|
409
|
-
for (const g of gates)
|
|
410
|
-
addMeasurement(g, batch);
|
|
411
|
-
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
412
|
-
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
413
|
-
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
414
|
-
// reported as the red it is: the round already ends, and the command output is the better lead.
|
|
415
|
-
const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
|
|
416
|
-
const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
|
|
417
|
-
for (const r of toolResults) {
|
|
418
|
-
await emitStart(r.gate);
|
|
419
|
-
await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
|
|
420
|
-
}
|
|
421
|
-
return;
|
|
422
|
-
}
|
|
423
620
|
// T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
|
|
424
621
|
// any later tool before anyone reads its verdict.
|
|
425
622
|
for (const g of gates) {
|
|
@@ -441,20 +638,30 @@ export async function runGates(task, ctx) {
|
|
|
441
638
|
const hit = verdictStore.get(identity);
|
|
442
639
|
if (hit)
|
|
443
640
|
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
444
|
-
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
641
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
642
|
+
&& !(await discardCachedRed(g, hit))) {
|
|
445
643
|
r = formatReusedRow(hit, identity);
|
|
446
644
|
cached = true;
|
|
645
|
+
if (g === "build")
|
|
646
|
+
await noBuild("reused-result", "verdict reused; no fresh build ran in this invocation");
|
|
447
647
|
await noteReuse(g, r, identity);
|
|
448
648
|
}
|
|
449
649
|
}
|
|
450
650
|
if (!r) {
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
651
|
+
if (g === "build" && !cmd)
|
|
652
|
+
await noBuild("skipped", "no build command detected");
|
|
653
|
+
try {
|
|
654
|
+
// VL-1: a detected vitest test command is judged by its own invocation-bound report — the
|
|
655
|
+
// stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
|
|
656
|
+
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
657
|
+
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
658
|
+
r = useManifest
|
|
659
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
|
|
660
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
661
|
+
}
|
|
662
|
+
finally {
|
|
663
|
+
await receiptNotes;
|
|
664
|
+
}
|
|
458
665
|
}
|
|
459
666
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
460
667
|
if (g === "test" && selected)
|
|
@@ -467,7 +674,7 @@ export async function runGates(task, ctx) {
|
|
|
467
674
|
if (!cached && r.pass && commands[g]) {
|
|
468
675
|
const dirt = await dirtyWorktree();
|
|
469
676
|
if (dirt) {
|
|
470
|
-
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
677
|
+
await record(await dirtyRefusal(g, dirt, commands[g]));
|
|
471
678
|
return;
|
|
472
679
|
}
|
|
473
680
|
}
|
|
@@ -666,24 +873,46 @@ export async function runGates(task, ctx) {
|
|
|
666
873
|
const rv = captured.value;
|
|
667
874
|
if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
|
|
668
875
|
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
|
|
669
|
-
if (
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
876
|
+
if (typeof rv.meta.reviewer === "string") {
|
|
877
|
+
const reviewer = rv.meta.reviewer;
|
|
878
|
+
const cause = String(rv.meta.cause);
|
|
879
|
+
// OBS-1025 add.2: the run-scoped tally; two no-verdicts retire the seat for the rest of the run.
|
|
880
|
+
const causes = rv.meta.noVerdict === true && ctx.reviewNoVerdicts
|
|
881
|
+
? [...(ctx.reviewNoVerdicts.get(reviewer) ?? []), cause] : undefined;
|
|
882
|
+
if (causes)
|
|
883
|
+
ctx.reviewNoVerdicts.set(reviewer, causes);
|
|
884
|
+
// OBS-1039: a seat below the byte floor at the beat is silent — demoted like a zero-byte seat.
|
|
885
|
+
const silent = rv.meta.seatAuthoredBytes === 0 || cause === "silent" || cause === "launch-never-started";
|
|
886
|
+
const twice = (causes?.length ?? 0) >= 2;
|
|
887
|
+
if ((silent || twice) && !ctx.demotedReviewers?.has(reviewer)) {
|
|
888
|
+
ctx.demotedReviewers?.add(reviewer);
|
|
889
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
|
|
890
|
+
payload: { reviewer, cause, seatAuthoredBytes: rv.meta.seatAuthoredBytes ?? 0, ...(twice ? { causes } : {}) }, result: rv });
|
|
891
|
+
}
|
|
674
892
|
}
|
|
675
893
|
}
|
|
676
894
|
return rv;
|
|
677
895
|
};
|
|
896
|
+
// OBS-1025 add.2: seats with two no-verdicts this run are out of the rotation for every pick below.
|
|
897
|
+
const retired = [...(ctx.reviewNoVerdicts ?? [])].filter(([, causes]) => causes.length >= 2).map(([seat]) => seat);
|
|
678
898
|
// RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
|
|
679
899
|
// it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
|
|
680
900
|
// retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
|
|
681
901
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
//
|
|
686
|
-
|
|
902
|
+
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
903
|
+
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
904
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
|
|
905
|
+
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
906
|
+
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
907
|
+
// verdict never enters results; an exhausted pool preserves its cause.
|
|
908
|
+
// OBS-1013 add.3: the re-route LOOPS while seats return no verdict — a closure mismatch or a silent
|
|
909
|
+
// seat is re-routed to a third seat of another vendor, and only an exhausted pool ends the round,
|
|
910
|
+
// as a terminal infra result that keeps the carried materials. Never a worker.
|
|
911
|
+
let retryPrior = [...priorReviewers];
|
|
912
|
+
const routes = [];
|
|
913
|
+
let hop = 0;
|
|
914
|
+
while ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
|
|
915
|
+
hop++;
|
|
687
916
|
const flaked = rv.meta.reviewer;
|
|
688
917
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
689
918
|
if (emptyOutput) {
|
|
@@ -694,29 +923,30 @@ export async function runGates(task, ctx) {
|
|
|
694
923
|
});
|
|
695
924
|
}
|
|
696
925
|
const retryVia = ctx.via
|
|
697
|
-
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) +
|
|
926
|
+
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
|
|
698
927
|
: undefined;
|
|
699
|
-
const priorExclusions = ctx.excludeReviewers ?? [];
|
|
700
928
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
701
929
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
702
930
|
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
703
931
|
// and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
|
|
704
|
-
|
|
932
|
+
retryPrior = [...retryPrior, flaked];
|
|
705
933
|
const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
|
|
706
|
-
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...
|
|
934
|
+
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
|
|
707
935
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
708
|
-
|
|
709
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia,
|
|
936
|
+
exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
937
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
|
|
710
938
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
711
939
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
712
940
|
const route = exclusion === "adapter"
|
|
713
941
|
? `different-adapter retry; excluded flaked adapter ${flakedAdapter}`
|
|
714
942
|
: `same-adapter fallback; excluded flaked channel ${flaked}`;
|
|
943
|
+
const produced = emptyOutput ? "EMPTY output" : rv.meta.cause === "closure-mismatch" ? "a verdict closing no carried fingerprint" : "no parseable verdict";
|
|
944
|
+
// `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep every
|
|
945
|
+
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
946
|
+
routes.push(`review re-route (${route}): ${flaked} produced ${produced}; replaced by ${retried}`);
|
|
715
947
|
rv = {
|
|
716
948
|
...second,
|
|
717
|
-
|
|
718
|
-
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
719
|
-
details: `review re-route (${route}): ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
|
|
949
|
+
details: `${routes.join("\n")}\n${second.details}`,
|
|
720
950
|
meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
|
|
721
951
|
};
|
|
722
952
|
}
|
|
@@ -725,8 +955,15 @@ export async function runGates(task, ctx) {
|
|
|
725
955
|
// floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
|
|
726
956
|
const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
|
|
727
957
|
rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
|
|
958
|
+
break;
|
|
728
959
|
}
|
|
729
960
|
}
|
|
961
|
+
if (rv.meta?.noVerdict === true) {
|
|
962
|
+
// Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
|
|
963
|
+
// row is not a passing review — and the sibling judge result is untouched beside it.
|
|
964
|
+
const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
|
|
965
|
+
rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
|
|
966
|
+
}
|
|
730
967
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
731
968
|
};
|
|
732
969
|
// v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
|
|
@@ -745,10 +982,12 @@ export async function runGates(task, ctx) {
|
|
|
745
982
|
if (entryDirt) {
|
|
746
983
|
addMeasurement(sequence[0], entryMeasurement);
|
|
747
984
|
await emitStart(sequence[0]);
|
|
748
|
-
|
|
985
|
+
if (enabled("build"))
|
|
986
|
+
await noBuild("refused", "dirty worktree; no build start confirmed");
|
|
987
|
+
await record(await dirtyRefusal(sequence[0], entryDirt));
|
|
749
988
|
return done();
|
|
750
989
|
}
|
|
751
|
-
if (
|
|
990
|
+
if (await screenBlocks())
|
|
752
991
|
return done();
|
|
753
992
|
await runBattery(ctx.commands);
|
|
754
993
|
if (failed())
|
|
@@ -765,13 +1004,33 @@ export async function runGates(task, ctx) {
|
|
|
765
1004
|
}
|
|
766
1005
|
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
767
1006
|
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
768
|
-
|
|
1007
|
+
let selected = ctx.selectTests && enabled("test") && ctx.commands.test
|
|
769
1008
|
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
770
1009
|
: undefined;
|
|
1010
|
+
let selectionReason = ctx.selectionReason ?? (selected ? "affected-tests" : "full-suite-fallback");
|
|
1011
|
+
if (selected && ctx.requiredRepairTests?.length) {
|
|
1012
|
+
const required = ctx.requiredRepairTests;
|
|
1013
|
+
const safe = required.every((path) => posix.normalize(path) === path && !path.startsWith("../")
|
|
1014
|
+
&& !path.startsWith("/") && TEST_FILE_RE.test(path) && existsSync(join(ctx.worktree, path)));
|
|
1015
|
+
const listed = safe ? await shGit(`git ls-files -z -- ${required.map(shq).join(" ")}`, ctx.worktree) : undefined;
|
|
1016
|
+
const tracked = new Set(listed?.stdout.split("\0").filter(Boolean));
|
|
1017
|
+
if (!listed || listed.code !== 0 || required.some((path) => !tracked.has(path))) {
|
|
1018
|
+
selected = undefined;
|
|
1019
|
+
selectionReason = "required-repair-test-unavailable";
|
|
1020
|
+
}
|
|
1021
|
+
else
|
|
1022
|
+
selected = [...new Set([...selected, ...required])].sort();
|
|
1023
|
+
}
|
|
1024
|
+
if (ctx.selectionReason)
|
|
1025
|
+
selectionDecision = {
|
|
1026
|
+
scope: selected ? "selected" : "full", reason: selected ? selectionReason
|
|
1027
|
+
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
|
|
1028
|
+
requiredFiles: [...(ctx.requiredRepairTests ?? [])],
|
|
1029
|
+
};
|
|
771
1030
|
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
|
|
772
1031
|
if (failed())
|
|
773
1032
|
return done();
|
|
774
|
-
if (
|
|
1033
|
+
if (enabled("acceptance") || enabled("review")) {
|
|
775
1034
|
// Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
|
|
776
1035
|
// unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
|
|
777
1036
|
// verdict: each gets the same commit and the same brief it always got, and neither promise is
|
|
@@ -792,25 +1051,19 @@ export async function runGates(task, ctx) {
|
|
|
792
1051
|
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
793
1052
|
const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
|
|
794
1053
|
const reviewed = reviewing?.then((outcome) => record(outcome));
|
|
795
|
-
|
|
1054
|
+
if (executionSignal()) {
|
|
1055
|
+
// A cancelled sibling still owns a process until it unwinds; do not settle the task early.
|
|
1056
|
+
const settled = await Promise.allSettled([judged, reviewed]);
|
|
1057
|
+
const rejected = settled.find((result) => result.status === "rejected");
|
|
1058
|
+
if (rejected?.status === "rejected")
|
|
1059
|
+
throw rejected.reason;
|
|
1060
|
+
executionSignal()?.throwIfAborted();
|
|
1061
|
+
}
|
|
1062
|
+
else
|
|
1063
|
+
await Promise.all([judged, reviewed]);
|
|
796
1064
|
if (failed())
|
|
797
1065
|
return done();
|
|
798
1066
|
}
|
|
799
|
-
else if (!v185) {
|
|
800
|
-
// Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
|
|
801
|
-
if (enabled("acceptance")) {
|
|
802
|
-
await emitStart("acceptance");
|
|
803
|
-
const judged = await measure("acceptance", runAcceptance);
|
|
804
|
-
await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
|
|
805
|
-
if (failed())
|
|
806
|
-
return done();
|
|
807
|
-
}
|
|
808
|
-
if (enabled("review")) {
|
|
809
|
-
await emitStart("review");
|
|
810
|
-
await record(await measure("review", runReview));
|
|
811
|
-
}
|
|
812
|
-
return done();
|
|
813
|
-
}
|
|
814
1067
|
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
815
1068
|
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
816
1069
|
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
|
|
@@ -839,7 +1092,8 @@ export async function runGates(task, ctx) {
|
|
|
839
1092
|
const hit = verdictStore.get(identity);
|
|
840
1093
|
if (hit)
|
|
841
1094
|
classifySignalOnlyTest(hit);
|
|
842
|
-
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1095
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1096
|
+
&& !(await discardCachedRed("test", hit))) {
|
|
843
1097
|
full = formatReusedRow(hit, identity);
|
|
844
1098
|
cached = true;
|
|
845
1099
|
await noteReuse("test", full, identity);
|
|
@@ -848,8 +1102,8 @@ export async function runGates(task, ctx) {
|
|
|
848
1102
|
if (!full) {
|
|
849
1103
|
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
850
1104
|
full = fullUsesManifest
|
|
851
|
-
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
|
|
852
|
-
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
|
|
1105
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
|
|
1106
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
|
|
853
1107
|
}
|
|
854
1108
|
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
855
1109
|
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
@@ -859,7 +1113,7 @@ export async function runGates(task, ctx) {
|
|
|
859
1113
|
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
860
1114
|
}
|
|
861
1115
|
const merged = withTelemetry(dirt
|
|
862
|
-
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
1116
|
+
? await dirtyRefusal("test", dirt, ctx.commands.test)
|
|
863
1117
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
864
1118
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
865
1119
|
heldTest = undefined;
|