tickmarkr 2.5.5 → 2.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +3 -1
  2. package/dist/cli/commands/approve.js +64 -3
  3. package/dist/cli/commands/doctor.d.ts +10 -0
  4. package/dist/cli/commands/fleet.js +4 -0
  5. package/dist/cli/commands/plan.js +20 -4
  6. package/dist/cli/commands/verify.js +102 -81
  7. package/dist/config/config.d.ts +11 -0
  8. package/dist/config/config.js +21 -12
  9. package/dist/config/fleet-overlay.js +55 -17
  10. package/dist/drivers/index.js +2 -1
  11. package/dist/drivers/orca.d.ts +21 -1
  12. package/dist/drivers/orca.js +209 -27
  13. package/dist/gates/baseline.d.ts +14 -3
  14. package/dist/gates/baseline.js +61 -14
  15. package/dist/gates/cache.d.ts +14 -13
  16. package/dist/gates/cache.js +17 -5
  17. package/dist/gates/llm.d.ts +3 -0
  18. package/dist/gates/llm.js +11 -0
  19. package/dist/gates/review.d.ts +28 -3
  20. package/dist/gates/review.js +106 -14
  21. package/dist/gates/run-gates.d.ts +6 -1
  22. package/dist/gates/run-gates.js +299 -101
  23. package/dist/gates/test-manifest.d.ts +30 -1
  24. package/dist/gates/test-manifest.js +113 -39
  25. package/dist/gates/test-reporter.js +14 -6
  26. package/dist/graph/graph.d.ts +2 -0
  27. package/dist/graph/graph.js +45 -1
  28. package/dist/run/consult.js +5 -4
  29. package/dist/run/daemon.d.ts +2 -0
  30. package/dist/run/daemon.js +480 -99
  31. package/dist/run/execution-budget.d.ts +25 -0
  32. package/dist/run/execution-budget.js +142 -0
  33. package/dist/run/git.d.ts +40 -0
  34. package/dist/run/git.js +89 -4
  35. package/dist/run/journal.d.ts +3 -1
  36. package/dist/run/journal.js +38 -16
  37. package/dist/run/lease.d.ts +44 -0
  38. package/dist/run/lease.js +226 -3
  39. package/dist/run/recovery.d.ts +8 -0
  40. package/dist/run/recovery.js +25 -0
  41. package/dist/run/repair-selection.d.ts +12 -0
  42. package/dist/run/repair-selection.js +56 -0
  43. package/dist/run/stall.d.ts +6 -1
  44. package/dist/run/stall.js +60 -3
  45. package/dist/tui/ink/fleet-app.d.ts +10 -2
  46. package/dist/tui/ink/fleet-app.js +33 -15
  47. package/package.json +2 -2
  48. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
@@ -5,6 +5,7 @@ import { type Baseline } from "./baseline.js";
5
5
  import { type GateVia } from "./llm.js";
6
6
  import { type PriorReviewer } from "./review.js";
7
7
  import type { GateResult } from "./types.js";
8
+ import { type VerificationRetryCause } from "../run/recovery.js";
8
9
  import { type StructuredFinding } from "../run/journal.js";
9
10
  import { type VerificationScope } from "./cache.js";
10
11
  export type LoadProvider = () => number;
@@ -44,6 +45,7 @@ export type GateEvent = {
44
45
  result: GateResult;
45
46
  };
46
47
  export interface GateContext {
48
+ authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
47
49
  verificationScope?: VerificationScope;
48
50
  worktree: string;
49
51
  baseRef: string;
@@ -59,11 +61,14 @@ export interface GateContext {
59
61
  carriedFindings?: readonly StructuredFinding[];
60
62
  excludeReviewers?: string[];
61
63
  demotedReviewers?: Set<string>;
64
+ reviewNoVerdicts?: Map<string, string[]>;
65
+ carriedAuthors?: readonly string[];
62
66
  reviewHistory?: string[];
63
67
  priorReviewers?: PriorReviewer[];
64
68
  artifactDir?: string;
65
- pipeline?: "v185" | "legacy";
66
69
  selectTests?: boolean;
70
+ requiredRepairTests?: readonly string[];
71
+ selectionReason?: string;
67
72
  collateral?: ReadonlyArray<string>;
68
73
  onGate?: (e: GateEvent) => void | Promise<void>;
69
74
  stateDir?: string;
@@ -1,4 +1,4 @@
1
- import { mkdtempSync, readFileSync, rmSync } from "node:fs";
1
+ import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
2
2
  import { loadavg, tmpdir } from "node:os";
3
3
  import { join, posix } from "node:path";
4
4
  import { channelKey, shq } from "../adapters/types.js";
@@ -6,17 +6,19 @@ import { TIER_RANK } from "../config/config.js";
6
6
  import { getAdapter } from "../adapters/registry.js";
7
7
  import { GATE_NAMES } from "../graph/schema.js";
8
8
  import { acceptanceGate } from "./acceptance.js";
9
- import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
9
+ import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
10
10
  import { evidenceGate } from "./evidence.js";
11
11
  import { captureLlmOutput } from "./llm.js";
12
12
  import { disallowedBy } from "../route/preference.js";
13
13
  import { marginalCostRank } from "../route/router.js";
14
- import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
14
+ import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
15
15
  import { scopeGate } from "./scope.js";
16
16
  import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
17
- import { shGit, resolvedCapacity } from "../run/git.js";
17
+ import { executionSignal } from "../run/execution-budget.js";
18
+ import { failureDisposition } from "../run/recovery.js";
19
+ import { preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
18
20
  import { withJudgeInvocationEvidence } from "../run/journal.js";
19
- import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
21
+ import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
20
22
  const productionLoadProvider = () => loadavg()[0] ?? 0;
21
23
  let loadProvider = productionLoadProvider;
22
24
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
@@ -96,6 +98,23 @@ async function captureLlmDispatches(adapters, run) {
96
98
  throw error;
97
99
  }
98
100
  }
101
+ /** Parse `git status --porcelain --untracked-files=all -z` output. A rename/copy carries its
102
+ * ORIGINAL path in a second NUL field immediately after the current one; skip it — dirt only
103
+ * cares about paths that exist in the worktree now. */
104
+ function parseStatusZ(stdout) {
105
+ const fields = stdout.split("\0");
106
+ const entries = [];
107
+ for (let i = 0; i < fields.length; i++) {
108
+ const rec = fields[i];
109
+ if (!rec)
110
+ continue;
111
+ const status = rec.slice(0, 2);
112
+ entries.push({ status, path: rec.slice(3) });
113
+ if (status.includes("R") || status.includes("C"))
114
+ i++; // consume the paired original-path field
115
+ }
116
+ return entries;
117
+ }
99
118
  const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
100
119
  // relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
101
120
  const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
@@ -195,7 +214,7 @@ export function testCommandForFiles(testCmd, files) {
195
214
  return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
196
215
  }
197
216
  /** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
198
- async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
217
+ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
199
218
  const entry = baseline.commands.test;
200
219
  const outcome = await evaluateManifestedTest(cmd, worktree, {
201
220
  baselineDurations: entry?.fileDurations,
@@ -204,6 +223,19 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
204
223
  artifactDir,
205
224
  });
206
225
  const reportPath = outcome.reportPath;
226
+ if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
227
+ && outcome.meta?.retryable !== false) {
228
+ const waitedMs = await waitForCalmWindow(executionSignal());
229
+ if (!calmWindowReady())
230
+ return { gate: "test", pass: false, details: outcome.details,
231
+ meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
232
+ if (!retry.authorizeRetry("infra")) {
233
+ return { gate: "test", pass: false, details: outcome.details,
234
+ meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
235
+ }
236
+ const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
237
+ return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
238
+ }
207
239
  return {
208
240
  gate: "test",
209
241
  pass: outcome.pass,
@@ -226,9 +258,16 @@ function classifySignalOnlyTest(g) {
226
258
  }
227
259
  export async function runGates(task, ctx) {
228
260
  const results = [];
261
+ let selectionDecision;
229
262
  let commits = [];
230
263
  const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
231
264
  const verdictStore = getVerdictStore(stateDir);
265
+ // R41: the verification protocol and the EFFECTIVE npm lifecycle policy measured for THIS
266
+ // checkout — the policy its runner children receive (an explicit process export, else npm's own
267
+ // resolved config for this worktree, project npmrc included). Every result leaves through
268
+ // withTelemetry carrying it, so the daemon's gate-result row records what the gate measured under
269
+ // rather than a session-wide value resolved somewhere else.
270
+ const verification = verificationProtocol(process.env, ctx.worktree);
232
271
  // VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
233
272
  // the ledger names the reuse and the identity even where the gate-result row's details must stay
234
273
  // the fresh verdict's (see formatReusedRow).
@@ -236,7 +275,6 @@ export async function runGates(task, ctx) {
236
275
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
237
276
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
238
277
  const failed = () => results.some((r) => !r.pass);
239
- const v185 = ctx.pipeline === "v185";
240
278
  // T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
241
279
  // round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
242
280
  // exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
@@ -288,13 +326,17 @@ export async function runGates(task, ctx) {
288
326
  // path that forgets to measure is visibly missing its telemetry rather than carrying a fabricated
289
327
  // zero. The daemon lifts these off `meta` onto the gate-result row (src/run/daemon.ts).
290
328
  const withTelemetry = (result) => {
329
+ if (result.gate === "test" && selectionDecision) {
330
+ result = { ...result, meta: { ...result.meta, selectionDecision } };
331
+ }
291
332
  const span = spans.get(result.gate);
292
333
  if (!span)
293
- return result;
334
+ return { ...result, meta: { ...result.meta, verification } };
294
335
  return {
295
336
  ...result,
296
337
  meta: {
297
338
  ...result.meta,
339
+ verification,
298
340
  ...span,
299
341
  ...(result.gate === "test" && selectedDurationMs !== undefined ? { selectedDurationMs } : {}),
300
342
  ...(result.gate === "test" && fullDurationMs !== undefined ? { fullDurationMs } : {}),
@@ -335,7 +377,7 @@ export async function runGates(task, ctx) {
335
377
  if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
336
378
  const dirt = await dirtyWorktree();
337
379
  if (dirt) {
338
- const refusal = withTelemetry(dirtyRoundRefusal(last.gate, dirt));
380
+ const refusal = withTelemetry(await dirtyRoundRefusal(last.gate, dirt));
339
381
  results[results.indexOf(last)] = refusal;
340
382
  sorted[sorted.length - 1] = refusal;
341
383
  await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
@@ -360,66 +402,178 @@ export async function runGates(task, ctx) {
360
402
  * exempt: an untracked source file is uncommitted work by every reading git offers.
361
403
  */
362
404
  const dirtyWorktree = async () => {
363
- const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
405
+ const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain --untracked-files=all -z", ctx.worktree);
364
406
  if (r.code !== 0)
365
- return `git status failed (exit ${r.code}) — the worktree cannot be proven clean`;
366
- const entries = r.stdout
367
- .split("\n")
368
- .map((l) => l.trimEnd())
369
- .filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l));
370
- return entries.length ? entries.join("\n") : undefined;
407
+ return { text: `git status failed (exit ${r.code}) — the worktree cannot be proven clean`, entries: [] };
408
+ const entries = parseStatusZ(r.stdout).filter((e) => !/^\.tickmarkr-[^/]*$/.test(e.path));
409
+ if (!entries.length)
410
+ return undefined;
411
+ return { text: entries.map((e) => `${e.status} ${e.path}`).join("\n"), entries };
371
412
  };
372
413
  const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
373
414
  + `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
374
415
  + `and never merged (and the committed diff would never be run)`;
375
416
  // `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
376
- const dirtyRefusal = (gate, dirt, left) => ({
377
- gate,
378
- pass: false,
379
- details: DIRTY_WHY
380
- + (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
381
- + dirt,
382
- meta: { dirtyWorktree: true, ...(left ? { dirtiedBy: gate } : {}) },
383
- });
417
+ const dirtyRefusal = async (gate, dirt, left) => {
418
+ let preservedRef;
419
+ let preservationError;
420
+ try {
421
+ preservedRef = await preserveWorktree(ctx.worktree);
422
+ }
423
+ catch (error) {
424
+ // Never masks the refusal, but never pretends a snapshot exists either — surfaced below.
425
+ preservationError = error instanceof Error ? error.message : String(error);
426
+ }
427
+ const dirtyPaths = [];
428
+ let allUntracked = true;
429
+ let totalBytes = 0;
430
+ for (const { status, path } of dirt.entries) {
431
+ dirtyPaths.push(path);
432
+ if (status !== "??") {
433
+ allUntracked = false;
434
+ }
435
+ try {
436
+ const st = statSync(join(ctx.worktree, path));
437
+ if (st.isFile()) {
438
+ totalBytes += st.size;
439
+ }
440
+ }
441
+ catch {
442
+ // ignore deleted or unreadable
443
+ }
444
+ }
445
+ let allAbsentFromDiff = true;
446
+ if (ctx.baseRef) {
447
+ try {
448
+ const diffOut = await shGit(`git diff --name-only -z ${shq(ctx.baseRef)}..HEAD`, ctx.worktree);
449
+ if (diffOut.code === 0) {
450
+ // Paths in Git's newline output are C-quoted, which is not a lossless path encoding
451
+ // (and must never be parsed as JSON). `-z` lets this comparison retain arbitrary
452
+ // filenames exactly, including whitespace and non-ASCII bytes.
453
+ const touched = new Set(diffOut.stdout.split("\0").filter(Boolean));
454
+ for (const p of dirtyPaths) {
455
+ if (touched.has(p)) {
456
+ allAbsentFromDiff = false;
457
+ break;
458
+ }
459
+ }
460
+ }
461
+ else {
462
+ // An unreadable committed diff cannot prove the litter is unrelated to the worker.
463
+ allAbsentFromDiff = false;
464
+ }
465
+ }
466
+ catch {
467
+ allAbsentFromDiff = false;
468
+ }
469
+ }
470
+ const isInfra = Boolean(left && allUntracked && allAbsentFromDiff);
471
+ const primaryFile = dirtyPaths[0] ?? "";
472
+ const meta = {
473
+ dirtyWorktree: true,
474
+ ref: preservedRef,
475
+ preservedRef,
476
+ paths: dirtyPaths,
477
+ files: dirtyPaths,
478
+ path: primaryFile,
479
+ file: primaryFile,
480
+ bytes: totalBytes,
481
+ byteCount: totalBytes,
482
+ };
483
+ if (left) {
484
+ meta.dirtiedBy = gate;
485
+ meta.culprit = left;
486
+ meta.culpritCommand = left;
487
+ meta.command = left;
488
+ }
489
+ if (isInfra) {
490
+ meta.infra = true;
491
+ meta.classification = "infra";
492
+ }
493
+ if (preservationError) {
494
+ meta.preservationFailed = true;
495
+ meta.preservationError = preservationError;
496
+ // A refusal without a durable snapshot must never enter the ordinary repair/escalation
497
+ // path: the daemon recognizes infra rows as terminal parks for this attempt. This also
498
+ // overrides a chargeable entry/tracked-dirt classification, because retrying could lose
499
+ // the only remaining copy of the worktree state.
500
+ meta.infra = true;
501
+ meta.classification = "infra";
502
+ meta.retryable = false;
503
+ meta.recoveryBlocked = "dirty worktree preservation failed; no recovery ref exists";
504
+ }
505
+ return {
506
+ gate,
507
+ pass: false,
508
+ details: DIRTY_WHY
509
+ + (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
510
+ + dirt.text
511
+ + (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
512
+ meta,
513
+ };
514
+ };
384
515
  // The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
385
516
  // the last cleanliness check, and naming a culprit this function cannot identify would be a worse
386
517
  // record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
387
- const dirtyRoundRefusal = (gate, dirt) => ({
388
- gate,
389
- pass: false,
390
- details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
391
- + `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
392
- + `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
393
- + `Uncommitted at round end:\n${dirt}`,
394
- meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
395
- });
518
+ const dirtyRoundRefusal = async (gate, dirt) => {
519
+ let preservedRef;
520
+ let preservationError;
521
+ try {
522
+ preservedRef = await preserveWorktree(ctx.worktree);
523
+ }
524
+ catch (error) {
525
+ preservationError = error instanceof Error ? error.message : String(error);
526
+ }
527
+ const dirtyPaths = [];
528
+ let totalBytes = 0;
529
+ for (const { path } of dirt.entries) {
530
+ dirtyPaths.push(path);
531
+ try {
532
+ const st = statSync(join(ctx.worktree, path));
533
+ if (st.isFile()) {
534
+ totalBytes += st.size;
535
+ }
536
+ }
537
+ catch { }
538
+ }
539
+ const primaryFile = dirtyPaths[0] ?? "";
540
+ return {
541
+ gate,
542
+ pass: false,
543
+ details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
544
+ + `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
545
+ + `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
546
+ + `Uncommitted at round end:\n${dirt.text}`
547
+ + (preservationError ? `\npreservation failed — the litter could not be snapshotted onto a recovery ref: ${preservationError}` : ""),
548
+ meta: {
549
+ dirtyWorktree: true,
550
+ dirtyAtRoundEnd: true,
551
+ ref: preservedRef,
552
+ preservedRef,
553
+ paths: dirtyPaths,
554
+ files: dirtyPaths,
555
+ path: primaryFile,
556
+ file: primaryFile,
557
+ bytes: totalBytes,
558
+ byteCount: totalBytes,
559
+ ...(preservationError ? {
560
+ preservationFailed: true,
561
+ preservationError,
562
+ infra: true,
563
+ classification: "infra",
564
+ retryable: false,
565
+ recoveryBlocked: "dirty worktree preservation failed; no recovery ref exists",
566
+ } : {}),
567
+ },
568
+ };
569
+ };
396
570
  // shell tools vs the shared baseline
571
+ const retryOptions = (identity) => ctx.authorizeInfraRetry
572
+ ? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
573
+ : {};
397
574
  const runBattery = async (commands, selected, gates = toolGates) => {
398
575
  if (!gates.length)
399
576
  return;
400
- if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
401
- // ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
402
- // not at true execution start. They are collectively sub-second (measured), so the debounce
403
- // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
404
- // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
405
- // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
406
- const finish = startMeasurement();
407
- const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
408
- const batch = finish();
409
- for (const g of gates)
410
- addMeasurement(g, batch);
411
- // The same refusal AFTER the commands, because a green command can dirty the tree the check
412
- // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
413
- // lands on the last gate that had one — the round dies there either way. A red battery is
414
- // reported as the red it is: the round already ends, and the command output is the better lead.
415
- const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
416
- const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
417
- for (const r of toolResults) {
418
- await emitStart(r.gate);
419
- await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
420
- }
421
- return;
422
- }
423
577
  // T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
424
578
  // any later tool before anyone reads its verdict.
425
579
  for (const g of gates) {
@@ -453,8 +607,8 @@ export async function runGates(task, ctx) {
453
607
  // other scripted test command keeps today's exit-code contract byte-identically.
454
608
  const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
455
609
  r = useManifest
456
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir))
457
- : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {})))[0];
610
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
611
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "test" && selected ? { selected } : {}) })))[0];
458
612
  }
459
613
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
460
614
  if (g === "test" && selected)
@@ -467,7 +621,7 @@ export async function runGates(task, ctx) {
467
621
  if (!cached && r.pass && commands[g]) {
468
622
  const dirt = await dirtyWorktree();
469
623
  if (dirt) {
470
- await record(dirtyRefusal(g, dirt, commands[g]));
624
+ await record(await dirtyRefusal(g, dirt, commands[g]));
471
625
  return;
472
626
  }
473
627
  }
@@ -666,24 +820,46 @@ export async function runGates(task, ctx) {
666
820
  const rv = captured.value;
667
821
  if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
668
822
  await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
669
- if (rv.meta.seatAuthoredBytes === 0 && typeof rv.meta.reviewer === "string"
670
- && !ctx.demotedReviewers?.has(rv.meta.reviewer)) {
671
- ctx.demotedReviewers?.add(rv.meta.reviewer);
672
- await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
673
- payload: { reviewer: rv.meta.reviewer, cause: rv.meta.cause, seatAuthoredBytes: 0 }, result: rv });
823
+ if (typeof rv.meta.reviewer === "string") {
824
+ const reviewer = rv.meta.reviewer;
825
+ const cause = String(rv.meta.cause);
826
+ // OBS-1025 add.2: the run-scoped tally; two no-verdicts retire the seat for the rest of the run.
827
+ const causes = rv.meta.noVerdict === true && ctx.reviewNoVerdicts
828
+ ? [...(ctx.reviewNoVerdicts.get(reviewer) ?? []), cause] : undefined;
829
+ if (causes)
830
+ ctx.reviewNoVerdicts.set(reviewer, causes);
831
+ // OBS-1039: a seat below the byte floor at the beat is silent — demoted like a zero-byte seat.
832
+ const silent = rv.meta.seatAuthoredBytes === 0 || cause === "silent" || cause === "launch-never-started";
833
+ const twice = (causes?.length ?? 0) >= 2;
834
+ if ((silent || twice) && !ctx.demotedReviewers?.has(reviewer)) {
835
+ ctx.demotedReviewers?.add(reviewer);
836
+ await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
837
+ payload: { reviewer, cause, seatAuthoredBytes: rv.meta.seatAuthoredBytes ?? 0, ...(twice ? { causes } : {}) }, result: rv });
838
+ }
674
839
  }
675
840
  }
676
841
  return rv;
677
842
  };
843
+ // OBS-1025 add.2: seats with two no-verdicts this run are out of the rotation for every pick below.
844
+ const retired = [...(ctx.reviewNoVerdicts ?? [])].filter(([, causes]) => causes.length >= 2).map(([seat]) => seat);
678
845
  // RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
679
846
  // it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
680
847
  // retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
681
848
  const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
682
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers));
683
- // OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
684
- // different adapter. Only a single-adapter eligible pool may fall back to another channel on the
685
- // flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
686
- if ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
849
+ const carriedAuthors = ctx.carriedAuthors ?? [];
850
+ let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
851
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
852
+ // OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
853
+ // a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
854
+ // verdict never enters results; an exhausted pool preserves its cause.
855
+ // OBS-1013 add.3: the re-route LOOPS while seats return no verdict — a closure mismatch or a silent
856
+ // seat is re-routed to a third seat of another vendor, and only an exhausted pool ends the round,
857
+ // as a terminal infra result that keeps the carried materials. Never a worker.
858
+ let retryPrior = [...priorReviewers];
859
+ const routes = [];
860
+ let hop = 0;
861
+ while ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
862
+ hop++;
687
863
  const flaked = rv.meta.reviewer;
688
864
  const emptyOutput = rv.meta.cause === "empty-output";
689
865
  if (emptyOutput) {
@@ -694,29 +870,30 @@ export async function runGates(task, ctx) {
694
870
  });
695
871
  }
696
872
  const retryVia = ctx.via
697
- ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
873
+ ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
698
874
  : undefined;
699
- const priorExclusions = ctx.excludeReviewers ?? [];
700
875
  const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
701
876
  const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
702
877
  // RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
703
878
  // and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
704
- const retryPrior = [...priorReviewers, flaked];
879
+ retryPrior = [...retryPrior, flaked];
705
880
  const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
706
- const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor);
881
+ const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
707
882
  const exclusion = crossAdapter ? "adapter" : "channel";
708
- const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
709
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior));
883
+ exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
884
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
710
885
  if (second.meta?.noEligibleReviewer !== true) {
711
886
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
712
887
  const route = exclusion === "adapter"
713
888
  ? `different-adapter retry; excluded flaked adapter ${flakedAdapter}`
714
889
  : `same-adapter fallback; excluded flaked channel ${flaked}`;
890
+ const produced = emptyOutput ? "EMPTY output" : rv.meta.cause === "closure-mismatch" ? "a verdict closing no carried fingerprint" : "no parseable verdict";
891
+ // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep every
892
+ // re-route visible in the result text a reader actually opens, including on a red retry.
893
+ routes.push(`review re-route (${route}): ${flaked} produced ${produced}; replaced by ${retried}`);
715
894
  rv = {
716
895
  ...second,
717
- // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
718
- // re-route visible in the result text a reader actually opens, including on a red retry.
719
- details: `review re-route (${route}): ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
896
+ details: `${routes.join("\n")}\n${second.details}`,
720
897
  meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
721
898
  };
722
899
  }
@@ -725,8 +902,15 @@ export async function runGates(task, ctx) {
725
902
  // floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
726
903
  const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
727
904
  rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
905
+ break;
728
906
  }
729
907
  }
908
+ if (rv.meta?.noVerdict === true) {
909
+ // Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
910
+ // row is not a passing review — and the sibling judge result is untouched beside it.
911
+ const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
912
+ rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
913
+ }
730
914
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
731
915
  };
732
916
  // v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
@@ -745,10 +929,10 @@ export async function runGates(task, ctx) {
745
929
  if (entryDirt) {
746
930
  addMeasurement(sequence[0], entryMeasurement);
747
931
  await emitStart(sequence[0]);
748
- await record(dirtyRefusal(sequence[0], entryDirt));
932
+ await record(await dirtyRefusal(sequence[0], entryDirt));
749
933
  return done();
750
934
  }
751
- if (v185 && await screenBlocks())
935
+ if (await screenBlocks())
752
936
  return done();
753
937
  await runBattery(ctx.commands);
754
938
  if (failed())
@@ -765,13 +949,33 @@ export async function runGates(task, ctx) {
765
949
  }
766
950
  // A non-final round may run only the tests covering its own diff; the merge-candidate round below
767
951
  // pays the full suite anyway, so a selection that misses costs a round and can never merge.
768
- const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
952
+ let selected = ctx.selectTests && enabled("test") && ctx.commands.test
769
953
  ? await coveringTests(ctx.worktree, ctx.baseRef)
770
954
  : undefined;
955
+ let selectionReason = ctx.selectionReason ?? (selected ? "affected-tests" : "full-suite-fallback");
956
+ if (selected && ctx.requiredRepairTests?.length) {
957
+ const required = ctx.requiredRepairTests;
958
+ const safe = required.every((path) => posix.normalize(path) === path && !path.startsWith("../")
959
+ && !path.startsWith("/") && TEST_FILE_RE.test(path) && existsSync(join(ctx.worktree, path)));
960
+ const listed = safe ? await shGit(`git ls-files -z -- ${required.map(shq).join(" ")}`, ctx.worktree) : undefined;
961
+ const tracked = new Set(listed?.stdout.split("\0").filter(Boolean));
962
+ if (!listed || listed.code !== 0 || required.some((path) => !tracked.has(path))) {
963
+ selected = undefined;
964
+ selectionReason = "required-repair-test-unavailable";
965
+ }
966
+ else
967
+ selected = [...new Set([...selected, ...required])].sort();
968
+ }
969
+ if (ctx.selectionReason)
970
+ selectionDecision = {
971
+ scope: selected ? "selected" : "full", reason: selected ? selectionReason
972
+ : selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
973
+ requiredFiles: [...(ctx.requiredRepairTests ?? [])],
974
+ };
771
975
  await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
772
976
  if (failed())
773
977
  return done();
774
- if (v185 && (enabled("acceptance") || enabled("review"))) {
978
+ if (enabled("acceptance") || enabled("review")) {
775
979
  // Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
776
980
  // unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
777
981
  // verdict: each gets the same commit and the same brief it always got, and neither promise is
@@ -792,25 +996,19 @@ export async function runGates(task, ctx) {
792
996
  // GATE_NAMES order by done(); the event stream truthfully records each independent completion.
793
997
  const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
794
998
  const reviewed = reviewing?.then((outcome) => record(outcome));
795
- await Promise.all([judged, reviewed]);
999
+ if (executionSignal()) {
1000
+ // A cancelled sibling still owns a process until it unwinds; do not settle the task early.
1001
+ const settled = await Promise.allSettled([judged, reviewed]);
1002
+ const rejected = settled.find((result) => result.status === "rejected");
1003
+ if (rejected?.status === "rejected")
1004
+ throw rejected.reason;
1005
+ executionSignal()?.throwIfAborted();
1006
+ }
1007
+ else
1008
+ await Promise.all([judged, reviewed]);
796
1009
  if (failed())
797
1010
  return done();
798
1011
  }
799
- else if (!v185) {
800
- // Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
801
- if (enabled("acceptance")) {
802
- await emitStart("acceptance");
803
- const judged = await measure("acceptance", runAcceptance);
804
- await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
805
- if (failed())
806
- return done();
807
- }
808
- if (enabled("review")) {
809
- await emitStart("review");
810
- await record(await measure("review", runReview));
811
- }
812
- return done();
813
- }
814
1012
  // The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
815
1013
  // the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
816
1014
  // on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
@@ -848,8 +1046,8 @@ export async function runGates(task, ctx) {
848
1046
  if (!full) {
849
1047
  const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
850
1048
  full = fullUsesManifest
851
- ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
852
- : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
1049
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
1050
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
853
1051
  }
854
1052
  fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
855
1053
  const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
@@ -859,7 +1057,7 @@ export async function runGates(task, ctx) {
859
1057
  verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
860
1058
  }
861
1059
  const merged = withTelemetry(dirt
862
- ? dirtyRefusal("test", dirt, ctx.commands.test)
1060
+ ? await dirtyRefusal("test", dirt, ctx.commands.test)
863
1061
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
864
1062
  results[results.findIndex((r) => r.gate === "test")] = merged;
865
1063
  heldTest = undefined;