tickmarkr 2.5.3 → 2.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/dist/adapters/registry.js +6 -1
  2. package/dist/adapters/types.d.ts +3 -0
  3. package/dist/cli/commands/approve.d.ts +1 -0
  4. package/dist/cli/commands/approve.js +54 -6
  5. package/dist/cli/commands/doctor.d.ts +4 -0
  6. package/dist/cli/commands/doctor.js +44 -29
  7. package/dist/cli/commands/fleet.js +195 -33
  8. package/dist/cli/commands/init.js +196 -6
  9. package/dist/cli/commands/plan.js +144 -3
  10. package/dist/cli/commands/resume.js +6 -2
  11. package/dist/cli/commands/run.js +21 -2
  12. package/dist/cli/commands/verify.js +1 -1
  13. package/dist/cli/help.d.ts +2 -0
  14. package/dist/cli/help.js +3 -1
  15. package/dist/compile/collateral.d.ts +2 -0
  16. package/dist/compile/collateral.js +50 -9
  17. package/dist/compile/ownership.d.ts +7 -0
  18. package/dist/compile/ownership.js +59 -22
  19. package/dist/config/config.d.ts +23 -8
  20. package/dist/config/config.js +42 -28
  21. package/dist/config/fleet-overlay.d.ts +3 -9
  22. package/dist/config/fleet-overlay.js +68 -11
  23. package/dist/config/fleet-why.d.ts +7 -0
  24. package/dist/config/fleet-why.js +5 -0
  25. package/dist/drivers/herdr.d.ts +5 -0
  26. package/dist/drivers/herdr.js +14 -3
  27. package/dist/drivers/index.d.ts +15 -1
  28. package/dist/drivers/index.js +38 -10
  29. package/dist/drivers/orca.d.ts +99 -10
  30. package/dist/drivers/orca.js +586 -97
  31. package/dist/drivers/types.d.ts +1 -0
  32. package/dist/gates/baseline.d.ts +3 -0
  33. package/dist/gates/baseline.js +2 -1
  34. package/dist/gates/cache.d.ts +100 -0
  35. package/dist/gates/cache.js +389 -0
  36. package/dist/gates/review.d.ts +35 -2
  37. package/dist/gates/review.js +64 -16
  38. package/dist/gates/run-gates.d.ts +5 -0
  39. package/dist/gates/run-gates.js +134 -18
  40. package/dist/gates/test-manifest.d.ts +99 -0
  41. package/dist/gates/test-manifest.js +389 -0
  42. package/dist/gates/test-reporter.d.ts +4 -0
  43. package/dist/gates/test-reporter.js +49 -0
  44. package/dist/route/preference.d.ts +22 -1
  45. package/dist/route/preference.js +123 -25
  46. package/dist/route/router.d.ts +13 -0
  47. package/dist/route/router.js +95 -17
  48. package/dist/run/daemon.d.ts +13 -0
  49. package/dist/run/daemon.js +398 -96
  50. package/dist/run/git.d.ts +35 -1
  51. package/dist/run/git.js +111 -10
  52. package/dist/run/journal.d.ts +14 -2
  53. package/dist/run/journal.js +78 -10
  54. package/dist/run/lease.d.ts +14 -0
  55. package/dist/run/lease.js +87 -0
  56. package/dist/run/merge.d.ts +2 -0
  57. package/dist/run/merge.js +91 -3
  58. package/dist/run/operator-state.d.ts +11 -0
  59. package/dist/run/operator-state.js +17 -3
  60. package/dist/tui/cockpit/board.d.ts +96 -0
  61. package/dist/tui/cockpit/board.js +346 -0
  62. package/dist/tui/cockpit/decision-actions.js +2 -0
  63. package/dist/tui/cockpit/layout.d.ts +5 -1
  64. package/dist/tui/cockpit/layout.js +8 -3
  65. package/dist/tui/cockpit/live-runtime.js +83 -31
  66. package/dist/tui/cockpit/run-view.d.ts +7 -5
  67. package/dist/tui/cockpit/run-view.js +12 -11
  68. package/dist/tui/ink/fleet-app.d.ts +41 -27
  69. package/dist/tui/ink/fleet-app.js +204 -31
  70. package/package.json +1 -1
  71. package/skills/tickmarkr-overseer/SKILL.md +180 -99
@@ -3,7 +3,7 @@ import { join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
6
- import { renderAcceptanceItem } from "../graph/schema.js";
6
+ import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
9
  import { structuredFindings } from "../run/journal.js";
@@ -200,9 +200,52 @@ function reviewPreferIndex(c, prefer) {
200
200
  const i = prefer.findIndex((p) => p === c.adapter || p === channelKey(c));
201
201
  return i === -1 ? prefer.length : i;
202
202
  }
203
+ /**
204
+ * RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
205
+ * task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
206
+ * reviewer's tier. The cause names the input that reached the maximum (earlier inputs win a tie, so a
207
+ * floor the author's tier already satisfies is attributed to the author).
208
+ */
209
+ export function resolveReviewerFloor(authorTier, taskFloor, configFloor, priorReviewerTier) {
210
+ const inputs = [
211
+ [authorTier, "author-tier"], [taskFloor, "task-floor"], [configFloor, "config"], [priorReviewerTier, "prior-reviewer"],
212
+ ];
213
+ let best = { floor: authorTier, cause: "author-tier" };
214
+ for (const [tier, cause] of inputs) {
215
+ if (tier !== undefined && TIER_RANK[tier] > TIER_RANK[best.floor])
216
+ best = { floor: tier, cause };
217
+ }
218
+ return best;
219
+ }
220
+ /**
221
+ * The highest tier among the task's prior reviewers — the prior reviewer's tier for RF-1. A recorded
222
+ * dispatch tier is historical evidence and wins over the current pool; an unrecorded one falls back to
223
+ * the seat's channel; a seat neither establishes (it left the pool on resume, or the journal holds
224
+ * garbage) holds frontier — fail closed, never silently dropped.
225
+ */
226
+ export function priorReviewerTier(channels, priorReviewers = []) {
227
+ let top;
228
+ for (const p of priorReviewers) {
229
+ const key = typeof p === "string" ? p : p.reviewer;
230
+ const seen = (typeof p === "string" ? undefined : p.tier) ?? channels.find((ch) => channelKey(ch) === key)?.tier;
231
+ const tier = TIERS.includes(seen) ? seen : "frontier";
232
+ if (top === undefined || TIER_RANK[tier] > TIER_RANK[top])
233
+ top = tier;
234
+ }
235
+ return top;
236
+ }
237
+ /**
238
+ * The gate's floor: author tier, task floor, review.floor (a tier — `worker` names none) and the seats
239
+ * the caller names as THIS TASK's prior reviewers (earlier rounds' seats, a flaked seat). Eligibility
240
+ * exclusions are NOT evidence — a retry bans a flaked seat's whole adapter, and those sibling channels
241
+ * never reviewed — and neither is the run-scoped LRU rotation history, which names unrelated tasks' seats.
242
+ */
243
+ export function gateReviewerFloor(task, cfg, author, channels, priorReviewers = []) {
244
+ return resolveReviewerFloor(author.tier, task.routingHints?.floor, cfg.review.floor === "worker" ? undefined : cfg.review.floor, priorReviewerTier(channels, priorReviewers));
245
+ }
203
246
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
204
247
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
205
- floor, // task-declared only; config floors govern workers and must not silently move review seats
248
+ floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
206
249
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
207
250
  onSeat, demoted = new Set()) {
208
251
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
@@ -213,6 +256,8 @@ onSeat, demoted = new Set()) {
213
256
  if (!authorChannel)
214
257
  return null;
215
258
  const authorProvider = modelProvider(author.model, authorChannel.vendor);
259
+ // RF-1: every caller inherits the author-tier floor — a reviewer is never seated below its author.
260
+ const effectiveFloor = resolveReviewerFloor(author.tier, floor).floor;
216
261
  const ranked = channels
217
262
  // Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
218
263
  // as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
@@ -222,7 +267,7 @@ onSeat, demoted = new Set()) {
222
267
  && modelProvider(c.model, c.vendor) !== authorProvider
223
268
  && modelId(c.model) !== modelId(author.model)
224
269
  && !exclude.includes(channelKey(c))
225
- && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
270
+ && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
226
271
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
227
272
  const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
228
273
  || history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
@@ -247,7 +292,10 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
247
292
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
248
293
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
249
294
  // direct tests) skips persistence and changes nothing else.
250
- artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
295
+ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
296
+ // RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
297
+ // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
298
+ priorReviewers = []) {
251
299
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
252
300
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
253
301
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -317,20 +365,19 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
317
365
  policy: "full",
318
366
  ...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
319
367
  };
320
- // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
321
- // historical seat for every task that never asked for review-tier coupling.
322
- const reviewerFloor = task.routingHints?.floor;
368
+ // RF-1: the floor is max(author tier, task floor, review.floor tier, prior reviewer's tier). Only
369
+ // review.floor is read from config — cfg.routing.floors governs workers and never moves review seats.
370
+ const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
371
+ const floorMeta = { reviewerFloor, reviewerFloorCause };
323
372
  let rotationSeat;
324
373
  const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
325
374
  if (!reviewer) {
326
375
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
327
376
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
328
- const reason = reviewerFloor
329
- ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
330
- : "no cross-vendor reviewer available (diversity rule)";
377
+ const reason = `no cross-vendor reviewer available at or above ${reviewerFloor} floor (${reviewerFloorCause}; diversity rule)`;
331
378
  return cfg.review.required || priorMaterials.length > 0
332
- ? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
333
- : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
379
+ ? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...floorMeta } }
380
+ : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...floorMeta } };
334
381
  }
335
382
  reviewHistory?.push(channelKey(reviewer));
336
383
  const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
@@ -369,6 +416,7 @@ Classify every concern as "material" (a correctness, security, or acceptance-cri
369
416
  block the merge) or "minor" (style, naming, or preference that should not block). ONLY material findings
370
417
  block approval. For a minor concern you have decided not to block on, set "defer": true and give a
371
418
  one-line "rationale" — it is recorded in the review, never dropped.
419
+ A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
372
420
 
373
421
  Respond with ONLY this JSON:
374
422
  {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
@@ -453,7 +501,9 @@ The top-level comments array is optional. Use it only for actionable line-anchor
453
501
  meta: {
454
502
  ...policyMeta,
455
503
  ...rotationMeta,
504
+ ...floorMeta,
456
505
  reviewer: channelKey(reviewer),
506
+ reviewerTier: reviewer.tier,
457
507
  vendor: reviewer.vendor,
458
508
  provider,
459
509
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
@@ -488,15 +538,13 @@ The top-level comments array is optional. Use it only for actionable line-anchor
488
538
  pass: decided.pass,
489
539
  details,
490
540
  meta: {
491
- ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
541
+ ...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier, vendor: reviewer.vendor, provider,
542
+ // OBS-990 b: the verbatim ids plus ONE normalised copy of each list — never a third alias.
492
543
  ...(priorMaterials.length ? {
493
544
  resolved: v.resolved,
494
545
  reraised: v.reraised,
495
- normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
496
546
  resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
497
547
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
498
- normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
499
- normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
500
548
  } : {}),
501
549
  ...(reraised.length ? { findings: [
502
550
  ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
@@ -3,8 +3,10 @@ import { type TickmarkrConfig } from "../config/config.js";
3
3
  import { type GateName, type Task } from "../graph/schema.js";
4
4
  import { type Baseline } from "./baseline.js";
5
5
  import { type GateVia } from "./llm.js";
6
+ import { type PriorReviewer } from "./review.js";
6
7
  import type { GateResult } from "./types.js";
7
8
  import { type StructuredFinding } from "../run/journal.js";
9
+ import { type VerificationScope } from "./cache.js";
8
10
  export type LoadProvider = () => number;
9
11
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
10
12
  export declare function setLoadProviderForTests(provider: LoadProvider): void;
@@ -42,6 +44,7 @@ export type GateEvent = {
42
44
  result: GateResult;
43
45
  };
44
46
  export interface GateContext {
47
+ verificationScope?: VerificationScope;
45
48
  worktree: string;
46
49
  baseRef: string;
47
50
  result: WorkerResult;
@@ -57,11 +60,13 @@ export interface GateContext {
57
60
  excludeReviewers?: string[];
58
61
  demotedReviewers?: Set<string>;
59
62
  reviewHistory?: string[];
63
+ priorReviewers?: PriorReviewer[];
60
64
  artifactDir?: string;
61
65
  pipeline?: "v185" | "legacy";
62
66
  selectTests?: boolean;
63
67
  collateral?: ReadonlyArray<string>;
64
68
  onGate?: (e: GateEvent) => void | Promise<void>;
69
+ stateDir?: string;
65
70
  }
66
71
  /**
67
72
  * The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
@@ -6,15 +6,17 @@ import { TIER_RANK } from "../config/config.js";
6
6
  import { getAdapter } from "../adapters/registry.js";
7
7
  import { GATE_NAMES } from "../graph/schema.js";
8
8
  import { acceptanceGate } from "./acceptance.js";
9
- import { compareToBaseline } from "./baseline.js";
9
+ import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
10
10
  import { evidenceGate } from "./evidence.js";
11
11
  import { captureLlmOutput } from "./llm.js";
12
12
  import { disallowedBy } from "../route/preference.js";
13
13
  import { marginalCostRank } from "../route/router.js";
14
- import { pickReviewer, reviewGate } from "./review.js";
14
+ import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
15
15
  import { scopeGate } from "./scope.js";
16
- import { shGit } from "../run/git.js";
16
+ import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
17
+ import { shGit, resolvedCapacity } from "../run/git.js";
17
18
  import { withJudgeInvocationEvidence } from "../run/journal.js";
19
+ import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
18
20
  const productionLoadProvider = () => loadavg()[0] ?? 0;
19
21
  let loadProvider = productionLoadProvider;
20
22
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
@@ -192,9 +194,45 @@ export function testCommandForFiles(testCmd, files) {
192
194
  const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
193
195
  return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
194
196
  }
197
+ /** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
198
+ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
199
+ const entry = baseline.commands.test;
200
+ const outcome = await evaluateManifestedTest(cmd, worktree, {
201
+ baselineDurations: entry?.fileDurations,
202
+ longestFile: entry?.longestFile,
203
+ overallCeilingMs: effectiveCeilingMs(entry),
204
+ artifactDir,
205
+ });
206
+ const reportPath = outcome.reportPath;
207
+ return {
208
+ gate: "test",
209
+ pass: outcome.pass,
210
+ details: outcome.details,
211
+ meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
212
+ };
213
+ }
214
+ const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
215
+ const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
216
+ /** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
217
+ * classification happens after persistence, too late to keep a scripted runner's non-verdict out.
218
+ * Keep named failures as work verdicts and preserve details for failure-policy fingerprinting. */
219
+ function classifySignalOnlyTest(g) {
220
+ if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
221
+ return;
222
+ const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
223
+ if (named || FAILURE_IDENTITY_RE.test(g.details))
224
+ return;
225
+ g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
226
+ }
195
227
  export async function runGates(task, ctx) {
196
228
  const results = [];
197
229
  let commits = [];
230
+ const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
231
+ const verdictStore = getVerdictStore(stateDir);
232
+ // VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
233
+ // the ledger names the reuse and the identity even where the gate-result row's details must stay
234
+ // the fresh verdict's (see formatReusedRow).
235
+ const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
198
236
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
199
237
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
200
238
  const failed = () => results.some((r) => !r.pass);
@@ -359,7 +397,7 @@ export async function runGates(task, ctx) {
359
397
  const runBattery = async (commands, selected, gates = toolGates) => {
360
398
  if (!gates.length)
361
399
  return;
362
- if (!v185) {
400
+ if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
363
401
  // ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
364
402
  // not at true execution start. They are collectively sub-second (measured), so the debounce
365
403
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
@@ -386,25 +424,60 @@ export async function runGates(task, ctx) {
386
424
  // any later tool before anyone reads its verdict.
387
425
  for (const g of gates) {
388
426
  await emitStart(g);
389
- const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
427
+ const cmd = commands[g];
428
+ let r;
429
+ let cached = false;
430
+ let identity;
431
+ if (cmd !== undefined) {
432
+ identity = await computeVerificationIdentity({
433
+ worktree: ctx.worktree,
434
+ gate: g,
435
+ scope: ctx.verificationScope,
436
+ command: cmd,
437
+ baseline: ctx.baseline,
438
+ selectedSet: g === "test" ? selected : undefined,
439
+ capacity: resolvedCapacity(),
440
+ });
441
+ const hit = verdictStore.get(identity);
442
+ if (hit)
443
+ classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
444
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
445
+ r = formatReusedRow(hit, identity);
446
+ cached = true;
447
+ await noteReuse(g, r, identity);
448
+ }
449
+ }
450
+ if (!r) {
451
+ // VL-1: a detected vitest test command is judged by its own invocation-bound report — the
452
+ // stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
453
+ // other scripted test command keeps today's exit-code contract byte-identically.
454
+ const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
455
+ r = useManifest
456
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir))
457
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {})))[0];
458
+ }
390
459
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
391
460
  if (g === "test" && selected)
392
- selectedDurationMs = spans.get("test").durationMs;
461
+ selectedDurationMs = spans.get("test")?.durationMs ?? 0;
393
462
  // The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
394
463
  // tracked file makes it dirty again, and every gate after it — including the next shell gate,
395
464
  // which would then run against bytes HEAD does not hold — inherits that. So re-check after each
396
465
  // command, the last one included, and fail the gate whose command did it. (A red command needs
397
466
  // no check: it already ends the round, and its own output is the truer verdict.)
398
- if (r.pass && commands[g]) {
467
+ if (!cached && r.pass && commands[g]) {
399
468
  const dirt = await dirtyWorktree();
400
469
  if (dirt) {
401
470
  await record(dirtyRefusal(g, dirt, commands[g]));
402
471
  return;
403
472
  }
404
473
  }
474
+ if (r)
475
+ classifySignalOnlyTest(r);
476
+ if (!cached && identity && r && !isInfraResult(r)) {
477
+ verdictStore.set(identity, { ...r, meta: { ...r.meta, source: "gate", runDir: ctx.artifactDir } });
478
+ }
405
479
  if (g === "test" && selected) {
406
480
  const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
407
- // green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
408
481
  if (!screened.pass)
409
482
  await record(screened);
410
483
  else {
@@ -602,7 +675,11 @@ export async function runGates(task, ctx) {
602
675
  }
603
676
  return rv;
604
677
  };
605
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
678
+ // RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
679
+ // it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
680
+ // retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
681
+ const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
682
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers));
606
683
  // OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
607
684
  // different adapter. Only a single-adapter eligible pool may fall back to another channel on the
608
685
  // flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
@@ -622,10 +699,14 @@ export async function runGates(task, ctx) {
622
699
  const priorExclusions = ctx.excludeReviewers ?? [];
623
700
  const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
624
701
  const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
625
- const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
702
+ // RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
703
+ // and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
704
+ const retryPrior = [...priorReviewers, flaked];
705
+ const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
706
+ const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor);
626
707
  const exclusion = crossAdapter ? "adapter" : "channel";
627
708
  const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
628
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
709
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior));
629
710
  if (second.meta?.noEligibleReviewer !== true) {
630
711
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
631
712
  const route = exclusion === "adapter"
@@ -639,10 +720,11 @@ export async function runGates(task, ctx) {
639
720
  meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
640
721
  };
641
722
  }
642
- else if (task.routingHints?.floor) {
643
- // Preserve the original no-answer cause when no replacement exists, but name the declared
644
- // floor that correctly refused a lower-tier fallback.
645
- rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}` };
723
+ else {
724
+ // Preserve the original no-answer cause when no replacement exists, but name the resolved
725
+ // floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
726
+ const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
727
+ rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
646
728
  }
647
729
  }
648
730
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
@@ -739,9 +821,43 @@ export async function runGates(task, ctx) {
739
821
  // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
740
822
  // may have run one before it, and every gate between the battery and here reads commits only, so
741
823
  // a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
742
- const [full] = await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]));
743
- fullDurationMs = spans.get("test").durationMs - (selectedDurationMs ?? 0);
744
- const dirt = full.pass ? await dirtyWorktree() : undefined;
824
+ // VL-1: the merge-candidate's manifest is the FULL set — a full suite whose report lacks one
825
+ // manifest file never reaches the pass branch below, so a selected-only green can never merge.
826
+ let full;
827
+ let cached = false;
828
+ let identity;
829
+ if (ctx.commands.test !== undefined) {
830
+ identity = await computeVerificationIdentity({
831
+ worktree: ctx.worktree,
832
+ gate: "test",
833
+ scope: ctx.verificationScope,
834
+ command: ctx.commands.test,
835
+ baseline: ctx.baseline,
836
+ selectedSet: undefined,
837
+ capacity: resolvedCapacity(),
838
+ });
839
+ const hit = verdictStore.get(identity);
840
+ if (hit)
841
+ classifySignalOnlyTest(hit);
842
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
843
+ full = formatReusedRow(hit, identity);
844
+ cached = true;
845
+ await noteReuse("test", full, identity);
846
+ }
847
+ }
848
+ if (!full) {
849
+ const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
850
+ full = fullUsesManifest
851
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
852
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
853
+ }
854
+ fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
855
+ const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
856
+ if (full)
857
+ classifySignalOnlyTest(full);
858
+ if (!cached && identity && full && !dirt && !isInfraResult(full)) {
859
+ verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
860
+ }
745
861
  const merged = withTelemetry(dirt
746
862
  ? dirtyRefusal("test", dirt, ctx.commands.test)
747
863
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
@@ -0,0 +1,99 @@
1
+ import type { BaselineFileDuration } from "./baseline.js";
2
+ export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
3
+ /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
4
+ * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
5
+ * through any symlink in it (e.g. macOS's `/var` -> `/private/var`, under which every OS temp dir —
6
+ * and so every test fixture worktree — lives); `cwd` as tickmarkr holds it may not be. Resolving cwd
7
+ * before computing the relative path is what makes the two actually comparable. Manifests built from
8
+ * a selected-test screen are already repo-relative and pass through unchanged. */
9
+ export declare function toManifestPath(file: string, cwd: string): string;
10
+ export interface TestReportCompletion {
11
+ at: number;
12
+ status: "passed" | "failed";
13
+ /** Failure fingerprints for this file; absent/empty on a passed file. */
14
+ failures?: string[];
15
+ }
16
+ /** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
17
+ export interface TestReport {
18
+ nonce: string;
19
+ requested: string[];
20
+ started: Record<string, number>;
21
+ completed: Record<string, TestReportCompletion>;
22
+ /** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
23
+ * this themselves (a second write silently overwrites the first), so the reporter records the
24
+ * evidence separately before it is lost. */
25
+ duplicateCompletions?: string[];
26
+ /** Written last, once, when the runner reaches its own terminal state. Its absence means the run
27
+ * never certified completion — killed, crashed, or still in flight — and is never a verdict. */
28
+ certificate?: {
29
+ at: number;
30
+ exitCode: number;
31
+ };
32
+ }
33
+ /** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
34
+ export declare function readTestReport(path: string): TestReport | undefined;
35
+ export type ManifestVerdictKind = "pass" | "infra" | "work" | "fail-closed";
36
+ export interface ManifestVerdict {
37
+ kind: ManifestVerdictKind;
38
+ pass: boolean;
39
+ details: string;
40
+ meta: Record<string, unknown>;
41
+ }
42
+ /**
43
+ * The independent validator: given the manifest THIS invocation was asked to prove, its bound nonce
44
+ * and the independently observed process exit code, decide the
45
+ * verdict from the report alone. `killedFile` short-circuits every report-shaped check — a job this
46
+ * module killed for a per-file hang never reaches its report.
47
+ */
48
+ export declare function verifyManifestReport(opts: {
49
+ manifest: readonly string[];
50
+ nonce: string;
51
+ exitCode: number | undefined;
52
+ report: TestReport | undefined;
53
+ killedFile?: string;
54
+ hangBudgetMs?: number;
55
+ }): ManifestVerdict;
56
+ /** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
57
+ export declare const FILE_HANG_SLACK = 3;
58
+ export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
59
+ export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
60
+ export interface ManifestRunResult {
61
+ exitCode: number | undefined;
62
+ stdout: string;
63
+ stderr: string;
64
+ report: TestReport | undefined;
65
+ killedFile?: string;
66
+ hangBudgetMs?: number;
67
+ /** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
68
+ * that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
69
+ pid?: number;
70
+ }
71
+ /** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
72
+ * timeout kills the detached process group, including descendants holding the output pipes. */
73
+ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
74
+ manifest: readonly string[];
75
+ nonce: string;
76
+ reportPath: string;
77
+ env?: NodeJS.ProcessEnv;
78
+ baselineDurations?: readonly BaselineFileDuration[] | null;
79
+ longestFile?: BaselineFileDuration | null;
80
+ pollMs?: number;
81
+ overallCeilingMs?: number;
82
+ }): Promise<ManifestRunResult>;
83
+ export interface ManifestGateOutcome {
84
+ pass: boolean;
85
+ kind: ManifestVerdictKind;
86
+ details: string;
87
+ classification?: "infra" | "regression";
88
+ meta: Record<string, unknown>;
89
+ exitCode: number;
90
+ reportPath: string;
91
+ }
92
+ /** One configured runner execution, and its own collection under the same arguments and environment.
93
+ * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
94
+ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
95
+ baselineDurations?: readonly BaselineFileDuration[] | null;
96
+ longestFile?: BaselineFileDuration | null;
97
+ overallCeilingMs?: number;
98
+ artifactDir?: string;
99
+ }): Promise<ManifestGateOutcome>;