tickmarkr 2.1.4 → 2.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,20 @@
1
1
  import type { TickmarkrConfig } from "../config/config.js";
2
2
  import type { AcceptanceItem } from "../graph/schema.js";
3
- import { type ShResult } from "../run/git.js";
3
+ import { type RunCapacity, type ShResult } from "../run/git.js";
4
4
  import type { GateResult } from "./types.js";
5
+ /**
6
+ * T7: the capacity a gate's own command ran under rides the RESULT, beside the verdict it explains,
7
+ * rather than inside `meta` — `meta` is a machine-readable extras bag several callers compare
8
+ * wholesale, and identity that a later session keys reuse on does not belong in a bag. Declared here,
9
+ * at the one producer, because only a battery gate has a command whose child received a fork cap:
10
+ * every other gate leaves the field absent, which is the honest reading of "this gate divided
11
+ * nothing". The daemon lifts it verbatim onto the journal's gate row (src/run/daemon.ts).
12
+ */
13
+ declare module "./types.js" {
14
+ interface GateResult {
15
+ capacity?: RunCapacity;
16
+ }
17
+ }
5
18
  export interface BaselineCommand {
6
19
  /**
7
20
  * Absent when the capture returned no verdict — see `infra`. A pre-v1.90 baseline can also lack it
@@ -21,8 +34,15 @@ export interface BaselineCommand {
21
34
  /** The ceiling that measurement implies, persisted so every later battery uses the same number. */
22
35
  ceilingMs?: number;
23
36
  /**
24
- * OBS-534 (T2): the capture was SIGKILLed at its ceiling. It never finished asking the question, so
25
- * the entry carries a CAUSE and no verdict: no exit code, no fingerprints, nothing forgivable.
37
+ * T7: the capacity this command's capture child ran under — the fork cap it received and the cores
38
+ * that cap was divided from. Absent in every pre-T7 baseline, which is exactly what makes those
39
+ * entries keep their current forgiveness; a MALFORMED one fails closed instead (git.ts readCapacity).
40
+ */
41
+ capacity?: RunCapacity;
42
+ /**
43
+ * The capture did not return a trustworthy verdict: it was SIGKILLed at its ceiling, or its output
44
+ * proves the machine was exhausted while it ran. The entry therefore carries a CAUSE and no
45
+ * verdict: no exit code, no fingerprints, nothing forgivable.
26
46
  */
27
47
  infra?: true;
28
48
  }
@@ -1,6 +1,6 @@
1
1
  import { existsSync, readFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
- import { DEFAULT_SHELL_TIMEOUT_MS, sh } from "../run/git.js";
3
+ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
4
4
  // incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
5
5
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
6
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
@@ -117,12 +117,22 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
117
117
  // shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
118
118
  // this runner-output classifier rather than applied to any judge-authored reason text. A real test
119
119
  // failure still dominates below because one regression-shaped line makes the whole output regression.
120
- const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
120
+ const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b/i;
121
+ // Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
122
+ // keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
123
+ // for evidence that the capture ran while the machine was resource-starved.
124
+ const CAPTURE_EXHAUSTION_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable/i;
121
125
  // A named error CLASS ("AssertionError", "TypeError", "MyDomainError") — never bare "Error", which
122
126
  // is what an errno report itself is headed with (`Error: spawn EAGAIN`). The prefix is required.
123
127
  const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
124
128
  const isInfraLine = (l) => INFRA_RE.test(l) && !ERROR_CLASS_RE.test(l) && !namesFailure(l) && !SUMMARY_FAIL_RE.test(l);
125
129
  const namesRegression = (l) => (isFailureShaped(l) || ERROR_CLASS_RE.test(l)) && !isInfraLine(l);
130
+ // Infrastructure signatures must enter the fingerprint diff too. Otherwise a signature such as
131
+ // birpc's assertion-free RPC death is classified correctly in the raw output but collapses to the
132
+ // content-free UNRECOGNIZED_FAILURE marker before the fresh-failure path can ask the same classifier.
133
+ // This does not make the vocabulary open-ended: INFRA_RE is still the one closed list, and
134
+ // classifyFailureOutput's ERROR_CLASS_RE/namesFailure vetoes still decide mixed lines.
135
+ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
126
136
  /**
127
137
  * What a nonzero runner exit is evidence OF. `undefined` when the output names neither — the
128
138
  * unreadable-runner case the existing fail-closed path already owns.
@@ -133,6 +143,18 @@ export function classifyFailureOutput(output) {
133
143
  return "regression";
134
144
  return lines.some(isInfraLine) ? "infra" : undefined;
135
145
  }
146
+ /**
147
+ * Capture validity asks a different question from gate classification. At a gate, one genuine
148
+ * regression line must outrank adjacent errno evidence so a real defect is never laundered as infra.
149
+ * At capture, any such evidence invalidates the whole measurement: once the machine was exhausted,
150
+ * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
151
+ * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
152
+ */
153
+ const captureHasInvalidatingInfra = (output) => output
154
+ .split("\n")
155
+ .map((l) => l.replace(ANSI_RE, ""))
156
+ .filter((l) => !PASS_LINE_RE.test(l))
157
+ .some((l) => CAPTURE_EXHAUSTION_RE.test(l));
136
158
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
137
159
  // Vitest's default reporter names file durations as
138
160
  // `✓ |project| tests/example.test.ts (12 tests) 1.23s` (❯ for a red file). The runner may be
@@ -180,10 +202,10 @@ export function fingerprint(output) {
180
202
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
181
203
  const shaped = [];
182
204
  for (const l of lines) {
183
- if (isFailureShaped(l))
205
+ if (isFingerprintShaped(l))
184
206
  shaped.push(l);
185
207
  const stripped = stripTurboPrefix(l);
186
- if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFailureShaped(stripped))
208
+ if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
187
209
  shaped.push(stripped);
188
210
  }
189
211
  if (!shaped.length)
@@ -370,6 +392,15 @@ export function ceilingKillResult(gate, r, ceilingMs) {
370
392
  * shell asks it to finish faster than the thing it is measuring.
371
393
  */
372
394
  export const CAPTURE_CEILING_MS = 1_800_000;
395
+ const invalidCaptureEntry = (durationMs) => ({
396
+ infra: true,
397
+ fingerprints: [],
398
+ durationMs,
399
+ fileDurationSumMs: null,
400
+ impliedParallelism: null,
401
+ longestFile: null,
402
+ ceilingMs: effectiveCeilingMs({ durationMs }),
403
+ });
373
404
  export async function captureBaseline(cwd, commands) {
374
405
  const base = { commands: {} };
375
406
  for (const [name, cmd] of Object.entries(commands)) {
@@ -396,18 +427,22 @@ export async function captureBaseline(cwd, commands) {
396
427
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
397
428
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
398
429
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
399
- base.commands[name] = {
400
- infra: true,
401
- fingerprints: [],
402
- durationMs,
403
- fileDurationSumMs: null,
404
- impliedParallelism: null,
405
- longestFile: null,
406
- ceilingMs: effectiveCeilingMs({ durationMs }),
407
- };
430
+ base.commands[name] = invalidCaptureEntry(durationMs);
408
431
  continue;
409
432
  }
410
433
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
434
+ // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
435
+ // discriminator correctly called the mixed output a regression. But the same output also said
436
+ // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
437
+ // being taken. A capture cannot know which red lines predated that shortage and which it caused,
438
+ // so none may become a fingerprint every later task gets to forgive.
439
+ if (captureHasInvalidatingInfra(raw)) {
440
+ console.error(`tickmarkr: baseline capture for "${name}" completed with process/resource-exhaustion evidence — `
441
+ + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
442
+ + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion.`);
443
+ base.commands[name] = invalidCaptureEntry(durationMs);
444
+ continue;
445
+ }
411
446
  base.commands[name] = {
412
447
  exitCode: r.code,
413
448
  // a command that exits 0 has no failures to fingerprint — recording any would be a lie the
@@ -417,6 +452,9 @@ export async function captureBaseline(cwd, commands) {
417
452
  durationMs,
418
453
  ...fileTiming(raw, durationMs),
419
454
  ceilingMs: effectiveCeilingMs({ durationMs }),
455
+ // T7: the world this measurement was taken in, so a later reader can ask whether its own world
456
+ // is the same one. Recorded from THIS command's own shell result, never re-derived here.
457
+ ...(r.capacity ? { capacity: r.capacity } : {}),
420
458
  };
421
459
  }
422
460
  const names = Object.keys(commands);
@@ -487,35 +525,31 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
487
525
  const entry = baseline.commands[name];
488
526
  const ceilingMs = effectiveCeilingMs(entry);
489
527
  const r = await sh(cmd, cwd, ceilingMs);
528
+ // T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
529
+ // result rather than re-derived after the fact. The skip row above ran no command and therefore
530
+ // states no capacity — a row that never divided the machine must not claim that it did.
531
+ const record = (g) => {
532
+ results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
533
+ };
534
+ // …and whether the entry that would forgive this command was measured in the same world. A
535
+ // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
536
+ // machine divided by a different number. Absent capacity (every pre-T7 baseline) still forgives
537
+ // exactly as it does today; a malformed one fails closed (git.ts sameCapacity).
538
+ const comparable = sameCapacity(entry?.capacity, r.capacity);
490
539
  // Q24: the kill is read BEFORE the exit code is interpreted at all. A SIGKILLed battery has
491
540
  // whatever partial output it had flushed — typically no failure shape — so every path below
492
541
  // would otherwise turn a timeout into a claim about the work: "no recognizable failure lines"
493
542
  // when the baseline was green, or a forgiven pre-existing red when it was not. Neither is true.
494
543
  const killed = ceilingKillResult(name, r, ceilingMs);
495
544
  if (killed) {
496
- results.push(killed);
545
+ record(killed);
497
546
  continue;
498
547
  }
499
548
  if (r.code === 0) {
500
- results.push({ gate: name, pass: true, details: "exit 0" });
549
+ record({ gate: name, pass: true, details: "exit 0" });
501
550
  continue;
502
551
  }
503
552
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
504
- // T9: classify BEFORE the baseline diff, and record it on every nonzero result. An infra-only
505
- // exit means the runner never completed a suite, so there is nothing to forgive and nothing
506
- // verified — it fails, and `meta.infra` marks it so the merge predicate cannot read it as a
507
- // satisfied gate even if some future producer reports it as a pass. Baseline forgiveness stays
508
- // exactly where it belongs: on failures the runner actually reported and the baseline already had.
509
- const classification = classifyFailureOutput(raw);
510
- if (classification === "infra") {
511
- results.push({
512
- gate: name,
513
- pass: false,
514
- details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
515
- meta: { classification, infra: true },
516
- });
517
- continue;
518
- }
519
553
  // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
520
554
  // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
521
555
  // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
@@ -524,6 +558,25 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
524
558
  // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
525
559
  // verdict); loosening back to vocabulary re-opens OBS-278.
526
560
  const { failing, unreadable } = freshFailures(entry, raw);
561
+ // T9: classify the FRESH diff before charging it. The complete runner output can legitimately
562
+ // contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
563
+ // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
564
+ // `classifyFailureOutput` remains the single discriminator. When there is no fresh fingerprint,
565
+ // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
566
+ const freshClassification = failing.length ? classifyFailureOutput(failing.join("\n")) : undefined;
567
+ const classification = freshClassification ?? (!failing.length ? classifyFailureOutput(raw) : undefined);
568
+ if (classification === "infra") {
569
+ const evidence = failing.length
570
+ ? failing.slice(0, 10).join("\n")
571
+ : unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n");
572
+ record({
573
+ gate: name,
574
+ pass: false,
575
+ details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
576
+ meta: { classification, infra: true },
577
+ });
578
+ continue;
579
+ }
527
580
  // OBS-534 (T2): only a recorded VERDICT can be forgiven. A green baseline has no red to forgive,
528
581
  // and neither has a capture that was killed at its ceiling — it recorded a cause instead, so it
529
582
  // fails closed on the same branch rather than reading as "only pre-existing failures". Legacy
@@ -534,7 +587,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
534
587
  ? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
535
588
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
536
589
  const evidence = unrecognizedEvidence(raw);
537
- results.push({
590
+ record({
538
591
  gate: name,
539
592
  pass: false,
540
593
  details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
@@ -545,10 +598,26 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
545
598
  if (failing.length) {
546
599
  const headlined = headlineDetails(raw, failing);
547
600
  const meta = { ...headlined.meta, ...(classification ? { classification } : {}) };
548
- results.push({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
601
+ record({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
602
+ continue;
603
+ }
604
+ // T7: everything below this line is forgiveness, and forgiveness is the one verdict that reads a
605
+ // record from another session. The failures are all baseline-recorded — but recorded under a
606
+ // capacity this command did not run under, so they are not evidence that these failures
607
+ // pre-existed the diff. A red does not become a green on fingerprints from a world it was not
608
+ // measured in; the operator gets both worlds named.
609
+ if (!comparable) {
610
+ record({
611
+ gate: name,
612
+ pass: false,
613
+ details: `exit ${r.code}; every failure is recorded in the baseline, but that capture ran under `
614
+ + `${describeCapacity(entry?.capacity)} and this command ran under ${describeCapacity(r.capacity)} — `
615
+ + `forgiveness across a changed capacity is not evidence, so this fails closed`,
616
+ meta: { capacityMismatch: true, ...(classification ? { classification } : {}) },
617
+ });
549
618
  continue;
550
619
  }
551
- results.push({
620
+ record({
552
621
  gate: name,
553
622
  pass: true,
554
623
  details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
@@ -25,6 +25,7 @@ export interface LlmVia {
25
25
  label?: string;
26
26
  keep?: boolean;
27
27
  onSlot?: (slot: Slot) => void;
28
+ onInactivity?: () => void;
28
29
  }
29
30
  export interface GateVia {
30
31
  driver: ExecutorDriver;
package/dist/gates/llm.js CHANGED
@@ -226,6 +226,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
226
226
  const snapshotQuietFor = now - quietSince;
227
227
  if (cpuFlatFor >= harvestCpuFlatWindowMs(cpu.resolutionMs)
228
228
  && snapshotQuietFor >= gateInactivityWindowMs) {
229
+ via.onInactivity?.();
229
230
  break;
230
231
  }
231
232
  }
@@ -1,5 +1,5 @@
1
1
  import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
2
- import { type TickmarkrConfig } from "../config/config.js";
2
+ import { type TickmarkrConfig, type Tier } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
4
  import { type GateVia } from "./llm.js";
5
5
  import type { GateResult } from "./types.js";
@@ -49,6 +49,13 @@ export declare function isDiffCapPark(result: GateResult): boolean;
49
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
50
  export declare function modelId(model: string): string;
51
51
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
52
- prefer?: string[]): BillingChannel | null;
52
+ prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
53
+ floor?: Tier): BillingChannel | null;
53
54
  export type ReviewUnparseableCause = VerdictUnparseableCause;
55
+ /**
56
+ * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
57
+ * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
58
+ * judgement rather than a guarantee made by this renderer.
59
+ */
60
+ export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
54
61
  export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
@@ -170,7 +170,8 @@ function reviewPreferIndex(c, prefer) {
170
170
  return i === -1 ? prefer.length : i;
171
171
  }
172
172
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
173
- prefer = []) {
173
+ prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
174
+ floor) {
174
175
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
175
176
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
176
177
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -182,9 +183,25 @@ prefer = []) {
182
183
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
183
184
  // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
184
185
  // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
185
- .filter((c) => c.vendor !== authorChannel.vendor && modelId(c.model) !== modelId(author.model) && !exclude.includes(channelKey(c)))
186
+ .filter((c) => c.vendor !== authorChannel.vendor
187
+ && modelId(c.model) !== modelId(author.model)
188
+ && !exclude.includes(channelKey(c))
189
+ && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
186
190
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
187
191
  }
192
+ /**
193
+ * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
194
+ * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
195
+ * judgement rather than a guarantee made by this renderer.
196
+ */
197
+ export function renderDeclaredWriteScope(files) {
198
+ if (files.length === 0) {
199
+ return "## Declared write scope\nUnrestricted: this task declared no write-scope patterns.";
200
+ }
201
+ return `## Declared write scope
202
+ The task DECLARED these write-scope patterns:
203
+ ${files.map((path) => `- ${path}`).join("\n")}`;
204
+ }
188
205
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
189
206
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
190
207
  // direct tests) skips persistence and changes nothing else.
@@ -257,13 +274,19 @@ artifactDir) {
257
274
  policy: "full",
258
275
  ...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
259
276
  };
260
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
277
+ // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
278
+ // historical seat for every task that never asked for review-tier coupling.
279
+ const reviewerFloor = task.routingHints?.floor;
280
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
261
281
  if (!reviewer) {
262
282
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
263
283
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
284
+ const reason = reviewerFloor
285
+ ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
286
+ : "no cross-vendor reviewer available (diversity rule)";
264
287
  return cfg.review.required
265
- ? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
266
- : { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
288
+ ? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
289
+ : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
267
290
  }
268
291
  const measuredDiff = await fetchTaskDiff(worktree, baseRef);
269
292
  // Keep the reader payload identical to the text charged to the strict cap:
@@ -284,6 +307,8 @@ ${COMPLETION_FAKING_CHECKLIST}
284
307
  ## Acceptance criteria
285
308
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
286
309
 
310
+ ${renderDeclaredWriteScope(task.files)}
311
+
287
312
  ## Diff
288
313
  \`\`\`diff
289
314
  ${diff}
@@ -301,7 +326,15 @@ Respond with ONLY this JSON:
301
326
  Approve iff no material finding remains; an empty findings list is a clean approval.
302
327
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
303
328
  `;
304
- const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
329
+ let concludedOnInactivity = false;
330
+ const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
331
+ driver: via.driver,
332
+ keep: via.keep,
333
+ onSlot: via.onSlot,
334
+ name: via.nameFor("review", reviewer.adapter),
335
+ label: via.labelFor("review"),
336
+ onInactivity: () => { concludedOnInactivity = true; },
337
+ } : undefined,
305
338
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
306
339
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
307
340
  // stdout that read as "unparseable" and escalated to re-implementation of green code
@@ -325,14 +358,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
325
358
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
326
359
  }
327
360
  }
328
- const failure = cause === "malformed-verdict"
329
- ? "review output unparseable"
330
- : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
361
+ const failure = concludedOnInactivity
362
+ ? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
363
+ : cause === "malformed-verdict"
364
+ ? "review output unparseable"
365
+ : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
331
366
  return {
332
367
  gate: "review",
333
368
  pass: false,
334
369
  details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
335
- meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
370
+ meta: {
371
+ ...policyMeta,
372
+ reviewer: channelKey(reviewer),
373
+ unparseable: true,
374
+ cause,
375
+ ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
376
+ },
336
377
  };
337
378
  }
338
379
  const decided = findings !== null
@@ -589,7 +589,18 @@ export async function runGates(task, ctx) {
589
589
  const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
590
590
  if (second.meta?.noEligibleReviewer !== true) {
591
591
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
592
- rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
592
+ rv = {
593
+ ...second,
594
+ // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
595
+ // re-route visible in the result text a reader actually opens, including on a red retry.
596
+ details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
597
+ meta: { ...second.meta, reviewRetry: { flaked, retried } },
598
+ };
599
+ }
600
+ else if (task.routingHints?.floor) {
601
+ // Preserve the original no-answer cause when no replacement exists, but name the declared
602
+ // floor that correctly refused a lower-tier fallback.
603
+ rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}` };
593
604
  }
594
605
  }
595
606
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;