tickmarkr 2.4.2 → 2.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/README.md +110 -24
  2. package/dist/adapters/fake.d.ts +1 -0
  3. package/dist/adapters/fake.js +2 -1
  4. package/dist/adapters/opencode.d.ts +1 -0
  5. package/dist/adapters/opencode.js +4 -1
  6. package/dist/adapters/types.d.ts +14 -0
  7. package/dist/adapters/types.js +25 -0
  8. package/dist/cli/commands/approve.d.ts +42 -0
  9. package/dist/cli/commands/approve.js +80 -13
  10. package/dist/cli/commands/doctor.d.ts +234 -0
  11. package/dist/cli/commands/doctor.js +139 -5
  12. package/dist/cli/commands/init.d.ts +2 -0
  13. package/dist/cli/commands/init.js +45 -5
  14. package/dist/cli/commands/plan.js +58 -6
  15. package/dist/cli/commands/report.js +18 -45
  16. package/dist/cli/commands/run.js +3 -0
  17. package/dist/cli/commands/scope.js +36 -6
  18. package/dist/cli/commands/stats.d.ts +2 -0
  19. package/dist/cli/commands/stats.js +42 -23
  20. package/dist/cli/commands/status.js +20 -4
  21. package/dist/cli/commands/ui.js +42 -53
  22. package/dist/cli/commands/unlock.d.ts +1 -1
  23. package/dist/cli/commands/unlock.js +59 -9
  24. package/dist/cli/help.d.ts +219 -0
  25. package/dist/cli/help.js +212 -0
  26. package/dist/cli/index.d.ts +42 -2
  27. package/dist/cli/index.js +23 -8
  28. package/dist/drivers/herdr.d.ts +7 -2
  29. package/dist/drivers/herdr.js +78 -48
  30. package/dist/drivers/orca.d.ts +3 -2
  31. package/dist/drivers/orca.js +43 -4
  32. package/dist/drivers/subprocess.d.ts +1 -0
  33. package/dist/drivers/subprocess.js +3 -0
  34. package/dist/drivers/types.d.ts +14 -0
  35. package/dist/gates/artifact-manifest.d.ts +50 -0
  36. package/dist/gates/artifact-manifest.js +23 -0
  37. package/dist/gates/llm.d.ts +5 -0
  38. package/dist/gates/llm.js +188 -3
  39. package/dist/gates/review.d.ts +6 -3
  40. package/dist/gates/review.js +79 -33
  41. package/dist/gates/run-gates.d.ts +3 -0
  42. package/dist/gates/run-gates.js +14 -4
  43. package/dist/plan/scope.d.ts +25 -0
  44. package/dist/plan/scope.js +92 -12
  45. package/dist/report/operator-record.d.ts +49 -0
  46. package/dist/report/operator-record.js +137 -0
  47. package/dist/run/daemon.d.ts +5 -5
  48. package/dist/run/daemon.js +166 -44
  49. package/dist/run/lock.d.ts +59 -2
  50. package/dist/run/lock.js +184 -26
  51. package/dist/run/operator-state.d.ts +86 -0
  52. package/dist/run/operator-state.js +165 -0
  53. package/dist/run/supervision.d.ts +32 -0
  54. package/dist/run/supervision.js +138 -17
  55. package/dist/tui/cockpit/capture.d.ts +19 -0
  56. package/dist/tui/cockpit/capture.js +89 -1
  57. package/dist/tui/cockpit/components.d.ts +15 -1
  58. package/dist/tui/cockpit/components.js +79 -9
  59. package/dist/tui/cockpit/decision-actions.d.ts +147 -0
  60. package/dist/tui/cockpit/decision-actions.js +315 -0
  61. package/dist/tui/cockpit/derive.d.ts +1 -1
  62. package/dist/tui/cockpit/derive.js +2 -0
  63. package/dist/tui/cockpit/evidence-view.d.ts +119 -0
  64. package/dist/tui/cockpit/evidence-view.js +210 -0
  65. package/dist/tui/cockpit/home-view.d.ts +88 -0
  66. package/dist/tui/cockpit/home-view.js +240 -0
  67. package/dist/tui/cockpit/keys.d.ts +125 -0
  68. package/dist/tui/cockpit/keys.js +31 -0
  69. package/dist/tui/cockpit/layout.d.ts +14 -0
  70. package/dist/tui/cockpit/layout.js +15 -0
  71. package/dist/tui/cockpit/live-runtime.d.ts +46 -0
  72. package/dist/tui/cockpit/live-runtime.js +683 -0
  73. package/dist/tui/cockpit/live-store.d.ts +289 -0
  74. package/dist/tui/cockpit/live-store.js +308 -0
  75. package/dist/tui/cockpit/live.d.ts +21 -1
  76. package/dist/tui/cockpit/live.js +12 -1
  77. package/dist/tui/cockpit/run-view.d.ts +100 -0
  78. package/dist/tui/cockpit/run-view.js +202 -0
  79. package/dist/tui/cockpit/shell.d.ts +50 -0
  80. package/dist/tui/cockpit/shell.js +74 -0
  81. package/dist/tui/cockpit/theme.d.ts +27 -0
  82. package/dist/tui/cockpit/theme.js +21 -0
  83. package/dist/tui/ink/init-app.d.ts +4 -0
  84. package/dist/tui/ink/init-app.js +18 -7
  85. package/package.json +1 -1
  86. package/skills/tickmarkr-auto/SKILL.md +10 -1
  87. package/skills/tickmarkr-loop/SKILL.md +72 -1
  88. package/skills/tickmarkr-overseer/SKILL.md +12 -0
  89. package/skills/tickmarkr-overseer/scripts/watch-context.sh +13 -2
package/dist/gates/llm.js CHANGED
@@ -1,3 +1,4 @@
1
+ import { stripVTControlCharacters } from "node:util";
1
2
  import { AsyncLocalStorage } from "node:async_hooks";
2
3
  import { randomBytes } from "node:crypto";
3
4
  import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
@@ -5,7 +6,7 @@ import { tmpdir } from "node:os";
5
6
  import { join } from "node:path";
6
7
  import { matchesTrustDialog } from "../adapters/types.js";
7
8
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
8
- import { bannerShell, paneDispatchCommand } from "../brand.js";
9
+ import { bannerShell, paneDispatchCommand, PLAIN_BANNER } from "../brand.js";
9
10
  import { sh } from "../run/git.js";
10
11
  import { harvestCpuFlatWindowMs, normalizeStallSnapshot, WorkerTreeCpuAccountant, } from "../run/stall.js";
11
12
  export const GATE_PANE_SEP = " · ";
@@ -129,13 +130,164 @@ export async function captureLlmOutput(run) {
129
130
  const value = await llmOutputCapture.run(outputs, run);
130
131
  return { value, outputs };
131
132
  }
133
+ // The harness preamble has a fixed row shape, in order: the dispatch echo rows (the identity export
134
+ // and the START printf, each possibly re-echoed behind the shell prompt), the START acknowledgement
135
+ // row, the banner rows, blank rows, and the identity row. RULING-229-15 add.1 makes the boundary
136
+ // STRUCTURAL: a pane read is line-terminated, so only the LAST row of a capture can be a partial
137
+ // paint. Every row before it is complete, and a complete row is harness only when it EQUALS a full
138
+ // harness row in the position the preamble allows. A complete row that merely starts with "T",
139
+ // "export", "review" or a banner glyph is seat text, and seat prose that MENTIONS a marker mid-row
140
+ // counts in full — "contains TICKMARKR_START_ somewhere" is never by itself a reason to measure zero.
141
+ const BANNER_ROWS = PLAIN_BANNER.replace(/\n$/, "").split("\n");
142
+ const ECHO_OPENER = "export HERDR_WORKSPACE_ID=";
143
+ const ECHO_START_OPENER = "printf '%s%s\\n' 'TICKMARKR_START_' '";
144
+ const ECHO_OPENERS = [ECHO_OPENER, ECHO_START_OPENER];
145
+ // RULING-229-15 add.5: the prompt glyph set is this CLOSED list — the corpus sweeps it, the grammar
146
+ // consumes the LONGEST glyph that matches (so ">>" is one glyph and ">" is still one).
147
+ export const PROMPT_GLYPHS = ["➜", "❯", "$", "%", ">>", ">"];
148
+ const GIT_SEGMENT_OPEN = "git:(";
149
+ function echoRowMatch(row) {
150
+ const opener = (at) => {
151
+ const body = row.slice(at);
152
+ if (ECHO_OPENERS.some((o) => body.startsWith(o)))
153
+ return "complete";
154
+ return ECHO_OPENERS.some((o) => o.startsWith(body)) ? "partial" : "none";
155
+ };
156
+ const ws = (at) => /^\s+/.exec(row.slice(at))?.[0].length;
157
+ const bare = opener(0);
158
+ if (bare !== "none")
159
+ return bare;
160
+ const glyph = PROMPT_GLYPHS.find((g) => row.startsWith(g));
161
+ if (glyph === undefined)
162
+ return "none";
163
+ let i = glyph.length;
164
+ const ws1 = ws(i);
165
+ if (ws1 === undefined)
166
+ return row.length === i ? "partial" : "none";
167
+ i += ws1;
168
+ const dir = /^\S+/.exec(row.slice(i));
169
+ if (!dir)
170
+ return "partial";
171
+ i += dir[0].length;
172
+ const ws2 = ws(i);
173
+ if (ws2 === undefined)
174
+ return "partial";
175
+ i += ws2;
176
+ const rest = row.slice(i);
177
+ if (rest.startsWith(GIT_SEGMENT_OPEN)) {
178
+ const close = row.indexOf(")", i + GIT_SEGMENT_OPEN.length);
179
+ if (close < 0)
180
+ return "partial";
181
+ i = close + 1;
182
+ const ws3 = ws(i);
183
+ if (ws3 === undefined)
184
+ return row.length === i ? "partial" : "none";
185
+ return opener(i + ws3);
186
+ }
187
+ if (GIT_SEGMENT_OPEN.startsWith(rest))
188
+ return "partial"; // "", "g", "gi", "git", "git:"
189
+ return opener(i);
190
+ }
191
+ // A complete dispatch echo row: the grammar reaches the opener. Never "contains the opener" — a row
192
+ // that mentions it mid-prose is the seat's.
193
+ function completeEchoRow(row) {
194
+ return echoRowMatch(row) === "complete";
195
+ }
196
+ // A LAST row still being painted: any prefix of a row in the grammar.
197
+ function partialEchoRow(row) {
198
+ return echoRowMatch(row) !== "none";
199
+ }
200
+ const START_MARKER = "TICKMARKR_START_";
201
+ const START_ROW = /^TICKMARKR_START_[\w-]+$/;
202
+ const IDENTITY_LINE = /^(?:review\s*·|tickmarkr(?::|$))/;
203
+ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
204
+ // Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
205
+ const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
206
+ // The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
207
+ function seatStart(output) {
208
+ const rows = output.split("\n");
209
+ if (rows.length > 1 && rows[rows.length - 1] === "")
210
+ rows.pop(); // the read's own line terminator
211
+ let stage = ECHO;
212
+ let bannerAt; // next banner row expected once the banner has begun
213
+ let offset = 0;
214
+ for (let i = 0; i < rows.length; i++) {
215
+ const row = rows[i].replace(/[ \t]+$/, "");
216
+ const t = row.trim();
217
+ const last = i === rows.length - 1;
218
+ // Complete harness rows: equality against the shape the preamble allows at this stage.
219
+ let accepted = false;
220
+ if (stage < SEAT && t.length === 0)
221
+ accepted = true; // blank rows between preamble rows
222
+ else if (stage <= ECHO && completeEchoRow(row))
223
+ accepted = true;
224
+ else if (stage <= START && START_ROW.test(row)) {
225
+ stage = BANNER;
226
+ accepted = true;
227
+ }
228
+ else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
229
+ stage = BANNER;
230
+ bannerAt = BANNER_ROWS.indexOf(row) + 1;
231
+ accepted = true;
232
+ }
233
+ else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
234
+ bannerAt++;
235
+ accepted = true;
236
+ }
237
+ else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
238
+ stage = SEAT;
239
+ accepted = true;
240
+ }
241
+ if (accepted) {
242
+ offset += rows[i].length + 1;
243
+ continue;
244
+ }
245
+ if (!last)
246
+ return offset;
247
+ // The last row may be a partial paint: a prefix of the next harness row the preamble allows.
248
+ if (stage <= ECHO && partialEchoRow(row))
249
+ return -1;
250
+ if (stage <= START && (START_MARKER.startsWith(row) || /^TICKMARKR_START_[\w-]*$/.test(row)))
251
+ return -1;
252
+ if (stage <= BANNER && (bannerAt === undefined
253
+ ? BANNER_ROWS.some((b) => b.startsWith(row))
254
+ : BANNER_ROWS[bannerAt]?.startsWith(row) === true))
255
+ return -1;
256
+ // The identity row is painted right after the banner, so its prefix is a partial paint only there;
257
+ // with no banner in the capture, "review" or "tick" alone is the seat's own first row.
258
+ if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
259
+ return -1;
260
+ return offset;
261
+ }
262
+ return -1; // every row was harness — the seat has not taken its turn
263
+ }
264
+ // A pane's dispatch echo, start acknowledgement, banner and identity are harness bytes.
265
+ // Remove only that leading preamble, never matching text later in the seat's response.
266
+ // Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
267
+ // which is what makes the caller's running Math.max safe: a partial banner counted once would be
268
+ // retained for the whole call and buy a silent seat its full ceiling.
269
+ export function reviewSeatOutput(raw, nonce) {
270
+ const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
271
+ const start = seatStart(output);
272
+ if (start < 0)
273
+ return "";
274
+ const seat = output.slice(start);
275
+ // Anything after the nonce-bound trailer belongs to the terminal (typically the next shell
276
+ // prompt), not the reviewer. Deleting only the marker would count that postlude as seat output and
277
+ // let a zero-byte seat escape first-silence demotion.
278
+ const trailer = new RegExp(`(?:^|\\n)TICKMARKR_EXIT_${nonce}:\\d+[^\\n]*(?:\\n|$)`).exec(seat);
279
+ // A trailer row still being typed is not stripped: after the preamble a last row "T" is far more
280
+ // often the seat's first byte than the harness's exit marker, and the next read completes either.
281
+ return trailer ? seat.slice(0, trailer.index) : seat;
282
+ }
283
+ export const REVIEW_FIRST_LIVENESS_MS = 30_000;
132
284
  async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
133
285
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
134
286
  try {
135
287
  const pf = join(dir, "prompt.md");
136
288
  writeFileSync(pf, prompt);
137
289
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
138
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
290
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength(r.stdout + r.stderr) };
139
291
  }
140
292
  finally {
141
293
  rmSync(dir, { recursive: true, force: true });
@@ -150,6 +302,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
150
302
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
151
303
  let slot;
152
304
  let accountant;
305
+ let forceClose = false;
153
306
  try {
154
307
  const pf = join(dir, "prompt.md");
155
308
  writeFileSync(pf, prompt);
@@ -180,6 +333,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
180
333
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
181
334
  let out;
182
335
  let timedOut = false;
336
+ let launchNeverStarted = false;
337
+ let seatAuthoredBytes = 0;
183
338
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
184
339
  if (!gatePrompt) {
185
340
  await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
@@ -193,6 +348,9 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
193
348
  await accountant.start();
194
349
  const startedAt = Date.now();
195
350
  out = await via.driver.read(slot, 400);
351
+ const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
352
+ seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
353
+ let firstLivenessObserved = false;
196
354
  let priorSnapshot = normalizeStallSnapshot(out);
197
355
  const anchoredAt = Date.now();
198
356
  let quietSince = anchoredAt;
@@ -207,12 +365,35 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
207
365
  const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
208
366
  const raw = await via.driver.read(slot, 400);
209
367
  out = raw;
368
+ seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
210
369
  // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
211
370
  // wait timed out at the same boundary the marker landed; either way a trailer completes
212
371
  // normally and is never mistaken for inactivity.
213
372
  if (matched || new RegExp(exitPattern).test(raw))
214
373
  break;
215
374
  const now = Date.now();
375
+ // The ceiling wins if it coincides with the first beat (or the read crosses it).
376
+ // That seat was killed by its configured timeout, not an early launch reroute.
377
+ if (now - startedAt >= timeoutMs)
378
+ break;
379
+ if (reviewing && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
380
+ firstLivenessObserved = true;
381
+ // RS-2: the beat reads seat-authored bytes ALONE. CPU evidence never holds a preamble-only
382
+ // capture open to the ceiling; a seat that has not written one byte of its own is re-routed.
383
+ // OBS-944 is why a buffering runner earns no exemption here: the claude-code seat's 901 s
384
+ // capture was byte-identical across two legs and ended at the pane-identity line — that
385
+ // seat never started, it was not quietly working. RULING-229-06 puts the beat on the PANE
386
+ // path only; a headless `-p` runner (runHeadlessDetailed, no pane, no beat) buffers every
387
+ // byte until completion and keeps its full ceiling.
388
+ if (seatAuthoredBytes === 0) {
389
+ launchNeverStarted = true;
390
+ forceClose = true;
391
+ break;
392
+ }
393
+ }
394
+ // Producing reviews own their full ceiling; inactivity is not a review verdict.
395
+ if (reviewing)
396
+ continue;
216
397
  const snapshot = normalizeStallSnapshot(raw);
217
398
  if (snapshot !== priorSnapshot) {
218
399
  priorSnapshot = snapshot;
@@ -246,18 +427,22 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
246
427
  }
247
428
  }
248
429
  timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
430
+ if (timedOut)
431
+ forceClose = true;
249
432
  }
250
433
  const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
251
434
  return {
252
435
  output: dewrapPaneVerdict(out, nonce),
253
436
  ...(Number.isFinite(exitCode) ? { exitCode } : {}),
254
437
  timedOut,
438
+ launchNeverStarted,
439
+ seatAuthoredBytes,
255
440
  };
256
441
  }
257
442
  finally {
258
443
  try {
259
444
  await accountant?.stop();
260
- if (slot && !via.keep)
445
+ if (slot && (forceClose || !via.keep))
261
446
  await via.driver.close(slot);
262
447
  }
263
448
  finally {
@@ -1,6 +1,7 @@
1
1
  import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type TickmarkrConfig, type Tier } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
+ import { type StructuredFinding } from "../run/journal.js";
4
5
  import { modelProvider } from "../route/preference.js";
5
6
  import { type GateVia } from "./llm.js";
6
7
  import type { GateResult } from "./types.js";
@@ -16,6 +17,8 @@ export interface ReviewFinding {
16
17
  }
17
18
  export interface ReviewVerdict {
18
19
  approve?: boolean;
20
+ resolved?: string[];
21
+ reraised?: string[];
19
22
  issues?: string[];
20
23
  findings?: ReviewFinding[];
21
24
  comments?: Array<{
@@ -54,12 +57,12 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
54
57
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
55
58
  floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
56
59
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
57
- onSeat?: (seat: number) => void): BillingChannel | null;
58
- export type ReviewUnparseableCause = VerdictUnparseableCause;
60
+ onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
61
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
59
62
  /**
60
63
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
61
64
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
62
65
  * judgement rather than a guarantee made by this renderer.
63
66
  */
64
67
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
65
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
68
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[]): Promise<GateResult>;
@@ -6,6 +6,7 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
+ import { structuredFindings } from "../run/journal.js";
9
10
  import { redactSecrets } from "../run/redact.js";
10
11
  import { marginalCostRank } from "../route/router.js";
11
12
  import { modelProvider } from "../route/preference.js";
@@ -178,7 +179,7 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
178
179
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
179
180
  floor, // task-declared only; config floors govern workers and must not silently move review seats
180
181
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
181
- onSeat) {
182
+ onSeat, demoted = new Set()) {
182
183
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
183
184
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
184
185
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -188,20 +189,21 @@ onSeat) {
188
189
  return null;
189
190
  const authorProvider = modelProvider(author.model, authorChannel.vendor);
190
191
  const ranked = channels
191
- // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
192
- // rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
193
- // true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
192
+ // Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
193
+ // as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
194
+ // and different base-model identity (ADDED TO the vendor rule, never replacing it). The diversity
194
195
  // filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
195
196
  .filter((c) => c.vendor !== authorChannel.vendor
196
- && (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
197
+ && modelProvider(c.model, c.vendor) !== authorProvider
197
198
  && modelId(c.model) !== modelId(author.model)
198
199
  && !exclude.includes(channelKey(c))
199
200
  && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
200
201
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
201
- const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
202
+ const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
203
+ || history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
202
204
  || ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
203
205
  if (reviewer)
204
- onSeat?.(ranked.indexOf(reviewer) + 1);
206
+ onSeat?.(ranked.indexOf(reviewer) + 1, ranked.length);
205
207
  return reviewer;
206
208
  }
207
209
  /**
@@ -220,7 +222,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
220
222
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
221
223
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
222
224
  // direct tests) skips persistence and changes nothing else.
223
- artifactDir, reviewHistory) {
225
+ artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
224
226
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
225
227
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
226
228
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -241,8 +243,9 @@ artifactDir, reviewHistory) {
241
243
  // so production rounds have journaled both siblings all along — only the fixtures were blind to it.
242
244
  // Fixed in the ledger rather than in the oracles, because determinism run-to-run is a property of
243
245
  // the journal, not of three test files that happen to assert it.
246
+ const priorMaterials = carriedFindings.filter((finding) => finding.class === "review:material");
244
247
  const declaredPolicy = declaredReviewPolicy(task.files);
245
- const policy = raiseReviewPolicy(declaredPolicy, cfg.review.policy);
248
+ const policy = priorMaterials.length ? "full" : raiseReviewPolicy(declaredPolicy, cfg.review.policy);
246
249
  // PROMOTION: the declared assignment is a claim about paths, and the diff is the evidence. A
247
250
  // judge-only task whose diff left the leaf class is reviewed in full — the claim never outranks
248
251
  // what actually happened, and an empty diff promotes too (a skip earned by an absence is not earned).
@@ -293,15 +296,15 @@ artifactDir, reviewHistory) {
293
296
  // historical seat for every task that never asked for review-tier coupling.
294
297
  const reviewerFloor = task.routingHints?.floor;
295
298
  let rotationSeat;
296
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
299
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
297
300
  if (!reviewer) {
298
301
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
299
302
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
300
303
  const reason = reviewerFloor
301
304
  ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
302
305
  : "no cross-vendor reviewer available (diversity rule)";
303
- return cfg.review.required
304
- ? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
306
+ return cfg.review.required || priorMaterials.length > 0
307
+ ? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
305
308
  : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
306
309
  }
307
310
  reviewHistory?.push(channelKey(reviewer));
@@ -327,7 +330,10 @@ ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
327
330
 
328
331
  ${renderDeclaredWriteScope(task.files)}
329
332
 
330
- ## Diff
333
+ ${priorMaterials.length ? `## Prior materials this attempt must close
334
+ ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}\n${finding.note}`).join("\n\n")}
335
+
336
+ ` : ""}## Diff
331
337
  \`\`\`diff
332
338
  ${diff}
333
339
  \`\`\`
@@ -340,18 +346,31 @@ block approval. For a minor concern you have decided not to block on, set "defer
340
346
  one-line "rationale" — it is recorded in the review, never dropped.
341
347
 
342
348
  Respond with ONLY this JSON:
343
- {"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
344
- Approve iff no material finding remains; an empty findings list is a clean approval.
349
+ {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
350
+ For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
351
+ (still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
352
+ Approve iff no material finding remains and every prior material is resolved.
345
353
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
346
354
  `;
347
- let concludedOnInactivity = false;
355
+ const artifactId = `${task.id}-${nonce}`;
356
+ const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
357
+ // Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
358
+ let savedBrief;
359
+ if (briefPath) {
360
+ try {
361
+ writeFileSync(briefPath, redactSecrets(prompt));
362
+ savedBrief = briefPath;
363
+ }
364
+ catch {
365
+ savedBrief = undefined;
366
+ }
367
+ }
348
368
  const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
349
369
  driver: via.driver,
350
370
  keep: via.keep,
351
371
  onSlot: via.onSlot,
352
372
  name: via.nameFor("review", reviewer.adapter),
353
373
  label: via.labelFor("review"),
354
- onInactivity: () => { concludedOnInactivity = true; },
355
374
  } : undefined,
356
375
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
357
376
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
@@ -362,53 +381,80 @@ The top-level comments array is optional. Use it only for actionable line-anchor
362
381
  const provider = modelProvider(reviewer.model, reviewer.vendor);
363
382
  const v = extractVerdictJson(raw, nonce);
364
383
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
384
+ const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
385
+ const closureLists = [v?.resolved, v?.reraised];
386
+ const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
387
+ || new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
388
+ || [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
365
389
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
366
- if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
390
+ if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
367
391
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
368
392
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
369
- const cause = classifyVerdictCause(raw, nonce, "approve", llm);
370
- const bytes = Buffer.byteLength(raw, "utf8");
393
+ const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
394
+ const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
395
+ : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
396
+ : classifyVerdictCause(raw, nonce, "approve", llm);
371
397
  let saved;
372
398
  if (artifactDir) {
373
399
  try {
374
- saved = join(artifactDir, `review-raw-${task.id}-${Date.now()}.txt`);
400
+ saved = join(artifactDir, `review-raw-${artifactId}.txt`);
375
401
  writeFileSync(saved, redactSecrets(raw));
376
402
  }
377
403
  catch {
378
404
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
379
405
  }
380
406
  }
381
- const failure = concludedOnInactivity
382
- ? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
383
- : cause === "malformed-verdict"
384
- ? "review output unparseable"
385
- : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
407
+ const failure = cause === "malformed-verdict"
408
+ ? "review output unparseable"
409
+ : "review dispatch failed — no structurally valid nonce-bound response";
386
410
  return {
387
411
  gate: "review",
388
412
  pass: false,
389
- details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
413
+ details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${llm.timedOut ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
390
414
  meta: {
391
415
  ...policyMeta,
392
416
  ...rotationMeta,
393
417
  reviewer: channelKey(reviewer),
394
418
  vendor: reviewer.vendor,
395
419
  provider,
396
- unparseable: true,
420
+ ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
397
421
  cause,
398
- ...(cause === "empty-output" ? { bytes } : {}),
399
- ...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
400
- ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
422
+ bytes, seatAuthoredBytes: bytes,
423
+ ...(saved ? { rawPath: saved } : {}),
424
+ ...(savedBrief ? { briefPath: savedBrief } : {}),
425
+ ...(llm.timedOut ? { timeoutMs: cfg.review.timeoutMs } : {}),
401
426
  },
402
427
  };
403
428
  }
404
429
  const decided = findings !== null
405
430
  ? classifyReviewFindings(findings)
406
431
  : classifyReviewIssues(v.approve, v.issues);
432
+ const reraised = priorMaterials.filter((finding) => v.reraised?.includes(finding.fingerprint));
433
+ if (reraised.length) {
434
+ if (decided.pass)
435
+ decided.headline = "requested changes";
436
+ decided.pass = false;
437
+ // A reviewer may also restate a re-raised material in findings. Preserve the original
438
+ // prose once so an unchanged defect keeps the same failure brief across repair rounds.
439
+ for (const finding of reraised) {
440
+ const line = `- [material] ${finding.note}`;
441
+ if (!decided.lines.includes(line))
442
+ decided.lines.push(line);
443
+ }
444
+ }
407
445
  const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
446
+ const details = appendAnchoredReview(prose, v);
408
447
  return {
409
448
  gate: "review",
410
449
  pass: decided.pass,
411
- details: appendAnchoredReview(prose, v),
412
- meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
450
+ details,
451
+ meta: {
452
+ ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
453
+ ...(priorMaterials.length ? { resolved: v.resolved, reraised: v.reraised } : {}),
454
+ ...(reraised.length ? { findings: [
455
+ ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
456
+ ...reraised,
457
+ ] } : {}),
458
+ },
413
459
  };
414
460
  }
@@ -4,6 +4,7 @@ import { type GateName, type Task } from "../graph/schema.js";
4
4
  import { type Baseline } from "./baseline.js";
5
5
  import { type GateVia } from "./llm.js";
6
6
  import type { GateResult } from "./types.js";
7
+ import { type StructuredFinding } from "../run/journal.js";
7
8
  export type LoadProvider = () => number;
8
9
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
9
10
  export declare function setLoadProviderForTests(provider: LoadProvider): void;
@@ -52,7 +53,9 @@ export interface GateContext {
52
53
  adapters: WorkerAdapter[];
53
54
  cfg: TickmarkrConfig;
54
55
  via?: GateVia;
56
+ carriedFindings?: readonly StructuredFinding[];
55
57
  excludeReviewers?: string[];
58
+ demotedReviewers?: Set<string>;
56
59
  reviewHistory?: string[];
57
60
  artifactDir?: string;
58
61
  pipeline?: "v185" | "legacy";
@@ -590,13 +590,23 @@ export async function runGates(task, ctx) {
590
590
  const dispatch = async (run) => {
591
591
  const captured = await captureLlmDispatches(ctx.adapters, run);
592
592
  invocations.push(...captured.invocations);
593
- return captured.value;
593
+ const rv = captured.value;
594
+ if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
595
+ await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
596
+ if (rv.meta.seatAuthoredBytes === 0 && typeof rv.meta.reviewer === "string"
597
+ && !ctx.demotedReviewers?.has(rv.meta.reviewer)) {
598
+ ctx.demotedReviewers?.add(rv.meta.reviewer);
599
+ await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
600
+ payload: { reviewer: rv.meta.reviewer, cause: rv.meta.cause, seatAuthoredBytes: 0 }, result: rv });
601
+ }
602
+ }
603
+ return rv;
594
604
  };
595
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
605
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
596
606
  // OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
597
607
  // different adapter. Only a single-adapter eligible pool may fall back to another channel on the
598
608
  // flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
599
- if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
609
+ if ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
600
610
  const flaked = rv.meta.reviewer;
601
611
  const emptyOutput = rv.meta.cause === "empty-output";
602
612
  if (emptyOutput) {
@@ -615,7 +625,7 @@ export async function runGates(task, ctx) {
615
625
  const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
616
626
  const exclusion = crossAdapter ? "adapter" : "channel";
617
627
  const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
618
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
628
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
619
629
  if (second.meta?.noEligibleReviewer !== true) {
620
630
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
621
631
  const route = exclusion === "adapter"
@@ -1,12 +1,14 @@
1
1
  import type { WorkerAdapter } from "../adapters/types.js";
2
2
  import type { TickmarkrConfig } from "../config/config.js";
3
3
  import type { ExecutorDriver } from "../drivers/types.js";
4
+ export declare const MAX_SCOPE_ATTEMPTS = 3;
4
5
  export declare function clarificationGate(intent: string): string[];
5
6
  export interface ScopeOptions {
6
7
  cfg: TickmarkrConfig;
7
8
  adapters: WorkerAdapter[];
8
9
  driver?: ExecutorDriver;
9
10
  force?: boolean;
11
+ candidate?: ScopeCandidate;
10
12
  }
11
13
  export interface ScopeResult {
12
14
  specFile: string;
@@ -14,4 +16,27 @@ export interface ScopeResult {
14
16
  attempts: number;
15
17
  }
16
18
  export declare function specPathForIntent(intentFile: string): string;
19
+ export interface ScopeCandidate {
20
+ adapter: string;
21
+ model: string;
22
+ }
23
+ export interface ScopePreview {
24
+ intentFile: string;
25
+ specFile: string;
26
+ specExists: boolean;
27
+ cached: boolean;
28
+ candidate?: ScopeCandidate;
29
+ authoringBudget: number;
30
+ probeCalls: number;
31
+ }
32
+ /**
33
+ * R11/R45 (C10): local-only disclosure — intent/clarification checks and a candidate read off the
34
+ * doctor cache, never a fresh probe or a model turn. `readDoctor` and `discoverChannels`/`route` are
35
+ * pure reads over that cache, so this never touches an adapter.
36
+ */
37
+ export declare function previewScope(intentFile: string, repoRoot: string, options: {
38
+ cfg: TickmarkrConfig;
39
+ adapters: WorkerAdapter[];
40
+ }): ScopePreview;
41
+ export declare function formatScopePreview(preview: ScopePreview): string;
17
42
  export declare function scopeIntent(intentFile: string, repoRoot: string, options: ScopeOptions): Promise<ScopeResult>;