tickmarkr 2.2.1 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/catalog-remote.d.ts +1 -4
  3. package/dist/adapters/catalog-remote.js +52 -42
  4. package/dist/adapters/catalog.js +5 -3
  5. package/dist/adapters/claude-code.d.ts +1 -1
  6. package/dist/adapters/claude-code.js +8 -5
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/types.d.ts +21 -1
  14. package/dist/adapters/types.js +43 -2
  15. package/dist/cli/commands/approve.js +5 -4
  16. package/dist/cli/commands/beat.js +7 -4
  17. package/dist/cli/commands/compile.js +32 -6
  18. package/dist/cli/commands/doctor.d.ts +9 -4
  19. package/dist/cli/commands/doctor.js +87 -13
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +53 -14
  22. package/dist/cli/commands/init.js +36 -21
  23. package/dist/cli/commands/plan.js +45 -7
  24. package/dist/cli/commands/report.js +37 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +45 -1
  27. package/dist/cli/commands/verify.d.ts +6 -0
  28. package/dist/cli/commands/verify.js +145 -25
  29. package/dist/cli/index.d.ts +1 -1
  30. package/dist/cli/index.js +2 -2
  31. package/dist/compile/collateral.d.ts +2 -9
  32. package/dist/compile/collateral.js +17 -18
  33. package/dist/compile/index.d.ts +4 -1
  34. package/dist/compile/index.js +41 -7
  35. package/dist/compile/native.d.ts +4 -2
  36. package/dist/compile/native.js +58 -9
  37. package/dist/compile/ownership.js +34 -9
  38. package/dist/config/config.d.ts +1 -0
  39. package/dist/config/config.js +52 -6
  40. package/dist/drivers/herdr.d.ts +1 -0
  41. package/dist/drivers/herdr.js +11 -1
  42. package/dist/drivers/index.d.ts +6 -0
  43. package/dist/drivers/index.js +19 -4
  44. package/dist/drivers/orca.d.ts +35 -1
  45. package/dist/drivers/orca.js +260 -20
  46. package/dist/drivers/subprocess.d.ts +3 -3
  47. package/dist/drivers/subprocess.js +16 -9
  48. package/dist/drivers/types.d.ts +12 -0
  49. package/dist/gates/baseline.d.ts +2 -0
  50. package/dist/gates/baseline.js +47 -11
  51. package/dist/gates/llm.d.ts +6 -0
  52. package/dist/gates/llm.js +25 -9
  53. package/dist/gates/review.d.ts +7 -3
  54. package/dist/gates/review.js +61 -22
  55. package/dist/gates/run-gates.d.ts +5 -2
  56. package/dist/gates/run-gates.js +50 -26
  57. package/dist/gates/verdict-cause.d.ts +6 -2
  58. package/dist/gates/verdict-cause.js +8 -4
  59. package/dist/route/preference.d.ts +4 -0
  60. package/dist/route/preference.js +40 -0
  61. package/dist/route/router.js +15 -2
  62. package/dist/run/consult.d.ts +1 -0
  63. package/dist/run/consult.js +39 -8
  64. package/dist/run/daemon.d.ts +16 -0
  65. package/dist/run/daemon.js +345 -74
  66. package/dist/run/git.d.ts +3 -0
  67. package/dist/run/git.js +40 -5
  68. package/dist/run/journal.d.ts +15 -2
  69. package/dist/run/journal.js +73 -12
  70. package/dist/run/supervision.d.ts +6 -0
  71. package/dist/run/supervision.js +29 -1
  72. package/dist/tui/ink/fleet-app.d.ts +4 -0
  73. package/dist/tui/ink/fleet-app.js +45 -16
  74. package/dist/tui/ink/init-app.js +4 -4
  75. package/package.json +59 -1
  76. package/skills/tickmarkr-overseer/SKILL.md +77 -18
  77. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  78. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  79. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  80. package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
  81. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
@@ -5,10 +5,19 @@ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "..
5
5
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
6
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
7
7
  // baselines stored by pre-hardening code.
8
- const ANSI_RE = /\x1b\[[\d;#]*[A-Za-z]/g;
8
+ // OBS-891 (run 3372): vitest toggles the cursor (`\x1b[?25l` / `\x1b[?25h`) around its progress
9
+ // output, and the private-mode parameter byte `?` never matched [\d;#], so an echo-block HEADER glued
10
+ // to a cursor-show sequence stayed invisible to withoutVitestEchoBlocks and its whole block leaked as
11
+ // runner evidence — seven prose-only "infra" parks in one night. Full CSI grammar: parameter bytes
12
+ // 0x30–0x3F (plus `#` for digit-normalized stored baselines), intermediates 0x20–0x2F, final 0x40–0x7E.
13
+ const ANSI_RE = /\x1b\[[0-?#]*[ -/]*[@-~]/g;
9
14
  // ponytail: only leading ✓/✔ after optional "label:" prefixes (turbo/vitest), or tickmarkr's own run
10
15
  // summary, counts as a pass line — other runners' pass markers (PASS, ok) stay fingerprintable
11
16
  const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:tickmarkr\s+[\w.-]+:\s+)?(?:\d+|#)\s+done,\s+(?:\d+|#)\s+failed(?:,\s+(?:\d+|#)\s+awaiting human)?\b)/;
17
+ // OBS-888: tickmarkr's own operator lines (`tickmarkr: baseline capture for "test" … spawn EAGAIN`)
18
+ // are printed by this product, never by a runner about the work. When this repository's tests exercise
19
+ // the capture path they print them too, carrying errno tokens INFRA_RE would read as host evidence.
20
+ const OPERATOR_LINE_RE = /^\s*tickmarkr: /;
12
21
  // HYG-08 (D-01, incident run-20260711-154920): a failing test went unnamed for 3 attempts because details
13
22
  // headlined benign fingerprint-diff noise. These anchors harvest the runner's OWN failure naming from fresh
14
23
  // output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
@@ -149,7 +158,10 @@ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
149
158
  * unreadable-runner case the existing fail-closed path already owns.
150
159
  */
151
160
  export function classifyFailureOutput(output) {
152
- const lines = output.split("\n").map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l));
161
+ // OBS-891: the gate reads the WHOLE output here when the fresh-fingerprint diff is empty, so an errno
162
+ // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
163
+ // to fingerprint(): test-owned output is never runner evidence about the work.
164
+ const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
153
165
  if (lines.some(namesRegression))
154
166
  return "regression";
155
167
  return lines.some(isInfraLine) ? "infra" : undefined;
@@ -161,7 +173,11 @@ export function classifyFailureOutput(output) {
161
173
  * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
162
174
  * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
163
175
  */
164
- const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+\S+\.(?:test|spec)\.[cm]?[jt]sx?\s+>\s+\S/;
176
+ // OBS-888 row 1: vitest 3.2.7 heads an echo block `std{out,err} | <file> > <test>` when the log is
177
+ // attributed to a test, `stderr | unknown test` when it is not, `stderr | <file>` for file-level output
178
+ // and `stderr | <task id>` (digits and underscores) when the reporter no longer knows the task
179
+ // (dist/chunks/index.*.js, `headerText`). The stripper knew only the first form.
180
+ const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+(?:unknown test\s*$|\d+_[\d_]+\s*$|\S+\.(?:test|spec)\.[cm]?[jt]sx?(?:\s*$|\s+>\s+\S))/;
165
181
  /** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
166
182
  const withoutVitestEchoBlocks = (output) => {
167
183
  const outside = [];
@@ -186,7 +202,7 @@ const captureInvalidatingLines = (output) => {
186
202
  const invalidating = [];
187
203
  for (const line of withoutVitestEchoBlocks(output)) {
188
204
  const clean = line.replace(ANSI_RE, "");
189
- if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
205
+ if (!PASS_LINE_RE.test(clean) && !OPERATOR_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
190
206
  invalidating.push(line);
191
207
  }
192
208
  return invalidating;
@@ -230,16 +246,18 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
230
246
  export function fingerprint(output) {
231
247
  const lines = withoutVitestEchoBlocks(output)
232
248
  .map((l) => l.replace(ANSI_RE, ""))
233
- .filter((l) => !PASS_LINE_RE.test(l));
249
+ .filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
234
250
  // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
235
251
  // prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
236
252
  // failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
237
253
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
238
254
  const shaped = [];
239
255
  for (const l of lines) {
256
+ const stripped = stripTurboPrefix(l);
257
+ if (stripped !== undefined && OPERATOR_LINE_RE.test(stripped))
258
+ continue; // OBS-888: operator prose under a turbo prefix
240
259
  if (isFingerprintShaped(l))
241
260
  shaped.push(l);
242
- const stripped = stripTurboPrefix(l);
243
261
  if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
244
262
  shaped.push(stripped);
245
263
  }
@@ -427,8 +445,9 @@ export function ceilingKillResult(gate, r, ceilingMs) {
427
445
  * shell asks it to finish faster than the thing it is measuring.
428
446
  */
429
447
  export const CAPTURE_CEILING_MS = 1_800_000;
430
- const invalidCaptureEntry = (durationMs, invalidatingLines = []) => ({
448
+ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) => ({
431
449
  infra: true,
450
+ invalidCause,
432
451
  fingerprints: [],
433
452
  durationMs,
434
453
  fileDurationSumMs: null,
@@ -463,7 +482,7 @@ export async function captureBaseline(cwd, commands) {
463
482
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
464
483
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
465
484
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
466
- base.commands[name] = invalidCaptureEntry(durationMs);
485
+ base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
467
486
  continue;
468
487
  }
469
488
  const combinedOutput = r.stdout + "\n" + r.stderr;
@@ -479,7 +498,7 @@ export async function captureBaseline(cwd, commands) {
479
498
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
480
499
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
481
500
  + `First invalidating line: ${invalidatingLines[0]}`);
482
- base.commands[name] = invalidCaptureEntry(durationMs, invalidatingLines);
501
+ base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
483
502
  continue;
484
503
  }
485
504
  base.commands[name] = {
@@ -575,7 +594,8 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
575
594
  // result rather than re-derived after the fact. The skip row above ran no command and therefore
576
595
  // states no capacity — a row that never divided the machine must not claim that it did.
577
596
  const record = (g) => {
578
- results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
597
+ const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
598
+ results.push(r.capacity ? { ...withReap, capacity: r.capacity } : withReap);
579
599
  };
580
600
  // …and whether the entry that would forgive this command was measured in the same world. A
581
601
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -591,6 +611,19 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
591
611
  record(killed);
592
612
  continue;
593
613
  }
614
+ const runner = shellToken(cmd) ?? name;
615
+ if (entry?.missingCommand === true || r.code === 127) {
616
+ const cause = entry?.missingCommand === true && r.code !== 127
617
+ ? "the baseline capture recorded this runner as missing"
618
+ : `the head command exited 127${/\bENOENT\b/i.test(`${r.stdout}\n${r.stderr}`) ? " (spawn ENOENT)" : ""}`;
619
+ record({
620
+ gate: name,
621
+ pass: false,
622
+ details: `unreadable — ${cause}; runner ${JSON.stringify(runner)} did not produce a trustworthy verdict`,
623
+ meta: { unreadable: true, runner },
624
+ });
625
+ continue;
626
+ }
594
627
  if (r.code === 0) {
595
628
  record({ gate: name, pass: true, details: "exit 0" });
596
629
  continue;
@@ -629,8 +662,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
629
662
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
630
663
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
631
664
  if (!failing.length && !baselineRed) {
665
+ const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
666
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
667
+ : "was killed at its ceiling";
632
668
  const closed = entry?.infra === true
633
- ? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
669
+ ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
634
670
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
635
671
  const evidence = unrecognizedEvidence(raw);
636
672
  record({
@@ -57,9 +57,15 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
57
57
  value: T;
58
58
  outputs: string[];
59
59
  }>;
60
+ export interface LlmRunResult {
61
+ output: string;
62
+ exitCode?: number;
63
+ timedOut: boolean;
64
+ }
60
65
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
61
66
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
62
67
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
68
+ export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
63
69
  export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
64
70
  export declare function extractJson<T>(raw: string): T | null;
65
71
  /** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
package/dist/gates/llm.js CHANGED
@@ -129,21 +129,24 @@ export async function captureLlmOutput(run) {
129
129
  const value = await llmOutputCapture.run(outputs, run);
130
130
  return { value, outputs };
131
131
  }
132
- export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
132
+ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
133
133
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
134
134
  try {
135
135
  const pf = join(dir, "prompt.md");
136
136
  writeFileSync(pf, prompt);
137
137
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
138
- return r.stdout + "\n" + r.stderr;
138
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
139
139
  }
140
140
  finally {
141
141
  rmSync(dir, { recursive: true, force: true });
142
142
  }
143
143
  }
144
+ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
145
+ return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
146
+ }
144
147
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
145
148
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
146
- export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
149
+ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
147
150
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
148
151
  let slot;
149
152
  let accountant;
@@ -176,6 +179,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
176
179
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
177
180
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
178
181
  let out;
182
+ let timedOut = false;
179
183
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
180
184
  if (!gatePrompt) {
181
185
  await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
@@ -241,8 +245,14 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
241
245
  break;
242
246
  }
243
247
  }
248
+ timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
244
249
  }
245
- return dewrapPaneVerdict(out, nonce);
250
+ const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
251
+ return {
252
+ output: dewrapPaneVerdict(out, nonce),
253
+ ...(Number.isFinite(exitCode) ? { exitCode } : {}),
254
+ timedOut,
255
+ };
246
256
  }
247
257
  finally {
248
258
  try {
@@ -256,6 +266,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
256
266
  }
257
267
  }
258
268
  }
269
+ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
270
+ return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
271
+ }
259
272
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
260
273
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
261
274
  // literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
@@ -315,12 +328,15 @@ export function dewrapPaneVerdict(out, nonce) {
315
328
  }
316
329
  return out;
317
330
  }
331
+ export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
332
+ const result = await (via
333
+ ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
334
+ : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
335
+ llmOutputCapture.getStore()?.push(result.output);
336
+ return result;
337
+ }
318
338
  export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
319
- const out = await (via
320
- ? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
321
- : runHeadless(adapter, model, prompt, cwd, timeoutMs));
322
- llmOutputCapture.getStore()?.push(out);
323
- return out;
339
+ return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
324
340
  }
325
341
  export function extractJson(raw) {
326
342
  const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
@@ -41,16 +41,20 @@ export type TaskDiffMeasurement = {
41
41
  readonly fullMeasurement: ArtifactDiffMeasurement;
42
42
  readonly capMeasurement: ArtifactDiffMeasurement;
43
43
  };
44
- export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
44
+ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?: readonly string[]): Promise<TaskDiffMeasurement>;
45
45
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
46
46
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
47
47
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
48
48
  export declare function isDiffCapPark(result: GateResult): boolean;
49
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
50
  export declare function modelId(model: string): string;
51
+ /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
52
+ export declare function modelProvider(model: string, fallback?: string): string;
51
53
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
52
54
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
53
- floor?: Tier): BillingChannel | null;
55
+ floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
56
+ history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
57
+ onSeat?: (seat: number) => void): BillingChannel | null;
54
58
  export type ReviewUnparseableCause = VerdictUnparseableCause;
55
59
  /**
56
60
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -58,4 +62,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause;
58
62
  * judgement rather than a guarantee made by this renderer.
59
63
  */
60
64
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
61
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
65
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
@@ -2,12 +2,13 @@ import { writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
+ import { filesGlob } from "../graph/files-glob.js";
5
6
  import { renderAcceptanceItem } from "../graph/schema.js";
6
7
  import { getAdapter } from "../adapters/registry.js";
7
8
  import { shOk } from "../run/git.js";
8
9
  import { redactSecrets } from "../run/redact.js";
9
10
  import { marginalCostRank } from "../route/router.js";
10
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
11
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
11
12
  import { classifyVerdictCause } from "./verdict-cause.js";
12
13
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
13
14
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -99,12 +100,14 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
99
100
  const changed = diff.split("\n").filter((l) => /^[+-]/.test(l) && !/^(?:\+\+\+|---)/.test(l));
100
101
  return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
101
102
  }
102
- export async function fetchTaskDiff(worktree, baseRef) {
103
+ export async function fetchTaskDiff(worktree, baseRef, files = []) {
103
104
  // --full-index: abbreviated index lines vary with object-store density, so two measurements of
104
105
  // the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
105
- const [rawFull, rawForCap] = await Promise.all([
106
- shOk(`git diff --full-index '${baseRef}..HEAD'`, worktree),
107
- shOk(`git diff --full-index -U0 '${baseRef}..HEAD'`, worktree),
106
+ const matched = files.length ? (await changedPaths(worktree, baseRef)).filter(filesGlob([...files])) : [];
107
+ const pathspec = matched.length ? ` -- ${matched.map(shq).join(" ")}` : "";
108
+ const [rawFull, rawForCap] = files.length && matched.length === 0 ? ["", ""] : await Promise.all([
109
+ shOk(`git diff --full-index '${baseRef}..HEAD'${pathspec}`, worktree),
110
+ shOk(`git diff --full-index -U0 '${baseRef}..HEAD'${pathspec}`, worktree),
108
111
  ]);
109
112
  const fullMeasurement = measureArtifactDiff(rawFull);
110
113
  const capMeasurement = measureArtifactDiff(rawForCap);
@@ -162,6 +165,24 @@ export function diffCapParkReason(results) {
162
165
  export function modelId(model) {
163
166
  return model.slice(model.lastIndexOf("/") + 1);
164
167
  }
168
+ /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
169
+ export function modelProvider(model, fallback = "unknown") {
170
+ const id = model.toLowerCase();
171
+ const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
172
+ if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
173
+ return "openai";
174
+ if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
175
+ return "anthropic";
176
+ if (prefix === "google" || /^gemini(?:-|$)/.test(id))
177
+ return "google";
178
+ if (prefix === "xai" || /^grok(?:-|$)/.test(id))
179
+ return "xai";
180
+ if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
181
+ return "zhipu";
182
+ if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
183
+ return "moonshot";
184
+ return fallback;
185
+ }
165
186
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
166
187
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
167
188
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -171,7 +192,9 @@ function reviewPreferIndex(c, prefer) {
171
192
  }
172
193
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
173
194
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
174
- floor) {
195
+ floor, // task-declared only; config floors govern workers and must not silently move review seats
196
+ history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
197
+ onSeat) {
175
198
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
176
199
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
177
200
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -179,15 +202,23 @@ floor) {
179
202
  const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
180
203
  if (!authorChannel)
181
204
  return null;
182
- return (channels
205
+ const authorProvider = modelProvider(author.model, authorChannel.vendor);
206
+ const ranked = channels
183
207
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
184
- // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
185
- // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
208
+ // rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
209
+ // true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
210
+ // filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
186
211
  .filter((c) => c.vendor !== authorChannel.vendor
212
+ && (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
187
213
  && modelId(c.model) !== modelId(author.model)
188
214
  && !exclude.includes(channelKey(c))
189
215
  && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
190
- .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
216
+ .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
217
+ const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
218
+ || ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
219
+ if (reviewer)
220
+ onSeat?.(ranked.indexOf(reviewer) + 1);
221
+ return reviewer;
191
222
  }
192
223
  /**
193
224
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -205,7 +236,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
205
236
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
206
237
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
207
238
  // direct tests) skips persistence and changes nothing else.
208
- artifactDir) {
239
+ artifactDir, reviewHistory) {
209
240
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
210
241
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
211
242
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -277,7 +308,8 @@ artifactDir) {
277
308
  // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
278
309
  // historical seat for every task that never asked for review-tier coupling.
279
310
  const reviewerFloor = task.routingHints?.floor;
280
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
311
+ let rotationSeat;
312
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
281
313
  if (!reviewer) {
282
314
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
283
315
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
@@ -285,10 +317,12 @@ artifactDir) {
285
317
  ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
286
318
  : "no cross-vendor reviewer available (diversity rule)";
287
319
  return cfg.review.required
288
- ? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
320
+ ? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
289
321
  : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
290
322
  }
291
- const measuredDiff = await fetchTaskDiff(worktree, baseRef);
323
+ reviewHistory?.push(channelKey(reviewer));
324
+ const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
325
+ const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
292
326
  // Keep the reader payload identical to the text charged to the strict cap:
293
327
  // whole-file source deletions are represented by their citable operation fact.
294
328
  const diff = reviewableLogicDiff(measuredDiff.full);
@@ -327,7 +361,7 @@ Approve iff no material finding remains; an empty findings list is a clean appro
327
361
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
328
362
  `;
329
363
  let concludedOnInactivity = false;
330
- const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
364
+ const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
331
365
  driver: via.driver,
332
366
  keep: via.keep,
333
367
  onSlot: via.onSlot,
@@ -338,16 +372,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
338
372
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
339
373
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
340
374
  // stdout that read as "unparseable" and escalated to re-implementation of green code
341
- // (run-20260709-104447 P87-09). ponytail: literal 15min; make it cfg.review.timeoutMs if a
342
- // second knob-turner appears.
343
- 900_000);
375
+ // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
376
+ cfg.review.timeoutMs);
377
+ const raw = llm.output;
378
+ const provider = modelProvider(reviewer.model, reviewer.vendor);
344
379
  const v = extractVerdictJson(raw, nonce);
345
380
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
346
381
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
347
382
  if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
348
383
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
349
384
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
350
- const cause = classifyVerdictCause(raw, nonce, "approve");
385
+ const cause = classifyVerdictCause(raw, nonce, "approve", llm);
351
386
  const bytes = Buffer.byteLength(raw, "utf8");
352
387
  let saved;
353
388
  if (artifactDir) {
@@ -367,13 +402,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
367
402
  return {
368
403
  gate: "review",
369
404
  pass: false,
370
- details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
405
+ details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
371
406
  meta: {
372
407
  ...policyMeta,
408
+ ...rotationMeta,
373
409
  reviewer: channelKey(reviewer),
410
+ vendor: reviewer.vendor,
411
+ provider,
374
412
  unparseable: true,
375
413
  cause,
376
414
  ...(cause === "empty-output" ? { bytes } : {}),
415
+ ...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
377
416
  ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
378
417
  },
379
418
  };
@@ -381,11 +420,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
381
420
  const decided = findings !== null
382
421
  ? classifyReviewFindings(findings)
383
422
  : classifyReviewIssues(v.approve, v.issues);
384
- const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
423
+ const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
385
424
  return {
386
425
  gate: "review",
387
426
  pass: decided.pass,
388
427
  details: appendAnchoredReview(prose, v),
389
- meta: { ...policyMeta, reviewer: channelKey(reviewer) },
428
+ meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
390
429
  };
391
430
  }
@@ -13,13 +13,15 @@ export declare function resetLoadProviderForTests(): void;
13
13
  * intervals and nothing between them, so the composite `test` gate (a selected screen, then other
14
14
  * gates, then the full suite) reports the two suites' cost rather than the span containing them —
15
15
  * and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
16
- * queue as well as the work. The load samples bracket the FIRST interval's start and the LAST
17
- * interval's end: start is what a scheduler would have decided on, end is the state it left behind.
16
+ * queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
17
+ * start preserves the scheduling input while max and mean retain sustained interior saturation.
18
18
  */
19
19
  export interface GateTelemetry {
20
20
  durationMs: number;
21
21
  load1Start: number;
22
22
  load1End: number;
23
+ load1Max: number;
24
+ load1Mean: number;
23
25
  }
24
26
  export type GateEvent = {
25
27
  phase: "start";
@@ -51,6 +53,7 @@ export interface GateContext {
51
53
  cfg: TickmarkrConfig;
52
54
  via?: GateVia;
53
55
  excludeReviewers?: string[];
56
+ reviewHistory?: string[];
54
57
  artifactDir?: string;
55
58
  pipeline?: "v185" | "legacy";
56
59
  selectTests?: boolean;