tickmarkr 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/model-lints.js +9 -0
  3. package/dist/adapters/pi.d.ts +1 -0
  4. package/dist/adapters/pi.js +15 -1
  5. package/dist/adapters/prompt.js +11 -3
  6. package/dist/adapters/registry.js +13 -4
  7. package/dist/adapters/types.d.ts +23 -1
  8. package/dist/adapters/types.js +43 -2
  9. package/dist/cli/commands/approve.d.ts +3 -7
  10. package/dist/cli/commands/approve.js +26 -20
  11. package/dist/cli/commands/beat.js +7 -4
  12. package/dist/cli/commands/doctor.d.ts +6 -2
  13. package/dist/cli/commands/doctor.js +79 -9
  14. package/dist/cli/commands/init.js +36 -21
  15. package/dist/cli/commands/plan.js +20 -3
  16. package/dist/cli/commands/report.js +37 -1
  17. package/dist/cli/commands/verify.d.ts +5 -0
  18. package/dist/cli/commands/verify.js +142 -25
  19. package/dist/cli/commands/version.d.ts +2 -1
  20. package/dist/cli/commands/version.js +25 -4
  21. package/dist/compile/collateral.js +15 -9
  22. package/dist/compile/native.js +5 -3
  23. package/dist/config/config.js +1 -1
  24. package/dist/drivers/index.d.ts +7 -0
  25. package/dist/drivers/index.js +40 -10
  26. package/dist/drivers/orca.d.ts +41 -1
  27. package/dist/drivers/orca.js +192 -15
  28. package/dist/drivers/subprocess.d.ts +3 -3
  29. package/dist/drivers/subprocess.js +16 -9
  30. package/dist/drivers/types.d.ts +2 -0
  31. package/dist/gates/baseline.d.ts +4 -0
  32. package/dist/gates/baseline.js +68 -16
  33. package/dist/gates/llm.d.ts +7 -1
  34. package/dist/gates/llm.js +66 -35
  35. package/dist/gates/review.d.ts +3 -1
  36. package/dist/gates/review.js +42 -12
  37. package/dist/gates/run-gates.d.ts +6 -0
  38. package/dist/gates/run-gates.js +25 -10
  39. package/dist/gates/verdict-cause.d.ts +6 -2
  40. package/dist/gates/verdict-cause.js +8 -4
  41. package/dist/run/consult.d.ts +7 -0
  42. package/dist/run/consult.js +21 -3
  43. package/dist/run/daemon.d.ts +14 -0
  44. package/dist/run/daemon.js +249 -27
  45. package/dist/run/git.d.ts +1 -0
  46. package/dist/run/git.js +4 -0
  47. package/dist/run/journal.d.ts +15 -2
  48. package/dist/run/journal.js +70 -12
  49. package/dist/run/supervision.d.ts +6 -0
  50. package/dist/run/supervision.js +29 -1
  51. package/dist/tui/ink/init-app.js +4 -4
  52. package/package.json +1 -1
  53. package/skills/tickmarkr-loop/SKILL.md +1 -0
  54. package/skills/tickmarkr-overseer/SKILL.md +88 -18
  55. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  56. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  57. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  58. package/skills/tickmarkr-overseer/scripts/watch-context.sh +35 -9
  59. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
@@ -23,6 +23,8 @@ export interface BaselineCommand {
23
23
  exitCode?: number;
24
24
  fingerprints: string[];
25
25
  missingCommand?: boolean;
26
+ /** Why a capture returned no verdict. */
27
+ invalidCause?: "ceiling-kill" | "resource-exhaustion";
26
28
  /** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
27
29
  durationMs?: number;
28
30
  /** Sum of the per-file durations named by the runner; null when its output names none. */
@@ -45,6 +47,8 @@ export interface BaselineCommand {
45
47
  * verdict: no exit code, no fingerprints, nothing forgivable.
46
48
  */
47
49
  infra?: true;
50
+ /** Runner-level exhaustion lines that invalidated a completed capture. */
51
+ invalidatingLines?: string[];
48
52
  }
49
53
  export interface BaselineFileDuration {
50
54
  file: string;
@@ -85,7 +85,7 @@ const TURBO_FAIL_RE = /^\s*(?:[\w@./-]+:\s*)*ELIFECYCLE\s+Command failed\b|^\s*F
85
85
  // at least TWO colon-joined segments (`<pkg>:<task>:`, tasks may nest — `pkg:test:unit:`), every
86
86
  // segment after the first starting with a letter — so `Error: boom` (one segment), `src/x.ts:12:`
87
87
  // (digit segment) and `12:34 error` (eslint stylish) are never stripped.
88
- const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:\s+/;
88
+ const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:(?:\s+|$)/;
89
89
  const stripTurboPrefix = (l) => {
90
90
  const m = TURBO_PREFIX_RE.exec(l);
91
91
  return m ? l.slice(m[0].length) : undefined;
@@ -161,11 +161,36 @@ export function classifyFailureOutput(output) {
161
161
  * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
162
162
  * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
163
163
  */
164
- const captureHasInvalidatingInfra = (output) => output
165
- .split("\n")
166
- .map((l) => l.replace(ANSI_RE, ""))
167
- .filter((l) => !PASS_LINE_RE.test(l))
168
- .some((l) => CAPTURE_EXHAUSTION_RE.test(l));
164
+ const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+\S+\.(?:test|spec)\.[cm]?[jt]sx?\s+>\s+\S/;
165
+ /** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
166
+ const withoutVitestEchoBlocks = (output) => {
167
+ const outside = [];
168
+ let inVitestEchoBlock = false;
169
+ for (const line of output.split("\n")) {
170
+ const clean = line.replace(ANSI_RE, "");
171
+ const runnerLine = stripTurboPrefix(clean) ?? clean;
172
+ if (inVitestEchoBlock) {
173
+ if (runnerLine.trim() === "")
174
+ inVitestEchoBlock = false;
175
+ continue;
176
+ }
177
+ if (VITEST_ECHO_BLOCK_RE.test(runnerLine)) {
178
+ inVitestEchoBlock = true;
179
+ continue;
180
+ }
181
+ outside.push(line);
182
+ }
183
+ return outside;
184
+ };
185
+ const captureInvalidatingLines = (output) => {
186
+ const invalidating = [];
187
+ for (const line of withoutVitestEchoBlocks(output)) {
188
+ const clean = line.replace(ANSI_RE, "");
189
+ if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
190
+ invalidating.push(line);
191
+ }
192
+ return invalidating;
193
+ };
169
194
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
170
195
  // Vitest's default reporter names file durations as
171
196
  // `✓ |project| tests/example.test.ts (12 tests) 1.23s` (❯ for a red file). The runner may be
@@ -203,8 +228,7 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
203
228
  // renders "zone +3 · run attempt 2 · 1 failed" (what T9 is chartered to draw) contributes nothing, with
204
229
  // or without box glyphs, so it cannot manufacture a fresh-fingerprint rejection on any future attempt.
205
230
  export function fingerprint(output) {
206
- const lines = output
207
- .split("\n")
231
+ const lines = withoutVitestEchoBlocks(output)
208
232
  .map((l) => l.replace(ANSI_RE, ""))
209
233
  .filter((l) => !PASS_LINE_RE.test(l));
210
234
  // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
@@ -403,14 +427,16 @@ export function ceilingKillResult(gate, r, ceilingMs) {
403
427
  * shell asks it to finish faster than the thing it is measuring.
404
428
  */
405
429
  export const CAPTURE_CEILING_MS = 1_800_000;
406
- const invalidCaptureEntry = (durationMs) => ({
430
+ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) => ({
407
431
  infra: true,
432
+ invalidCause,
408
433
  fingerprints: [],
409
434
  durationMs,
410
435
  fileDurationSumMs: null,
411
436
  impliedParallelism: null,
412
437
  longestFile: null,
413
438
  ceilingMs: effectiveCeilingMs({ durationMs }),
439
+ ...(invalidatingLines.length ? { invalidatingLines } : {}),
414
440
  });
415
441
  export async function captureBaseline(cwd, commands) {
416
442
  const base = { commands: {} };
@@ -438,20 +464,23 @@ export async function captureBaseline(cwd, commands) {
438
464
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
439
465
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
440
466
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
441
- base.commands[name] = invalidCaptureEntry(durationMs);
467
+ base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
442
468
  continue;
443
469
  }
444
- const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
470
+ const combinedOutput = r.stdout + "\n" + r.stderr;
471
+ const raw = combinedOutput.split(cwd).join("");
445
472
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
446
473
  // discriminator correctly called the mixed output a regression. But the same output also said
447
474
  // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
448
475
  // being taken. A capture cannot know which red lines predated that shortage and which it caused,
449
476
  // so none may become a fingerprint every later task gets to forgive.
450
- if (captureHasInvalidatingInfra(raw)) {
477
+ const invalidatingLines = captureInvalidatingLines(combinedOutput);
478
+ if (invalidatingLines.length) {
451
479
  console.error(`tickmarkr: baseline capture for "${name}" completed with process/resource-exhaustion evidence — `
452
480
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
453
- + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion.`);
454
- base.commands[name] = invalidCaptureEntry(durationMs);
481
+ + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
482
+ + `First invalidating line: ${invalidatingLines[0]}`);
483
+ base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
455
484
  continue;
456
485
  }
457
486
  base.commands[name] = {
@@ -530,7 +559,14 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
530
559
  if (!cmd) {
531
560
  // nothing detected for this gate in the target repo — journal an explicit skip instead of
532
561
  // vanishing silently (a lint gate with no lint script rendered as forever-open in status)
533
- results.push({ gate: name, pass: true, details: `no ${name} command detected — skipped`, meta: { skipped: true } });
562
+ const reason = `no ${name} command detected`;
563
+ const outcome = { kind: "skipped", reason };
564
+ results.push({
565
+ gate: name,
566
+ pass: true,
567
+ details: `${reason} — skipped`,
568
+ meta: { skipped: true, outcome },
569
+ });
534
570
  continue;
535
571
  }
536
572
  const entry = baseline.commands[name];
@@ -556,6 +592,19 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
556
592
  record(killed);
557
593
  continue;
558
594
  }
595
+ const runner = shellToken(cmd) ?? name;
596
+ if (entry?.missingCommand === true || r.code === 127) {
597
+ const cause = entry?.missingCommand === true && r.code !== 127
598
+ ? "the baseline capture recorded this runner as missing"
599
+ : `the head command exited 127${/\bENOENT\b/i.test(`${r.stdout}\n${r.stderr}`) ? " (spawn ENOENT)" : ""}`;
600
+ record({
601
+ gate: name,
602
+ pass: false,
603
+ details: `unreadable — ${cause}; runner ${JSON.stringify(runner)} did not produce a trustworthy verdict`,
604
+ meta: { unreadable: true, runner },
605
+ });
606
+ continue;
607
+ }
559
608
  if (r.code === 0) {
560
609
  record({ gate: name, pass: true, details: "exit 0" });
561
610
  continue;
@@ -594,8 +643,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
594
643
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
595
644
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
596
645
  if (!failing.length && !baselineRed) {
646
+ const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
647
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
648
+ : "was killed at its ceiling";
597
649
  const closed = entry?.infra === true
598
- ? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
650
+ ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
599
651
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
600
652
  const evidence = unrecognizedEvidence(raw);
601
653
  record({
@@ -1,4 +1,4 @@
1
- import type { WorkerAdapter } from "../adapters/types.js";
1
+ import { type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ExecutorDriver, type Slot } from "../drivers/types.js";
3
3
  export declare const GATE_PANE_SEP = " \u00B7 ";
4
4
  export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
@@ -57,9 +57,15 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
57
57
  value: T;
58
58
  outputs: string[];
59
59
  }>;
60
+ export interface LlmRunResult {
61
+ output: string;
62
+ exitCode?: number;
63
+ timedOut: boolean;
64
+ }
60
65
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
61
66
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
62
67
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
68
+ export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
63
69
  export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
64
70
  export declare function extractJson<T>(raw: string): T | null;
65
71
  /** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
package/dist/gates/llm.js CHANGED
@@ -3,6 +3,7 @@ import { randomBytes } from "node:crypto";
3
3
  import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
+ import { matchesTrustDialog } from "../adapters/types.js";
6
7
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
7
8
  import { bannerShell, paneDispatchCommand } from "../brand.js";
8
9
  import { sh } from "../run/git.js";
@@ -128,21 +129,24 @@ export async function captureLlmOutput(run) {
128
129
  const value = await llmOutputCapture.run(outputs, run);
129
130
  return { value, outputs };
130
131
  }
131
- export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
132
+ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
132
133
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
133
134
  try {
134
135
  const pf = join(dir, "prompt.md");
135
136
  writeFileSync(pf, prompt);
136
137
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
137
- return r.stdout + "\n" + r.stderr;
138
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
138
139
  }
139
140
  finally {
140
141
  rmSync(dir, { recursive: true, force: true });
141
142
  }
142
143
  }
144
+ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
145
+ return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
146
+ }
143
147
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
144
148
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
145
- export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
149
+ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
146
150
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
147
151
  let slot;
148
152
  let accountant;
@@ -161,10 +165,21 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
161
165
  slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
162
166
  via.onSlot?.(slot);
163
167
  await via.driver.run(slot, paneDispatchCommand(scriptPath));
168
+ if (via.driver.sendKey) {
169
+ try {
170
+ if (matchesTrustDialog(await via.driver.read(slot, 400), adapter.trustDialog)) {
171
+ await via.driver.sendKey(slot, adapter.trustDialog.key);
172
+ }
173
+ }
174
+ catch {
175
+ /* a failed pre-wait read must not replace the verdict wait */
176
+ }
177
+ }
164
178
  // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
165
179
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
166
180
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
167
181
  let out;
182
+ let timedOut = false;
168
183
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
169
184
  if (!gatePrompt) {
170
185
  await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
@@ -230,8 +245,14 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
230
245
  break;
231
246
  }
232
247
  }
248
+ timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
233
249
  }
234
- return dewrapPaneVerdict(out, nonce);
250
+ const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
251
+ return {
252
+ output: dewrapPaneVerdict(out, nonce),
253
+ ...(Number.isFinite(exitCode) ? { exitCode } : {}),
254
+ timedOut,
255
+ };
235
256
  }
236
257
  finally {
237
258
  try {
@@ -245,6 +266,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
245
266
  }
246
267
  }
247
268
  }
269
+ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
270
+ return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
271
+ }
248
272
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
249
273
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
250
274
  // literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
@@ -304,12 +328,15 @@ export function dewrapPaneVerdict(out, nonce) {
304
328
  }
305
329
  return out;
306
330
  }
331
+ export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
332
+ const result = await (via
333
+ ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
334
+ : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
335
+ llmOutputCapture.getStore()?.push(result.output);
336
+ return result;
337
+ }
307
338
  export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
308
- const out = await (via
309
- ? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
310
- : runHeadless(adapter, model, prompt, cwd, timeoutMs));
311
- llmOutputCapture.getStore()?.push(out);
312
- return out;
339
+ return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
313
340
  }
314
341
  export function extractJson(raw) {
315
342
  const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
@@ -369,37 +396,41 @@ export function extractVerdictJson(raw, nonce) {
369
396
  /* fall through */
370
397
  }
371
398
  }
372
- let pos = raw.length - 1;
373
- while (pos >= 0) {
374
- const end = raw.lastIndexOf("}", pos);
375
- if (end === -1)
376
- return null;
377
- let depth = 1;
378
- let stepped = false;
379
- for (let i = end - 1; i >= 0; i--) {
380
- if (raw[i] === "}")
399
+ for (let start = raw.lastIndexOf("{"); start >= 0;) {
400
+ const nextStart = start === 0 ? -1 : raw.lastIndexOf("{", start - 1);
401
+ let depth = 0;
402
+ let quoted = false;
403
+ let escaped = false;
404
+ for (let i = start; i < raw.length; i++) {
405
+ const char = raw[i];
406
+ if (quoted) {
407
+ if (escaped)
408
+ escaped = false;
409
+ else if (char === "\\")
410
+ escaped = true;
411
+ else if (char === '"')
412
+ quoted = false;
413
+ continue;
414
+ }
415
+ if (char === '"')
416
+ quoted = true;
417
+ else if (char === "{")
381
418
  depth++;
382
- else if (raw[i] === "{") {
383
- depth--;
384
- if (depth === 0) {
385
- stepped = true;
386
- try {
387
- const v = JSON.parse(raw.slice(i, end + 1));
388
- if (v && typeof v === "object" && v.nonce === nonce) {
389
- const { nonce: _n, ...rest } = v;
390
- return rest;
391
- }
419
+ else if (char === "}" && --depth === 0) {
420
+ try {
421
+ const v = JSON.parse(raw.slice(start, i + 1));
422
+ if (v && typeof v === "object" && v.nonce === nonce) {
423
+ const { nonce: _n, ...rest } = v;
424
+ return rest;
392
425
  }
393
- catch {
394
- /* keep scanning */
395
- }
396
- pos = i - 1;
397
- break;
398
426
  }
427
+ catch {
428
+ /* keep scanning */
429
+ }
430
+ break;
399
431
  }
400
432
  }
401
- if (!stepped)
402
- return null;
433
+ start = nextStart;
403
434
  }
404
435
  return null;
405
436
  }
@@ -41,13 +41,15 @@ export type TaskDiffMeasurement = {
41
41
  readonly fullMeasurement: ArtifactDiffMeasurement;
42
42
  readonly capMeasurement: ArtifactDiffMeasurement;
43
43
  };
44
- export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
44
+ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?: readonly string[]): Promise<TaskDiffMeasurement>;
45
45
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
46
46
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
47
47
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
48
48
  export declare function isDiffCapPark(result: GateResult): boolean;
49
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
50
  export declare function modelId(model: string): string;
51
+ /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
52
+ export declare function modelProvider(model: string, fallback?: string): string;
51
53
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
52
54
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
53
55
  floor?: Tier): BillingChannel | null;
@@ -2,12 +2,13 @@ import { writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
+ import { filesGlob } from "../graph/files-glob.js";
5
6
  import { renderAcceptanceItem } from "../graph/schema.js";
6
7
  import { getAdapter } from "../adapters/registry.js";
7
8
  import { shOk } from "../run/git.js";
8
9
  import { redactSecrets } from "../run/redact.js";
9
10
  import { marginalCostRank } from "../route/router.js";
10
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
11
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
11
12
  import { classifyVerdictCause } from "./verdict-cause.js";
12
13
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
13
14
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -99,12 +100,14 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
99
100
  const changed = diff.split("\n").filter((l) => /^[+-]/.test(l) && !/^(?:\+\+\+|---)/.test(l));
100
101
  return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
101
102
  }
102
- export async function fetchTaskDiff(worktree, baseRef) {
103
+ export async function fetchTaskDiff(worktree, baseRef, files = []) {
103
104
  // --full-index: abbreviated index lines vary with object-store density, so two measurements of
104
105
  // the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
105
- const [rawFull, rawForCap] = await Promise.all([
106
- shOk(`git diff --full-index '${baseRef}..HEAD'`, worktree),
107
- shOk(`git diff --full-index -U0 '${baseRef}..HEAD'`, worktree),
106
+ const matched = files.length ? (await changedPaths(worktree, baseRef)).filter(filesGlob([...files])) : [];
107
+ const pathspec = matched.length ? ` -- ${matched.map(shq).join(" ")}` : "";
108
+ const [rawFull, rawForCap] = files.length && matched.length === 0 ? ["", ""] : await Promise.all([
109
+ shOk(`git diff --full-index '${baseRef}..HEAD'${pathspec}`, worktree),
110
+ shOk(`git diff --full-index -U0 '${baseRef}..HEAD'${pathspec}`, worktree),
108
111
  ]);
109
112
  const fullMeasurement = measureArtifactDiff(rawFull);
110
113
  const capMeasurement = measureArtifactDiff(rawForCap);
@@ -162,6 +165,24 @@ export function diffCapParkReason(results) {
162
165
  export function modelId(model) {
163
166
  return model.slice(model.lastIndexOf("/") + 1);
164
167
  }
168
+ /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
169
+ export function modelProvider(model, fallback = "unknown") {
170
+ const id = model.toLowerCase();
171
+ const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
172
+ if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
173
+ return "openai";
174
+ if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
175
+ return "anthropic";
176
+ if (prefix === "google" || /^gemini(?:-|$)/.test(id))
177
+ return "google";
178
+ if (prefix === "xai" || /^grok(?:-|$)/.test(id))
179
+ return "xai";
180
+ if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
181
+ return "zhipu";
182
+ if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
183
+ return "moonshot";
184
+ return fallback;
185
+ }
165
186
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
166
187
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
167
188
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -179,11 +200,14 @@ floor) {
179
200
  const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
180
201
  if (!authorChannel)
181
202
  return null;
203
+ // Failover also guards true provider; the initial pick keeps the established stamped-vendor contract.
204
+ const authorProvider = modelProvider(author.model, authorChannel.vendor);
182
205
  return (channels
183
206
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
184
207
  // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
185
208
  // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
186
209
  .filter((c) => c.vendor !== authorChannel.vendor
210
+ && (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
187
211
  && modelId(c.model) !== modelId(author.model)
188
212
  && !exclude.includes(channelKey(c))
189
213
  && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
@@ -285,10 +309,10 @@ artifactDir) {
285
309
  ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
286
310
  : "no cross-vendor reviewer available (diversity rule)";
287
311
  return cfg.review.required
288
- ? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
312
+ ? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
289
313
  : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
290
314
  }
291
- const measuredDiff = await fetchTaskDiff(worktree, baseRef);
315
+ const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
292
316
  // Keep the reader payload identical to the text charged to the strict cap:
293
317
  // whole-file source deletions are represented by their citable operation fact.
294
318
  const diff = reviewableLogicDiff(measuredDiff.full);
@@ -327,7 +351,7 @@ Approve iff no material finding remains; an empty findings list is a clean appro
327
351
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
328
352
  `;
329
353
  let concludedOnInactivity = false;
330
- const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
354
+ const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
331
355
  driver: via.driver,
332
356
  keep: via.keep,
333
357
  onSlot: via.onSlot,
@@ -341,13 +365,16 @@ The top-level comments array is optional. Use it only for actionable line-anchor
341
365
  // (run-20260709-104447 P87-09). ponytail: literal 15min; make it cfg.review.timeoutMs if a
342
366
  // second knob-turner appears.
343
367
  900_000);
368
+ const raw = llm.output;
369
+ const provider = modelProvider(reviewer.model, reviewer.vendor);
344
370
  const v = extractVerdictJson(raw, nonce);
345
371
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
346
372
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
347
373
  if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
348
374
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
349
375
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
350
- const cause = classifyVerdictCause(raw, nonce, "approve");
376
+ const cause = classifyVerdictCause(raw, nonce, "approve", llm);
377
+ const bytes = Buffer.byteLength(raw, "utf8");
351
378
  let saved;
352
379
  if (artifactDir) {
353
380
  try {
@@ -366,12 +393,15 @@ The top-level comments array is optional. Use it only for actionable line-anchor
366
393
  return {
367
394
  gate: "review",
368
395
  pass: false,
369
- details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
396
+ details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
370
397
  meta: {
371
398
  ...policyMeta,
372
399
  reviewer: channelKey(reviewer),
400
+ vendor: reviewer.vendor,
401
+ provider,
373
402
  unparseable: true,
374
403
  cause,
404
+ ...(cause === "empty-output" ? { bytes } : {}),
375
405
  ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
376
406
  },
377
407
  };
@@ -379,11 +409,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
379
409
  const decided = findings !== null
380
410
  ? classifyReviewFindings(findings)
381
411
  : classifyReviewIssues(v.approve, v.issues);
382
- const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
412
+ const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
383
413
  return {
384
414
  gate: "review",
385
415
  pass: decided.pass,
386
416
  details: appendAnchoredReview(prose, v),
387
- meta: { ...policyMeta, reviewer: channelKey(reviewer) },
417
+ meta: { ...policyMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
388
418
  };
389
419
  }
@@ -31,6 +31,12 @@ export type GateEvent = {
31
31
  phase: "end";
32
32
  gate: GateName;
33
33
  result: GateResult;
34
+ } | {
35
+ phase: "note";
36
+ gate: GateName;
37
+ name: string;
38
+ payload: Record<string, unknown>;
39
+ result: GateResult;
34
40
  };
35
41
  export interface GateContext {
36
42
  worktree: string;
@@ -11,7 +11,7 @@ import { evidenceGate } from "./evidence.js";
11
11
  import { captureLlmOutput } from "./llm.js";
12
12
  import { disallowedBy } from "../route/preference.js";
13
13
  import { marginalCostRank } from "../route/router.js";
14
- import { reviewGate } from "./review.js";
14
+ import { pickReviewer, reviewGate } from "./review.js";
15
15
  import { scopeGate } from "./scope.js";
16
16
  import { shGit } from "../run/git.js";
17
17
  import { withJudgeInvocationEvidence } from "../run/journal.js";
@@ -540,7 +540,7 @@ export async function runGates(task, ctx) {
540
540
  .sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y))[0];
541
541
  // v1.87 T2: the failover seat obeys the same policy the primary judge just passed — a denied
542
542
  // channel is refused here too, never reached by falling through the exclusion arms below.
543
- const judgePool = (ctx.judgeChannels ?? ctx.channels).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
543
+ const judgePool = (ctx.judgeChannels ?? []).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
544
544
  const crossAdapter = pick(judgePool.filter((c) => c.adapter !== flakedAdapter));
545
545
  const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
546
546
  // Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
@@ -576,25 +576,40 @@ export async function runGates(task, ctx) {
576
576
  return captured.value;
577
577
  };
578
578
  let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
579
- // OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
580
- // never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
581
- // the flaked verdict never enters results). The exclusion rides reviewGate's own excludeReviewers
582
- // parameter, so pickReviewer's diversity rules still govern the retry seat; a fleet with no second
583
- // eligible seat keeps the ORIGINAL result so the recorded cause stays truthful (OBS-196).
579
+ // OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
580
+ // different adapter. Only a single-adapter eligible pool may fall back to another channel on the
581
+ // flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
584
582
  if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
585
583
  const flaked = rv.meta.reviewer;
584
+ const emptyOutput = rv.meta.cause === "empty-output";
585
+ if (emptyOutput) {
586
+ await ctx.onGate?.({
587
+ phase: "note", gate: "review", name: "reviewer-empty-output",
588
+ payload: { reviewer: flaked, bytes: typeof rv.meta.bytes === "number" ? rv.meta.bytes : 0 },
589
+ result: { ...rv, meta: { ...rv.meta, skipped: true } },
590
+ });
591
+ }
586
592
  const retryVia = ctx.via
587
593
  ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
588
594
  : undefined;
589
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
595
+ const priorExclusions = ctx.excludeReviewers ?? [];
596
+ const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
597
+ const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
598
+ const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
599
+ const exclusion = crossAdapter ? "adapter" : "channel";
600
+ const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
601
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir));
590
602
  if (second.meta?.noEligibleReviewer !== true) {
591
603
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
604
+ const route = exclusion === "adapter"
605
+ ? `different-adapter retry; excluded flaked adapter ${flakedAdapter}`
606
+ : `same-adapter fallback; excluded flaked channel ${flaked}`;
592
607
  rv = {
593
608
  ...second,
594
609
  // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
595
610
  // re-route visible in the result text a reader actually opens, including on a red retry.
596
- details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
597
- meta: { ...second.meta, reviewRetry: { flaked, retried } },
611
+ details: `review re-route (${route}): ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
612
+ meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
598
613
  };
599
614
  }
600
615
  else if (task.routingHints?.floor) {
@@ -1,4 +1,8 @@
1
1
  export type VerdictDiscriminator = "approve" | "pass" | "ok" | "action";
2
- export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
2
+ export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict" | "timeout" | "startup-failure";
3
+ export interface VerdictProcessFacts {
4
+ timedOut?: boolean;
5
+ exitCode?: number;
6
+ }
3
7
  export declare function hasVerdictParticipationWitness(raw: string, nonce: string, discriminator: VerdictDiscriminator): boolean;
4
- export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator): VerdictUnparseableCause;
8
+ export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator, process?: VerdictProcessFacts): VerdictUnparseableCause;