tickmarkr 1.84.0 → 1.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/README.md +4 -2
  2. package/dist/adapters/catalog-remote.d.ts +64 -0
  3. package/dist/adapters/catalog-remote.js +287 -0
  4. package/dist/adapters/catalog.d.ts +96 -0
  5. package/dist/adapters/catalog.js +176 -0
  6. package/dist/adapters/claude-code.d.ts +1 -0
  7. package/dist/adapters/claude-code.js +59 -1
  8. package/dist/adapters/fake.js +42 -4
  9. package/dist/adapters/model-lints.d.ts +25 -5
  10. package/dist/adapters/model-lints.js +184 -50
  11. package/dist/adapters/model-windows.d.ts +31 -0
  12. package/dist/adapters/model-windows.js +69 -0
  13. package/dist/adapters/prompt.d.ts +5 -1
  14. package/dist/adapters/prompt.js +13 -4
  15. package/dist/adapters/registry.d.ts +25 -26
  16. package/dist/adapters/registry.js +173 -110
  17. package/dist/adapters/types.d.ts +3 -0
  18. package/dist/adapters/types.js +36 -3
  19. package/dist/brand.d.ts +5 -1
  20. package/dist/brand.js +18 -2
  21. package/dist/cli/commands/doctor.d.ts +3 -0
  22. package/dist/cli/commands/doctor.js +43 -21
  23. package/dist/cli/commands/fleet.d.ts +7 -0
  24. package/dist/cli/commands/fleet.js +94 -74
  25. package/dist/cli/commands/init.js +118 -5
  26. package/dist/cli/commands/status.js +202 -46
  27. package/dist/compile/collateral.d.ts +86 -2
  28. package/dist/compile/collateral.js +294 -3
  29. package/dist/compile/gsd.d.ts +2 -1
  30. package/dist/compile/gsd.js +68 -2
  31. package/dist/compile/native.d.ts +14 -0
  32. package/dist/compile/native.js +161 -12
  33. package/dist/config/config.d.ts +82 -5
  34. package/dist/config/config.js +253 -66
  35. package/dist/config/fleet-overlay.d.ts +25 -20
  36. package/dist/config/fleet-overlay.js +195 -77
  37. package/dist/config/fleet-why.d.ts +23 -0
  38. package/dist/config/fleet-why.js +42 -0
  39. package/dist/drivers/herdr.d.ts +21 -3
  40. package/dist/drivers/herdr.js +344 -110
  41. package/dist/gates/acceptance.js +7 -2
  42. package/dist/gates/baseline.d.ts +1 -0
  43. package/dist/gates/baseline.js +91 -13
  44. package/dist/gates/llm.d.ts +0 -1
  45. package/dist/gates/llm.js +5 -30
  46. package/dist/gates/review.d.ts +9 -1
  47. package/dist/gates/review.js +105 -10
  48. package/dist/gates/run-gates.d.ts +9 -0
  49. package/dist/gates/run-gates.js +285 -41
  50. package/dist/gates/verdict-cause.d.ts +4 -0
  51. package/dist/gates/verdict-cause.js +63 -0
  52. package/dist/graph/schema.d.ts +6 -0
  53. package/dist/graph/schema.js +8 -5
  54. package/dist/route/router.d.ts +0 -5
  55. package/dist/route/router.js +16 -20
  56. package/dist/run/consult.d.ts +6 -0
  57. package/dist/run/consult.js +35 -25
  58. package/dist/run/daemon.d.ts +48 -2
  59. package/dist/run/daemon.js +1488 -330
  60. package/dist/run/journal.d.ts +56 -3
  61. package/dist/run/journal.js +358 -4
  62. package/dist/run/stall.d.ts +35 -1
  63. package/dist/run/stall.js +118 -8
  64. package/dist/tui/cockpit/capture.d.ts +12 -0
  65. package/dist/tui/cockpit/capture.js +37 -1
  66. package/dist/tui/cockpit/components.d.ts +2 -0
  67. package/dist/tui/cockpit/components.js +8 -8
  68. package/dist/tui/cockpit/derive.d.ts +29 -2
  69. package/dist/tui/cockpit/derive.js +219 -23
  70. package/dist/tui/cockpit/run-cockpit.js +128 -27
  71. package/dist/tui/cockpit/theme.d.ts +32 -26
  72. package/dist/tui/cockpit/theme.js +11 -5
  73. package/dist/tui/ink/components.d.ts +0 -15
  74. package/dist/tui/ink/components.js +0 -17
  75. package/dist/tui/ink/fleet-app.d.ts +4 -1
  76. package/dist/tui/ink/fleet-app.js +134 -13
  77. package/fixtures/sample.native.md +1 -1
  78. package/package.json +1 -1
  79. package/skills/tickmarkr-overseer/SKILL.md +354 -34
  80. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
  81. package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
  82. package/dist/tui/ink/studio-app.d.ts +0 -59
  83. package/dist/tui/ink/studio-app.js +0 -320
  84. package/dist/tui/save.d.ts +0 -38
  85. package/dist/tui/save.js +0 -96
  86. package/dist/tui/staging.d.ts +0 -29
  87. package/dist/tui/staging.js +0 -78
@@ -5,6 +5,7 @@ import { renderAcceptanceItem } from "../graph/schema.js";
5
5
  import { sh } from "../run/git.js";
6
6
  import { checkDiffCap, fetchTaskDiff, isProtectedEvidence, setAsideReceiptPath } from "./review.js";
7
7
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
8
+ import { classifyVerdictCause } from "./verdict-cause.js";
8
9
  // Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
9
10
  const JUDGE_TIMEOUT_MS = 900_000;
10
11
  const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
@@ -328,12 +329,16 @@ The top-level comments array is optional. Use it only for actionable line-anchor
328
329
  const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
329
330
  const extracted = extractVerdictJson(raw, nonce);
330
331
  if (!extracted) {
332
+ const cause = classifyVerdictCause(raw, nonce, "pass");
331
333
  // GATE-09: structured meta names the flaked judge channel (mirrors review.ts:99 meta precedent) so
332
334
  // run-gates can retry the judge on a failover channel without string-matching details (D-03).
333
335
  // The parsed-verdict paths below are untouched — the flake signal is exactly extractJson→null here.
336
+ const failure = cause === "malformed-verdict"
337
+ ? "judge output unparseable — failing closed"
338
+ : "judge dispatch failed — no structurally valid nonce-bound response; output unparseable — failing closed";
334
339
  return { gate: "acceptance", pass: false,
335
- details: warn + detBlock + "judge output unparseable — failing closed",
336
- meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
340
+ details: warn + detBlock + failure,
341
+ meta: { unparseable: true, cause, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
337
342
  }
338
343
  const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
339
344
  // v1.70 / OBS-129: each citation must fall inside a changed hunk of `diff` (the exact string embedded in
@@ -14,6 +14,7 @@ export interface BaselineWarning {
14
14
  commands: string[];
15
15
  reason: string;
16
16
  }
17
+ export declare const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
17
18
  export declare function fingerprint(output: string): string[];
18
19
  export declare function detectGateCommands(repoRoot: string, cfg: TickmarkrConfig): Record<string, string>;
19
20
  export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
@@ -14,16 +14,76 @@ const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:
14
14
  // output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
15
15
  // OBS-42: vitest's diagnostic failure headings are shared anchors for baseline and tip verification.
16
16
  const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Exception)\b)/;
17
- const SUMMARY_FAIL_RE = /^\s*Tests?\s+(?:Files?\s+)?\d+\s+failed/; // " Tests N failed | M passed (T)"
17
+ // OBS-278: a failure fingerprint is a SHAPE — the way a runner reports a failure — never error/fail
18
+ // vocabulary. A status surface that draws those words (zone +N, run attempt N, "1 failed") used to
19
+ // fingerprint as a fresh failure on every attempt, rejecting a task for rendering what it was chartered
20
+ // to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
21
+ // Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
22
+ // "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
23
+ const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
24
+ const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
25
+ const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
26
+ const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
27
+ // The failure shapes non-Vitest runners name their tests with: pytest's short summary
28
+ // (`FAILED tests/t.py::test_x - AssertionError`, `ERROR tests/t.py::fixture`), go test
29
+ // (`--- FAIL: TestFoo (0.00s)`), TAP / node:test (`not ok 1 - name`, whose summary line lands in
30
+ // SUMMARY_FAIL_RE) and python unittest's failure-block header (`FAIL: test_x (mod.Class.test_x)`,
31
+ // verbatim capture, python3 -m unittest -v). Each anchors the runner's token at line START; the
32
+ // pytest/go forms also require a file/test identifier after it (`::`, a path separator, or an
33
+ // extension) — so a surface drawing "FAILED" mid-line, or a plain sentence like "FAILED to reach
34
+ // the zone", still matches nothing. Without these a genuinely new pytest/go/TAP/unittest failure on
35
+ // an already-red baseline was forgiven as pre-existing, because only shaped lines can reject.
36
+ const RUNNER_FAIL_RE = /^\s*(?:(?:FAILED|ERROR)\s+\S*(?:::|[/\\]|\.[A-Za-z]\w*\b)|FAIL:\s+\S|---\s+FAIL:\s+\S|not ok\b)/;
37
+ // The other half of the position rule: runners that put the verdict LAST. cargo/libtest names each
38
+ // failing test as `test tests::name ... FAILED` (verbatim capture, cargo 1.95.0) and details it with
39
+ // `thread 'tests::name' (81651804) panicked at src/lib.rs:7:24:`; python unittest -v writes
40
+ // `test_x (mod.Class.test_x) ... FAIL` (and `... ERROR`) — the same rule with the runner's
41
+ // qualified-name token between identifier and separator, and the short verdict spelling (verbatim
42
+ // capture, python3 -m unittest -v). Recognition is positional, not a vendor name: an identifier, the
43
+ // runner's own separator, then the verdict ENDING the line. A drawn status strip ("│ ✗ tip-verify
44
+ // FAILED · zone +3 │") has the verdict mid-line inside chrome, so it matches nothing here — which is
45
+ // why this can generalize without re-opening OBS-278.
46
+ const TRAILING_FAIL_RE = /^\s*(?:test\s+)?\S+(?:\s+\([^)]*\))?\s+(?:\.{3}|-{3,})\s+(?:FAIL(?:ED)?|ERROR)\s*$|^\s*thread\s+'[^']*'\s+.*panicked at\s+\S/;
47
+ // node:test's spec reporter speaks glyphs, not words: a failure is named by a leading ✖ (verbatim
48
+ // capture, Node v22 `--test-reporter=spec`: `✖ old failure (0.585875ms)`, totalled as `ℹ fail 1` in
49
+ // SUMMARY_FAIL_RE) — the exact counterpart of the ✓/✔ pass markers PASS_LINE_RE drops, with the same
50
+ // optional label prefixes. Recognition is positional — the runner's own marker at line start — so any
51
+ // runner sharing the glyph protocol is read without being enumerated; a status strip drawing ✖
52
+ // mid-line inside chrome matches nothing, the same position rule as the other shapes.
53
+ const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
54
+ // Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
55
+ // and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
56
+ const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l);
57
+ const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
58
+ || TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
59
+ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
18
60
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
61
+ // A failing command whose output holds no shape any runner here names. The marker is content-free and
62
+ // constant: downstream consumers (tip verify journals fingerprint counts) still see that the command
63
+ // failed, while no line of the output — narration, status strip, stack trace — can enter the set. Being
64
+ // constant it is identical on every attempt, so it can never surface as a fresh fingerprint either.
65
+ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
66
+ // OBS-278 dies here: only a failure SHAPE is fingerprintable. Vocabulary is not a shape — a surface that
67
+ // renders "zone +3 · run attempt 2 · 1 failed" (what T9 is chartered to draw) contributes nothing, with
68
+ // or without box glyphs, so it cannot manufacture a fresh-fingerprint rejection on any future attempt.
19
69
  export function fingerprint(output) {
20
70
  const lines = output
21
71
  .split("\n")
22
72
  .map((l) => l.replace(ANSI_RE, ""))
23
- .filter((l) => !PASS_LINE_RE.test(l) && (FAIL_ANCHOR_RE.test(l) || /\b(error|fail(ed|ure|ing)?)\b/i.test(l)))
24
- .map(normalizeLine);
25
- return [...new Set(lines)];
73
+ .filter((l) => !PASS_LINE_RE.test(l));
74
+ const shaped = lines.filter(isFailureShaped);
75
+ if (!shaped.length)
76
+ return lines.some((l) => l.trim()) ? [UNRECOGNIZED_FAILURE] : [];
77
+ return [...new Set(shaped.map(normalizeLine))];
26
78
  }
79
+ // The diagnostic channel for a runner we cannot read: raw output lines, never fingerprints, never a
80
+ // verdict. They exist so the operator sees SOMETHING when a green command turns red unrecognizably.
81
+ const unrecognizedEvidence = (raw) => raw
82
+ .split("\n")
83
+ .map((l) => l.replace(ANSI_RE, "").trimEnd())
84
+ .filter((l) => VOCAB_RE.test(l))
85
+ .slice(0, 10)
86
+ .join("\n");
27
87
  // stored baselines may predate ANSI/pass-marker hardening — renormalize at compare time so existing
28
88
  // on-disk baseline.json files stay comparable without recapture (compat invariant, CLAUDE.md)
29
89
  const renormalize = (fp) => normalizeLine(fp.replace(ANSI_RE, ""));
@@ -68,7 +128,9 @@ export async function captureBaseline(cwd, commands) {
68
128
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
69
129
  base.commands[name] = {
70
130
  exitCode: r.code,
71
- fingerprints: fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")),
131
+ // a command that exits 0 has no failures to fingerprint — recording any would be a lie the
132
+ // compare step then has to forgive
133
+ fingerprints: r.code === 0 ? [] : fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")),
72
134
  missingCommand: missingConfiguredCommand(cmd, r),
73
135
  };
74
136
  }
@@ -119,12 +181,12 @@ function headlineDetails(raw, fresh) {
119
181
  const headlines = raw
120
182
  .split("\n")
121
183
  .map((l) => l.replace(ANSI_RE, ""))
122
- .filter((l) => FAIL_ANCHOR_RE.test(l) || SUMMARY_FAIL_RE.test(l));
184
+ .filter((l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l));
123
185
  if (!headlines.length)
124
186
  return { details: `new failures vs baseline:\n${fresh.join("\n")}` };
125
187
  return {
126
188
  details: `failing tests:\n${headlines.join("\n")}\n\nnew failure fingerprints vs baseline (secondary):\n${fresh.join("\n")}`,
127
- meta: { failingTests: headlines.filter((l) => FAIL_ANCHOR_RE.test(l)) },
189
+ meta: { failingTests: headlines.filter(namesFailure) },
128
190
  };
129
191
  }
130
192
  export async function compareToBaseline(cwd, commands, baseline, enabled) {
@@ -145,18 +207,34 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
145
207
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
146
208
  const known = new Set((baseline.commands[name]?.fingerprints ?? []).map(renormalize));
147
209
  // OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
148
- const fresh = fingerprint(raw).filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
149
- if (!fresh.length && (baseline.commands[name]?.exitCode ?? 1) === 0) {
210
+ const current = fingerprint(raw);
211
+ const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
212
+ // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
213
+ // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
214
+ // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
215
+ // already red has its new failures forgiven, so forgiveness that rests on the marker SAYS so
216
+ // below rather than reading as a verified green. Raise the ceiling by teaching isFailureShaped
217
+ // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
218
+ // verdict); loosening back to vocabulary re-opens OBS-278.
219
+ const unreadable = current.includes(UNRECOGNIZED_FAILURE);
220
+ const failing = fresh.filter((f) => f !== UNRECOGNIZED_FAILURE);
221
+ if (!failing.length && (baseline.commands[name]?.exitCode ?? 1) === 0) {
222
+ const closed = `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
223
+ const evidence = unrecognizedEvidence(raw);
150
224
  results.push({
151
225
  gate: name,
152
226
  pass: false,
153
- details: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`,
227
+ details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
154
228
  });
155
229
  continue;
156
230
  }
157
- results.push(fresh.length
158
- ? { gate: name, pass: false, ...headlineDetails(raw, fresh) }
159
- : { gate: name, pass: true, details: `exit ${r.code} but only pre-existing failures (forgiven)` });
231
+ results.push(failing.length
232
+ ? { gate: name, pass: false, ...headlineDetails(raw, failing) }
233
+ : {
234
+ gate: name,
235
+ pass: true,
236
+ details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
237
+ });
160
238
  }
161
239
  return results;
162
240
  }
@@ -15,7 +15,6 @@ export declare function appendAnchoredReview(prose: string, verdict: unknown): s
15
15
  export declare function verdictNonceLine(nonce: string): string;
16
16
  export declare function extractPromptNonce(prompt: string): string | null;
17
17
  export declare function gateExitTrailer(nonce: string): string;
18
- export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string, prompt?: string): string;
19
18
  export type GatePaneRole = "judge" | "review" | "consult";
20
19
  /** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
21
20
  export declare function gatePaneName(role: GatePaneRole, taskId: string, suffix?: string): string;
package/dist/gates/llm.js CHANGED
@@ -69,29 +69,6 @@ export function extractPromptNonce(prompt) {
69
69
  export function gateExitTrailer(nonce) {
70
70
  return `printf '\\nTICKMARKR_''EXIT_${nonce}:%s\\n' $?`;
71
71
  }
72
- // v1.64: scripted fake judge verdicts predate the required per-criterion evidence field — quote the
73
- // first line of the prompt's own diff block into rows lacking one so zero-token fixtures keep their
74
- // outcomes. Rows scripting an explicit evidence value pass through verbatim (tests exercise both paths).
75
- function injectFakeEvidence(obj, prompt) {
76
- if (!prompt.startsWith("TICKMARKR-JUDGE") || !Array.isArray(obj.criteria))
77
- return obj;
78
- const line = /```diff\n([\s\S]*?)```/.exec(prompt)?.[1].split("\n").find((l) => l.trim());
79
- if (!line)
80
- return obj;
81
- const criteria = obj.criteria.map((row) => row && typeof row === "object" && !("evidence" in row) ? { ...row, evidence: line } : row);
82
- return { ...obj, criteria };
83
- }
84
- // ponytail: fake adapter serves static verdict JSON without nonce; append a bound copy for zero-token tests.
85
- export function augmentFakeVerdictOutput(adapter, out, nonce, prompt = "") {
86
- if (adapter.id !== "fake")
87
- return out;
88
- const obj = extractJson(out);
89
- if (!obj || typeof obj !== "object" || obj.nonce === nonce)
90
- return out;
91
- if (typeof obj.nonce === "string")
92
- return out;
93
- return `${out}\n${JSON.stringify(injectFakeEvidence({ ...obj, nonce }, prompt))}`;
94
- }
95
72
  /** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
96
73
  export function gatePaneName(role, taskId, suffix = "") {
97
74
  return `${role}${GATE_PANE_SEP}${taskId}${suffix}`;
@@ -130,11 +107,7 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
130
107
  const pf = join(mkdtempSync(join(tmpdir(), "tickmarkr-llm-")), "prompt.md");
131
108
  writeFileSync(pf, prompt);
132
109
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
133
- const nonce = extractPromptNonce(prompt);
134
- let out = r.stdout + "\n" + r.stderr;
135
- if (nonce)
136
- out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
137
- return out;
110
+ return r.stdout + "\n" + r.stderr;
138
111
  }
139
112
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
140
113
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
@@ -157,10 +130,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
157
130
  // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
158
131
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
159
132
  await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
160
- let out = await via.driver.read(slot, 400);
133
+ const out = await via.driver.read(slot, 400);
161
134
  if (!via.keep)
162
135
  await via.driver.close(slot);
163
- out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
164
136
  return dewrapPaneVerdict(out, nonce);
165
137
  }
166
138
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
@@ -179,6 +151,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
179
151
  export function dewrapPaneVerdict(out, nonce) {
180
152
  if (!out.includes(nonce))
181
153
  return out;
154
+ // Preserve already-readable responder bytes; only a genuinely wrapped verdict needs reconstruction.
155
+ if (extractVerdictJson(out, nonce))
156
+ return out;
182
157
  const lines = out.split("\n");
183
158
  // OBS-209: EVERY brace-start is a candidate, scanned newest-first. findIndex took only the first,
184
159
  // so any earlier line beginning with `{` — a quoted snippet, a lone brace in the reviewer's own
@@ -3,6 +3,7 @@ import { type TickmarkrConfig } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
4
  import { type GateVia } from "./llm.js";
5
5
  import type { GateResult } from "./types.js";
6
+ import { type VerdictUnparseableCause } from "./verdict-cause.js";
6
7
  export type ReviewSeverity = "material" | "minor";
7
8
  export interface ReviewFinding {
8
9
  note: string;
@@ -27,6 +28,13 @@ export declare function isProtectedEvidence(path: string): boolean;
27
28
  export declare function setAsideReceiptPath(section: string): string | null;
28
29
  /** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
29
30
  export declare function setAsideRegenerableCaptures(diff: string): string;
31
+ /**
32
+ * The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
33
+ * never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
34
+ * the leaf class must be visible to the promotion test, and rename detection would hide the old side.
35
+ */
36
+ export declare function changedPaths(worktree: string, baseRef: string): Promise<string[]>;
37
+ export declare function mirrorsVersionOnly(worktree: string, baseRef: string, path: string): Promise<boolean>;
30
38
  export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
31
39
  full: string;
32
40
  forCap: string;
@@ -37,5 +45,5 @@ export declare function diffCapParkReason(results: GateResult[]): string | null;
37
45
  export declare function modelId(model: string): string;
38
46
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
39
47
  prefer?: string[]): BillingChannel | null;
40
- export type ReviewUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
48
+ export type ReviewUnparseableCause = VerdictUnparseableCause;
41
49
  export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
@@ -1,13 +1,14 @@
1
1
  import { writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
- import { channelKey } from "../adapters/types.js";
4
- import { DEFAULT_DIFF_CAP, TIER_RANK } from "../config/config.js";
3
+ import { channelKey, shq } from "../adapters/types.js";
4
+ import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
5
  import { renderAcceptanceItem } from "../graph/schema.js";
6
6
  import { getAdapter } from "../adapters/registry.js";
7
7
  import { shOk } from "../run/git.js";
8
8
  import { redactSecrets } from "../run/redact.js";
9
9
  import { marginalCostRank } from "../route/router.js";
10
10
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
11
+ import { classifyVerdictCause } from "./verdict-cause.js";
11
12
  // legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
12
13
  function classifyReviewIssues(approve, issues) {
13
14
  const inconsistencies = [];
@@ -211,6 +212,34 @@ export function setAsideRegenerableCaptures(diff) {
211
212
  const kindOnly = kindOnlyPaths(sections);
212
213
  return sections.map((s) => setAsideSection(s, kindOnly)).join("");
213
214
  }
215
+ /**
216
+ * The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
217
+ * never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
218
+ * the leaf class must be visible to the promotion test, and rename detection would hide the old side.
219
+ */
220
+ export async function changedPaths(worktree, baseRef) {
221
+ const out = await shOk(`git diff --name-only --no-renames -z '${baseRef}..HEAD'`, worktree);
222
+ return [...new Set(out.split("\0").map((p) => p.trim()).filter(Boolean))].sort();
223
+ }
224
+ /**
225
+ * A root manifest is a version MIRROR only when the bump is all it changed. `package.json` carries the
226
+ * gate commands and the dependency set, so a diff that moves `scripts` or `dependencies` is executable
227
+ * behaviour wearing a leaf-class path — the one thing a path predicate can never see for itself. Every
228
+ * added or removed line must be a `"version":` line (a lockfile bump rewrites several of them); a
229
+ * manifest whose diff cannot be read at all fails closed, out of the class.
230
+ */
231
+ const VERSION_FIELD_LINE_RE = /^[+-]\s*"version":\s*"[^"]*",?\s*$/;
232
+ export async function mirrorsVersionOnly(worktree, baseRef, path) {
233
+ let diff;
234
+ try {
235
+ diff = await shOk(`git diff -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
236
+ }
237
+ catch {
238
+ return false;
239
+ }
240
+ const changed = diff.split("\n").filter((l) => /^[+-]/.test(l) && !/^(?:\+\+\+|---)/.test(l));
241
+ return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
242
+ }
214
243
  export async function fetchTaskDiff(worktree, baseRef) {
215
244
  const full = setAsideRegenerableCaptures(await shOk(`git diff '${baseRef}..HEAD'`, worktree));
216
245
  const forCap = setAsideRegenerableCaptures(await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree));
@@ -269,9 +298,74 @@ export async function reviewGate(task, worktree, baseRef, author, channels, adap
269
298
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
270
299
  // direct tests) skips persistence and changes nothing else.
271
300
  artifactDir) {
272
- if (task.complexity < cfg.review.complexityThreshold) {
273
- return { gate: "review", pass: true, details: `skipped — complexity ${task.complexity} < threshold ${cfg.review.complexityThreshold}`, meta: { skipped: true } };
301
+ // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
302
+ // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
303
+ // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
304
+ // "a law caps complexity at 3, the gate starts at 7" unreachability OBS-186 measured.
305
+ //
306
+ // COLLATERAL this rescoped task closed: the run-gates/daemon participation assertions are rewritten
307
+ // path-keyed, the NamedFake review fixtures author their own nonce-bound verdict (a renamed fake is
308
+ // a distinct responder and does not inherit the registered fake's producer contract — the check is
309
+ // not weakened, the fixture is fixed), the merge decision reads `gateSatisfied`, and the daemon writes a
310
+ // parallel round's gate-result rows in GATE_NAMES order (src/run/daemon.ts).
311
+ //
312
+ // That last one is why: retiring the switch makes fixtures that used to SKIP review journal TWO
313
+ // verdict rows per round instead of one, and judge ‖ review publish in COMPLETION order — so the
314
+ // three journal-determinism oracles (tests/run/narration.test.ts's byte comparison and its
315
+ // throwing-sink event-order check, tests/run/notify-identity.test.ts's two-run equality, and this
316
+ // repo's own phase-start/gate-result pairing in tests/run/daemon.test.ts) start seeing a race.
317
+ // LATENT, not introduced: the operator's config has run `complexityThreshold: 0` since 2026-07-31,
318
+ // so production rounds have journaled both siblings all along — only the fixtures were blind to it.
319
+ // Fixed in the ledger rather than in the oracles, because determinism run-to-run is a property of
320
+ // the journal, not of three test files that happen to assert it.
321
+ const declaredPolicy = declaredReviewPolicy(task.files);
322
+ const policy = raiseReviewPolicy(declaredPolicy, cfg.review.policy);
323
+ // PROMOTION: the declared assignment is a claim about paths, and the diff is the evidence. A
324
+ // judge-only task whose diff left the leaf class is reviewed in full — the claim never outranks
325
+ // what actually happened, and an empty diff promotes too (a skip earned by an absence is not earned).
326
+ let promotedBy = null;
327
+ if (policy === "judge-only") {
328
+ const touched = await changedPaths(worktree, baseRef);
329
+ // Two ways a path leaves the leaf class. It is not a member (`docs/tool.ts`, `docs/Makefile`) — or
330
+ // it is a root version mirror whose diff moved more than the version field, which no path predicate
331
+ // can see. `package.json` carries the gate commands, so a scripts edit hiding behind a leaf-class
332
+ // path is exactly the promotion this buys.
333
+ const nonMembers = touched.filter((p) => !isReviewLeafPath(p));
334
+ const impostorMirrors = (await Promise.all(touched.filter((p) => REVIEW_VERSION_MIRRORS.has(p))
335
+ .map(async (p) => await mirrorsVersionOnly(worktree, baseRef, p) ? null : p))).filter((p) => p !== null);
336
+ // The fail-closed backstop for the compile lint. reviewParticipationErrors runs at a repo root the
337
+ // compile seam cannot always name (collateral.ts); THIS gate is handed the run's real config, so a
338
+ // critical path that reached dispatch is reviewed here whatever the lint saw. The shipped defaults
339
+ // are unioned in for the same reason they are there: a config that names none still has a floor.
340
+ const criticalHits = criticalPathHits([...task.files, ...touched], [...new Set([...DEFAULT_REVIEW_CRITICAL_PATHS, ...(cfg.review.criticalPaths ?? [])])]);
341
+ const escaped = [...new Set([...nonMembers, ...impostorMirrors, ...criticalHits])].sort();
342
+ if (touched.length > 0 && escaped.length === 0) {
343
+ // A declined review makes NO green claim. types.ts's T11 note ("pass stays true so enforcement is
344
+ // unchanged") describes the baseline skips — a build/test/lint command the repo never configured.
345
+ // R3 overrules it here: a cross-vendor review that did not run cannot report a pass, so the record
346
+ // carries verdict "skipped" with the policy that declined it and the reason, and pass is never
347
+ // true. The merge-predicate seam this opens is named in the collateral note above.
348
+ return {
349
+ gate: "review",
350
+ pass: false,
351
+ details: `skipped — reviewPolicy judge-only: every declared path is docs/CHANGELOG/RELEASING/version-mirror leaf work and the diff stayed in that class (${touched.join(", ")})`,
352
+ meta: {
353
+ skipped: true,
354
+ verdict: "skipped",
355
+ policy: "judge-only",
356
+ reason: "every declared path and every path this diff touched is provably leaf-class work",
357
+ paths: touched,
358
+ },
359
+ };
360
+ }
361
+ promotedBy = escaped.length > 0 ? escaped : [];
274
362
  }
363
+ // Every verdict this gate reports from here on was produced under `full` — either declared full, or
364
+ // promoted here. `promotedBy` names the paths that bought the promotion, so the record shows WHY.
365
+ const policyMeta = {
366
+ policy: "full",
367
+ ...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
368
+ };
275
369
  const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
276
370
  if (!reviewer) {
277
371
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
@@ -326,9 +420,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
326
420
  if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
327
421
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
328
422
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
329
- const cause = raw.trim().length === 0
330
- ? "empty-output"
331
- : !raw.includes(nonce) ? "no-verdict" : "malformed-verdict";
423
+ const cause = classifyVerdictCause(raw, nonce, "approve");
332
424
  let saved;
333
425
  if (artifactDir) {
334
426
  try {
@@ -339,11 +431,14 @@ The top-level comments array is optional. Use it only for actionable line-anchor
339
431
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
340
432
  }
341
433
  }
434
+ const failure = cause === "malformed-verdict"
435
+ ? "review output unparseable"
436
+ : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
342
437
  return {
343
438
  gate: "review",
344
439
  pass: false,
345
- details: `review output unparseable (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
346
- meta: { reviewer: channelKey(reviewer), unparseable: true, cause },
440
+ details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
441
+ meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
347
442
  };
348
443
  }
349
444
  const decided = findings !== null
@@ -354,6 +449,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
354
449
  gate: "review",
355
450
  pass: decided.pass,
356
451
  details: appendAnchoredReview(prose, v),
357
- meta: { reviewer: channelKey(reviewer) },
452
+ meta: { ...policyMeta, reviewer: channelKey(reviewer) },
358
453
  };
359
454
  }
@@ -9,6 +9,7 @@ export type GateEvent = {
9
9
  gate: GateName;
10
10
  index: number;
11
11
  total: number;
12
+ parentAt?: number;
12
13
  } | {
13
14
  phase: "end";
14
15
  gate: GateName;
@@ -27,8 +28,16 @@ export interface GateContext {
27
28
  via?: GateVia;
28
29
  excludeReviewers?: string[];
29
30
  artifactDir?: string;
31
+ pipeline?: "v185" | "legacy";
32
+ selectTests?: boolean;
30
33
  onGate?: (e: GateEvent) => void | Promise<void>;
31
34
  }
35
+ /**
36
+ * The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
37
+ * npm/yarn/pnpm/npx script wrappers need one `--` to forward positional filters to the underlying runner;
38
+ * a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
39
+ */
40
+ export declare function testCommandForFiles(testCmd: string, files: string[]): string;
32
41
  export declare function runGates(task: Task, ctx: GateContext): Promise<{
33
42
  results: GateResult[];
34
43
  commits: string[];