tickmarkr 1.87.0 → 1.90.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/adapters/catalog.d.ts +18 -1
  2. package/dist/adapters/catalog.js +44 -1
  3. package/dist/adapters/fake.d.ts +2 -1
  4. package/dist/adapters/fake.js +7 -0
  5. package/dist/adapters/grok.js +11 -0
  6. package/dist/adapters/kimi.d.ts +2 -1
  7. package/dist/adapters/kimi.js +36 -0
  8. package/dist/adapters/opencode.js +17 -0
  9. package/dist/adapters/pi.js +11 -0
  10. package/dist/adapters/prompt.js +8 -1
  11. package/dist/adapters/registry.js +76 -57
  12. package/dist/adapters/types.d.ts +34 -3
  13. package/dist/adapters/types.js +99 -1
  14. package/dist/cli/commands/approve.d.ts +2 -0
  15. package/dist/cli/commands/approve.js +104 -84
  16. package/dist/cli/commands/compile.d.ts +1 -1
  17. package/dist/cli/commands/compile.js +29 -12
  18. package/dist/cli/commands/doctor.d.ts +16 -0
  19. package/dist/cli/commands/doctor.js +52 -0
  20. package/dist/cli/commands/init.js +2 -1
  21. package/dist/cli/commands/plan.d.ts +1 -1
  22. package/dist/cli/commands/plan.js +10 -1
  23. package/dist/cli/commands/report.js +49 -0
  24. package/dist/cli/commands/status.js +298 -96
  25. package/dist/cli/commands/verify.d.ts +9 -0
  26. package/dist/cli/commands/verify.js +177 -0
  27. package/dist/cli/harness.d.ts +13 -0
  28. package/dist/cli/harness.js +50 -0
  29. package/dist/cli/index.d.ts +1 -1
  30. package/dist/cli/index.js +3 -1
  31. package/dist/compile/collateral.js +11 -11
  32. package/dist/compile/common.js +2 -2
  33. package/dist/compile/index.d.ts +14 -3
  34. package/dist/compile/index.js +36 -10
  35. package/dist/compile/native.d.ts +15 -1
  36. package/dist/compile/native.js +310 -28
  37. package/dist/config/config.js +2 -2
  38. package/dist/drivers/subprocess.d.ts +6 -1
  39. package/dist/drivers/subprocess.js +9 -4
  40. package/dist/gates/acceptance.d.ts +21 -1
  41. package/dist/gates/acceptance.js +67 -22
  42. package/dist/gates/artifact-manifest.d.ts +119 -0
  43. package/dist/gates/artifact-manifest.js +357 -0
  44. package/dist/gates/baseline.d.ts +45 -5
  45. package/dist/gates/baseline.js +119 -15
  46. package/dist/gates/llm.js +37 -26
  47. package/dist/gates/review.d.ts +16 -11
  48. package/dist/gates/review.js +44 -150
  49. package/dist/gates/run-gates.js +124 -7
  50. package/dist/gates/scope.js +3 -3
  51. package/dist/graph/files-glob.d.ts +18 -0
  52. package/dist/graph/files-glob.js +22 -0
  53. package/dist/graph/schema.d.ts +3 -1
  54. package/dist/graph/schema.js +4 -1
  55. package/dist/run/daemon.d.ts +44 -0
  56. package/dist/run/daemon.js +2334 -1973
  57. package/dist/run/git.d.ts +53 -0
  58. package/dist/run/git.js +119 -5
  59. package/dist/run/interactive-seed.d.ts +6 -2
  60. package/dist/run/interactive-seed.js +72 -5
  61. package/dist/run/journal.d.ts +9 -1
  62. package/dist/run/journal.js +99 -9
  63. package/dist/run/lock.d.ts +11 -0
  64. package/dist/run/lock.js +97 -6
  65. package/dist/run/merge.d.ts +4 -1
  66. package/dist/run/merge.js +26 -7
  67. package/dist/run/outcome.d.ts +50 -0
  68. package/dist/run/outcome.js +152 -0
  69. package/dist/run/protocol.d.ts +460 -0
  70. package/dist/run/protocol.js +433 -0
  71. package/dist/run/supervision.d.ts +29 -0
  72. package/dist/run/supervision.js +189 -0
  73. package/fixtures/authoring-lints/01-awk-range-self-pass.spec.md +12 -0
  74. package/fixtures/authoring-lints/02-judge-text-key-miss.spec.md +7 -0
  75. package/fixtures/authoring-lints/03-c1-t41-rendered-observable.spec.md +8 -0
  76. package/fixtures/authoring-lints/04-c1-t24-named-file.spec.md +8 -0
  77. package/fixtures/authoring-lints/05-c2-t24-t28-dep-inversion.spec.md +7 -0
  78. package/fixtures/authoring-lints/06-c2-denumbered-coupling.spec.md +7 -0
  79. package/fixtures/authoring-lints/07-c3a-t41-line-count-proxy.spec.md +7 -0
  80. package/fixtures/authoring-lints/08-c3b-t41-governance-referent.spec.md +7 -0
  81. package/fixtures/authoring-lints/09-c4-universals-without-pointer.spec.md +7 -0
  82. package/fixtures/authoring-lints/10-c5-t34-conjunct-flood.spec.md +7 -0
  83. package/fixtures/authoring-lints/11-c6-t34-q3-q9-q20-bundle.spec.md +7 -0
  84. package/fixtures/authoring-lints/12-c7-t24-prose-seam.spec.md +8 -0
  85. package/fixtures/wrapped-acceptance.native.md +29 -0
  86. package/package.json +1 -1
  87. package/schema/rungraph.schema.json +21 -2
  88. package/skills/tickmarkr-overseer/SKILL.md +262 -5
  89. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
  90. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +80 -0
  91. package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
  92. package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
  93. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +201 -0
@@ -1,6 +1,6 @@
1
1
  import { existsSync, readFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
- import { sh } from "../run/git.js";
3
+ import { DEFAULT_SHELL_TIMEOUT_MS, sh } from "../run/git.js";
4
4
  // incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
5
5
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
6
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
@@ -57,6 +57,30 @@ const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) ||
57
57
  const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
58
58
  || TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
59
59
  const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
60
+ // T9 — the infra/regression discriminator. A runner that died because the MACHINE ran out of
61
+ // processes, file descriptors or memory never finished asking the question, so its nonzero exit is
62
+ // not evidence about the work. But the reverse mistake is the expensive one: a real regression that
63
+ // happens to be printed next to an errno token must never be laundered into "infra" and forgiven.
64
+ // So the errno tokens below classify a line as infra only when NOTHING on that line also names a
65
+ // test-level failure — "AssertionError after spawn EAGAIN" names one and is a regression; "spawn
66
+ // EAGAIN" and "Error: spawn EAGAIN" name none and are infra. One regression line anywhere in the
67
+ // output makes the whole output a regression, whatever else the runner printed.
68
+ const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable/;
69
+ // A named error CLASS ("AssertionError", "TypeError", "MyDomainError") — never bare "Error", which
70
+ // is what an errno report itself is headed with (`Error: spawn EAGAIN`). The prefix is required.
71
+ const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
72
+ const isInfraLine = (l) => INFRA_RE.test(l) && !ERROR_CLASS_RE.test(l) && !namesFailure(l) && !SUMMARY_FAIL_RE.test(l);
73
+ const namesRegression = (l) => (isFailureShaped(l) || ERROR_CLASS_RE.test(l)) && !isInfraLine(l);
74
+ /**
75
+ * What a nonzero runner exit is evidence OF. `undefined` when the output names neither — the
76
+ * unreadable-runner case the existing fail-closed path already owns.
77
+ */
78
+ export function classifyFailureOutput(output) {
79
+ const lines = output.split("\n").map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l));
80
+ if (lines.some(namesRegression))
81
+ return "regression";
82
+ return lines.some(isInfraLine) ? "infra" : undefined;
83
+ }
60
84
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
61
85
  // A failing command whose output holds no shape any runner here names. The marker is content-free and
62
86
  // constant: downstream consumers (tip verify journals fingerprint counts) still see that the command
@@ -87,6 +111,18 @@ const unrecognizedEvidence = (raw) => raw
87
111
  // stored baselines may predate ANSI/pass-marker hardening — renormalize at compare time so existing
88
112
  // on-disk baseline.json files stay comparable without recapture (compat invariant, CLAUDE.md)
89
113
  const renormalize = (fp) => normalizeLine(fp.replace(ANSI_RE, ""));
114
+ /**
115
+ * Q121s: the battery's forgiveness math, exported so tip-verify applies the IDENTICAL rule.
116
+ * Returns the failure fingerprints of `raw` that are NOT in the baseline entry (fresh), and
117
+ * whether the output carried no recognizable failure shape at all.
118
+ */
119
+ export function freshFailures(entry, raw) {
120
+ const known = new Set((entry?.fingerprints ?? []).map(renormalize));
121
+ // OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
122
+ const current = fingerprint(raw);
123
+ const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
124
+ return { failing: fresh.filter((f) => f !== UNRECOGNIZED_FAILURE), unreadable: current.includes(UNRECOGNIZED_FAILURE) };
125
+ }
90
126
  export function detectGateCommands(repoRoot, cfg) {
91
127
  const out = {};
92
128
  const pkgPath = join(repoRoot, "package.json");
@@ -121,17 +157,59 @@ function missingConfiguredCommand(cmd, result) {
121
157
  const output = `${result.stdout}\n${result.stderr}`;
122
158
  return new RegExp(`(?:^|[:\\s])${reEscape(token)}:\\s+(?:command not found|No such file or directory)`, "i").test(output);
123
159
  }
160
+ /**
161
+ * How much longer than the baseline a battery may legitimately take. Gate batteries run inside a
162
+ * worktree beside up to `concurrency` siblings, so the same suite is genuinely slower than its
163
+ * pristine-tree capture — 3× is the headroom, not a performance budget.
164
+ */
165
+ const CEILING_SLACK = 3;
166
+ /**
167
+ * The ceiling a battery runs under: whichever is larger of the shipped constant and the slack applied
168
+ * to what this command actually measured. A pre-v1.90 baseline (no measurement) gets exactly the
169
+ * shipped constant, so nothing regresses on an existing on-disk baseline.json — the same read-old
170
+ * compat rule the fingerprint renormalizer follows.
171
+ */
172
+ export const effectiveCeilingMs = (entry) => entry?.ceilingMs ?? Math.max(DEFAULT_SHELL_TIMEOUT_MS, CEILING_SLACK * (entry?.durationMs ?? 0));
173
+ /**
174
+ * The ONLY producer of a `ceiling-kill` gate result. It is gated on `timedOut`, which git.ts sets in
175
+ * exactly one place — inside the timer callback that issues the SIGKILL — so a process that exited on
176
+ * its own can never reach this text however it exited: a slow red suite, an unreadable runner, exit
177
+ * 137 from a kill somebody ELSE sent. Returning `undefined` rather than a result keeps that structural:
178
+ * every caller must go on to the ordinary verdict path, it cannot fall through into this one.
179
+ *
180
+ * A kill is not a verdict. The battery was still running when the machine took it away, so its exit
181
+ * code is evidence about the ceiling and nothing about the work — which is why this reports `infra`
182
+ * and never enters baseline forgiveness (there is no runner output to forgive).
183
+ */
184
+ export function ceilingKillResult(gate, r, ceilingMs) {
185
+ if (r.timedOut !== true)
186
+ return undefined;
187
+ // no measurement only when a caller synthesized the result; a real kill fires AT the ceiling
188
+ const durationMs = r.durationMs ?? ceilingMs;
189
+ return {
190
+ gate,
191
+ pass: false,
192
+ details: `ceiling-kill: SIGKILLed after ${durationMs}ms at the configured ${ceilingMs}ms ceiling — `
193
+ + `the battery never returned a verdict, so this gate verified nothing (exit ${r.code} is the kill, not a test result)`,
194
+ meta: { classification: "infra", infra: true, kind: "ceiling-kill", durationMs, ceilingMs },
195
+ };
196
+ }
124
197
  export async function captureBaseline(cwd, commands) {
125
198
  const base = { commands: {} };
126
199
  for (const [name, cmd] of Object.entries(commands)) {
127
200
  const r = await sh(cmd, cwd);
128
201
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
202
+ // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
203
+ // scales the next ceiling up — the right direction for a suite that never finished once.
204
+ const durationMs = r.durationMs ?? 0;
129
205
  base.commands[name] = {
130
206
  exitCode: r.code,
131
207
  // a command that exits 0 has no failures to fingerprint — recording any would be a lie the
132
208
  // compare step then has to forgive
133
209
  fingerprints: r.code === 0 ? [] : fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")),
134
210
  missingCommand: missingConfiguredCommand(cmd, r),
211
+ durationMs,
212
+ ceilingMs: effectiveCeilingMs({ durationMs }),
135
213
  };
136
214
  }
137
215
  const names = Object.keys(commands);
@@ -199,16 +277,37 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
199
277
  results.push({ gate: name, pass: true, details: `no ${name} command detected — skipped`, meta: { skipped: true } });
200
278
  continue;
201
279
  }
202
- const r = await sh(cmd, cwd);
280
+ const ceilingMs = effectiveCeilingMs(baseline.commands[name]);
281
+ const r = await sh(cmd, cwd, ceilingMs);
282
+ // Q24: the kill is read BEFORE the exit code is interpreted at all. A SIGKILLed battery has
283
+ // whatever partial output it had flushed — typically no failure shape — so every path below
284
+ // would otherwise turn a timeout into a claim about the work: "no recognizable failure lines"
285
+ // when the baseline was green, or a forgiven pre-existing red when it was not. Neither is true.
286
+ const killed = ceilingKillResult(name, r, ceilingMs);
287
+ if (killed) {
288
+ results.push(killed);
289
+ continue;
290
+ }
203
291
  if (r.code === 0) {
204
292
  results.push({ gate: name, pass: true, details: "exit 0" });
205
293
  continue;
206
294
  }
207
295
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
208
- const known = new Set((baseline.commands[name]?.fingerprints ?? []).map(renormalize));
209
- // OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
210
- const current = fingerprint(raw);
211
- const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
296
+ // T9: classify BEFORE the baseline diff, and record it on every nonzero result. An infra-only
297
+ // exit means the runner never completed a suite, so there is nothing to forgive and nothing
298
+ // verified — it fails, and `meta.infra` marks it so the merge predicate cannot read it as a
299
+ // satisfied gate even if some future producer reports it as a pass. Baseline forgiveness stays
300
+ // exactly where it belongs: on failures the runner actually reported and the baseline already had.
301
+ const classification = classifyFailureOutput(raw);
302
+ if (classification === "infra") {
303
+ results.push({
304
+ gate: name,
305
+ pass: false,
306
+ details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
307
+ meta: { classification, infra: true },
308
+ });
309
+ continue;
310
+ }
212
311
  // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
213
312
  // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
214
313
  // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
@@ -216,8 +315,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
216
315
  // below rather than reading as a verified green. Raise the ceiling by teaching isFailureShaped
217
316
  // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
218
317
  // verdict); loosening back to vocabulary re-opens OBS-278.
219
- const unreadable = current.includes(UNRECOGNIZED_FAILURE);
220
- const failing = fresh.filter((f) => f !== UNRECOGNIZED_FAILURE);
318
+ const { failing, unreadable } = freshFailures(baseline.commands[name], raw);
221
319
  if (!failing.length && (baseline.commands[name]?.exitCode ?? 1) === 0) {
222
320
  const closed = `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
223
321
  const evidence = unrecognizedEvidence(raw);
@@ -225,16 +323,22 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
225
323
  gate: name,
226
324
  pass: false,
227
325
  details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
326
+ ...(classification ? { meta: { classification } } : {}),
228
327
  });
229
328
  continue;
230
329
  }
231
- results.push(failing.length
232
- ? { gate: name, pass: false, ...headlineDetails(raw, failing) }
233
- : {
234
- gate: name,
235
- pass: true,
236
- details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
237
- });
330
+ if (failing.length) {
331
+ const headlined = headlineDetails(raw, failing);
332
+ const meta = { ...headlined.meta, ...(classification ? { classification } : {}) };
333
+ results.push({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
334
+ continue;
335
+ }
336
+ results.push({
337
+ gate: name,
338
+ pass: true,
339
+ details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
340
+ ...(classification ? { meta: { classification } } : {}),
341
+ });
238
342
  }
239
343
  return results;
240
344
  }
package/dist/gates/llm.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import { AsyncLocalStorage } from "node:async_hooks";
2
2
  import { randomBytes } from "node:crypto";
3
- import { mkdtempSync, writeFileSync } from "node:fs";
3
+ import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
6
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
@@ -104,36 +104,47 @@ export async function captureLlmOutput(run) {
104
104
  return { value, outputs };
105
105
  }
106
106
  export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
107
- const pf = join(mkdtempSync(join(tmpdir(), "tickmarkr-llm-")), "prompt.md");
108
- writeFileSync(pf, prompt);
109
- const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
110
- return r.stdout + "\n" + r.stderr;
107
+ const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
108
+ try {
109
+ const pf = join(dir, "prompt.md");
110
+ writeFileSync(pf, prompt);
111
+ const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
112
+ return r.stdout + "\n" + r.stderr;
113
+ }
114
+ finally {
115
+ rmSync(dir, { recursive: true, force: true });
116
+ }
111
117
  }
112
118
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
113
119
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
114
120
  export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
115
121
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
116
- const pf = join(dir, "prompt.md");
117
- writeFileSync(pf, prompt);
118
- const scriptPath = join(dir, "dispatch.sh");
119
- // OBS-50: bootstrap in a script beside the prompt — pane sees one short bash line + banner, not the raw inline command
120
- const nonce = extractPromptNonce(prompt) ?? generateVerdictNonce();
121
- writeFileSync(scriptPath, [
122
- "export BASH_SILENCE_DEPRECATION_WARNING=1",
123
- bannerShell(),
124
- adapter.headlessCommand(pf, model),
125
- gateExitTrailer(nonce),
126
- ].join("\n"));
127
- const slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
128
- via.onSlot?.(slot);
129
- await via.driver.run(slot, paneDispatchCommand(scriptPath));
130
- // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
131
- // false-complete — same guard the worker path uses (daemon.ts:330-331).
132
- await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
133
- const out = await via.driver.read(slot, 400);
134
- if (!via.keep)
135
- await via.driver.close(slot);
136
- return dewrapPaneVerdict(out, nonce);
122
+ try {
123
+ const pf = join(dir, "prompt.md");
124
+ writeFileSync(pf, prompt);
125
+ const scriptPath = join(dir, "dispatch.sh");
126
+ // OBS-50: bootstrap in a script beside the prompt — pane sees one short bash line + banner, not the raw inline command
127
+ const nonce = extractPromptNonce(prompt) ?? generateVerdictNonce();
128
+ writeFileSync(scriptPath, [
129
+ "export BASH_SILENCE_DEPRECATION_WARNING=1",
130
+ bannerShell(),
131
+ adapter.headlessCommand(pf, model),
132
+ gateExitTrailer(nonce),
133
+ ].join("\n"));
134
+ const slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
135
+ via.onSlot?.(slot);
136
+ await via.driver.run(slot, paneDispatchCommand(scriptPath));
137
+ // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
138
+ // false-complete — same guard the worker path uses (daemon.ts:330-331).
139
+ await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
140
+ const out = await via.driver.read(slot, 400);
141
+ if (!via.keep)
142
+ await via.driver.close(slot);
143
+ return dewrapPaneVerdict(out, nonce);
144
+ }
145
+ finally {
146
+ rmSync(dir, { recursive: true, force: true });
147
+ }
137
148
  }
138
149
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
139
150
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
@@ -4,6 +4,8 @@ import { type Task } from "../graph/schema.js";
4
4
  import { type GateVia } from "./llm.js";
5
5
  import type { GateResult } from "./types.js";
6
6
  import { type VerdictUnparseableCause } from "./verdict-cause.js";
7
+ import { type ArtifactDiffMeasurement, type ArtifactDiffSection } from "./artifact-manifest.js";
8
+ export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
7
9
  export type ReviewSeverity = "material" | "minor";
8
10
  export interface ReviewFinding {
9
11
  note: string;
@@ -21,13 +23,6 @@ export interface ReviewVerdict {
21
23
  body: string;
22
24
  }>;
23
25
  }
24
- export declare const REGENERABLE_CAPTURE_PATHS: readonly string[];
25
- export declare const PROTECTED_EVIDENCE_PREFIXES: readonly ["tests/fixtures/cockpit/anchors/", "tests/fixtures/cockpit/sources/", "tests/fixtures/cockpit/colour/sources/"];
26
- export declare function isProtectedEvidence(path: string): boolean;
27
- /** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
28
- export declare function setAsideReceiptPath(section: string): string | null;
29
- /** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
30
- export declare function setAsideRegenerableCaptures(diff: string): string;
31
26
  /**
32
27
  * The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
33
28
  * never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
@@ -35,11 +30,21 @@ export declare function setAsideRegenerableCaptures(diff: string): string;
35
30
  */
36
31
  export declare function changedPaths(worktree: string, baseRef: string): Promise<string[]>;
37
32
  export declare function mirrorsVersionOnly(worktree: string, baseRef: string, path: string): Promise<boolean>;
38
- export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
39
- full: string;
40
- forCap: string;
41
- }>;
33
+ export type TaskDiffMeasurement = {
34
+ readonly full: string;
35
+ readonly forCap: string;
36
+ /** Strict-cap UTF-8 bytes left after capture payloads become receipts. */
37
+ readonly logicBytes: number;
38
+ /** UTF-8 bytes withheld by those receipts and charged to the larger cap. */
39
+ readonly captureBytes: number;
40
+ readonly classifications: readonly ArtifactDiffSection[];
41
+ readonly fullMeasurement: ArtifactDiffMeasurement;
42
+ readonly capMeasurement: ArtifactDiffMeasurement;
43
+ };
44
+ export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
42
45
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
46
+ /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
47
+ export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
43
48
  export declare function isDiffCapPark(result: GateResult): boolean;
44
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
45
50
  export declare function modelId(model: string): string;
@@ -9,6 +9,8 @@ import { redactSecrets } from "../run/redact.js";
9
9
  import { marginalCostRank } from "../route/router.js";
10
10
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
11
11
  import { classifyVerdictCause } from "./verdict-cause.js";
12
+ import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
13
+ export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
12
14
  // legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
13
15
  function classifyReviewIssues(approve, issues) {
14
16
  const inconsistencies = [];
@@ -69,149 +71,6 @@ function classifyReviewFindings(findings) {
69
71
  // OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
70
72
  // one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
71
73
  const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
72
- // v1.82 T1 — the cap bounds what a READER MUST READ, not what a run must write. A regeneration of the
73
- // frame corpora is ~134KB of `-U0` measurement before a source line changes, and nobody reads it: those
74
- // frames are asserted byte-for-byte by the corpus tests. Counting them is the category error this
75
- // removes. The two artifacts stay two (OBS-48: the cap measures -U0, the reader receives the full diff);
76
- // the exclusion is applied to both, identically, right here so BOTH measuring gates inherit it.
77
- //
78
- // Clause 1 — membership is an EXACT PATH match against the shipped capture manifest, the same lists the
79
- // regeneration path itself uses. Location, directory depth and file extension confer nothing: an
80
- // unmanifested file sitting beside real frames is measured and shown in full. (The anchors deliberately
81
- // share basenames with the frames; only the full path separates the oracle from its regenerable twin.)
82
- //
83
- // The members are LISTED here rather than imported from the manifest module, for one measured reason:
84
- // that module is the Ink/React renderer, and importing it puts the whole TUI in every gate's module
85
- // graph — which also memoises chalk's colour level at import time and turns the fleet suite red. A
86
- // drift test in tests/gates/diff-cap.test.ts asserts this list is exactly GOLDEN_FRAME_CASES +
87
- // COLOUR_FRAME_CASES and that every entry exists on disk, so it is a copy that cannot drift rather
88
- // than a second source of truth: add or rename a frame case and that test goes red until this matches.
89
- export const REGENERABLE_CAPTURE_PATHS = [
90
- "tests/fixtures/cockpit/frames/run.width-stacked.80x24.txt",
91
- "tests/fixtures/cockpit/frames/run.width-folded-keys.100x24.txt",
92
- "tests/fixtures/cockpit/frames/run.width-three-column.140x24.txt",
93
- "tests/fixtures/cockpit/frames/run.height-14.140x14.txt",
94
- "tests/fixtures/cockpit/frames/run.height-18.140x18.txt",
95
- "tests/fixtures/cockpit/frames/run.height-24.140x24.txt",
96
- "tests/fixtures/cockpit/frames/run.height-40.140x40.txt",
97
- "tests/fixtures/cockpit/frames/run.no-colour.140x24.txt",
98
- "tests/fixtures/cockpit/frames/run.non-tty.140x24.txt",
99
- "tests/fixtures/cockpit/frames/run.ci.140x24.txt",
100
- "tests/fixtures/cockpit/frames/setup.width-stacked.80x24.txt",
101
- "tests/fixtures/cockpit/frames/setup.width-folded-keys.100x24.txt",
102
- "tests/fixtures/cockpit/frames/setup.width-three-column.140x24.txt",
103
- "tests/fixtures/cockpit/frames/setup.height-14.140x14.txt",
104
- "tests/fixtures/cockpit/frames/setup.height-18.140x18.txt",
105
- "tests/fixtures/cockpit/frames/setup.height-24.140x24.txt",
106
- "tests/fixtures/cockpit/frames/setup.height-40.140x40.txt",
107
- "tests/fixtures/cockpit/frames/setup.no-colour.140x24.txt",
108
- "tests/fixtures/cockpit/frames/setup.non-tty.140x24.txt",
109
- "tests/fixtures/cockpit/frames/setup.ci.140x24.txt",
110
- "tests/fixtures/cockpit/colour/run-20260718-000943.colour.140x24.txt",
111
- "tests/fixtures/cockpit/colour/run-20260718-000943.no-colour.140x24.txt",
112
- "tests/fixtures/cockpit/colour/run-20260725-025004.interrupted.colour.140x24.txt",
113
- ];
114
- const CAPTURE_MANIFEST = new Set(REGENERABLE_CAPTURE_PATHS);
115
- // Clause 2 — the frozen appearance anchors and the captured engagement journals are NEVER set aside and
116
- // are exempt from every other reduction too (clause 5): they are reviewed, immutable evidence. The
117
- // anchors are the oracle this milestone declares, and the fixture law bans editing a captured journal to
118
- // satisfy an assertion — so both keep counting toward the cap and keep reaching a reader verbatim.
119
- export const PROTECTED_EVIDENCE_PREFIXES = [
120
- "tests/fixtures/cockpit/anchors/",
121
- "tests/fixtures/cockpit/sources/",
122
- "tests/fixtures/cockpit/colour/sources/",
123
- ];
124
- export function isProtectedEvidence(path) {
125
- return PROTECTED_EVIDENCE_PREFIXES.some((prefix) => path.startsWith(prefix));
126
- }
127
- const SET_ASIDE_RECEIPT = /^set aside: regenerable capture (.+?) — \d+ bytes withheld\b/m;
128
- /** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
129
- export function setAsideReceiptPath(section) {
130
- return SET_ASIDE_RECEIPT.exec(section)?.[1] ?? null;
131
- }
132
- // `--- a/x` / `+++ b/x` → "x"; "/dev/null" → null, which is the ABSENCE of a side, not a membership
133
- // failure (clause 3). git quotes paths carrying specials, so unquote before stripping the a/ b/ prefix.
134
- function diffSidePath(raw) {
135
- const v = raw.trim();
136
- if (v === "/dev/null")
137
- return null;
138
- const unquoted = v.startsWith('"') && v.endsWith('"') ? v.slice(1, -1) : v;
139
- return unquoted.replace(/^[ab]\//, "");
140
- }
141
- // The `--- a/x` / `+++ b/x` header pair and the hunk lines below it. Clause 4 — null when the section
142
- // carries NO content hunk at all (a mode-only change, a pure rename, a binary marker): such a section is
143
- // left exactly as git wrote it rather than handed a manufactured receipt.
144
- function parseSection(section) {
145
- const lines = section.split("\n");
146
- const minus = lines.findIndex((l) => l.startsWith("--- "));
147
- if (minus === -1 || !lines[minus + 1]?.startsWith("+++ "))
148
- return null;
149
- const body = lines.slice(minus + 2);
150
- if (!body.some((l) => l.startsWith("@@ ")))
151
- return null;
152
- return { lines, minus, sides: [diffSidePath(lines[minus].slice(4)), diffSidePath(lines[minus + 1].slice(4))], body };
153
- }
154
- // The content a one-sided section carries: its hunk lines with the sign stripped, keeping git's
155
- // `` markers so a trailing-newline difference still reads as a content
156
- // difference. Hunk headers are dropped — a delete and an add of the same bytes never share them.
157
- function hunkPayload(body, sign) {
158
- return body.filter((l) => l.startsWith(sign) || l.startsWith("\\")).map((l) => (l[0] === sign ? l.slice(1) : l)).join("\n");
159
- }
160
- // Clause 4 — a hunk is NOT proof of a content change. Git spells a KIND change (regular file ⇄ symlink)
161
- // as a delete plus an add of the SAME path, each carrying a hunk, so when the regular file's bytes are
162
- // exactly the link target both halves carry identical payloads: the kind changed and the content did
163
- // not. Nothing was withheld, so neither half earns a receipt — and neither half can see the other, so
164
- // the pairing is found across sections before any one of them is set aside.
165
- function kindOnlyPaths(sections) {
166
- const removed = new Map();
167
- const added = new Map();
168
- for (const section of sections) {
169
- const parsed = parseSection(section);
170
- if (!parsed)
171
- continue;
172
- const [a, b] = parsed.sides;
173
- if (a && !b)
174
- removed.set(a, hunkPayload(parsed.body, "-"));
175
- else if (b && !a)
176
- added.set(b, hunkPayload(parsed.body, "+"));
177
- }
178
- return new Set([...removed].filter(([path, payload]) => added.get(path) === payload).map(([path]) => path));
179
- }
180
- function setAsideSection(section, kindOnly) {
181
- const parsed = parseSection(section);
182
- if (!parsed)
183
- return section;
184
- const { lines, minus, sides } = parsed;
185
- // Clause 3 — the test is over the sides that name a real file. Requiring BOTH sides to be members
186
- // makes every corpus addition (a-side /dev/null) and deletion (b-side /dev/null) ineligible, which is
187
- // exactly the frames this milestone adds. A rename crossing the boundary in either direction has two
188
- // real sides and one of them is not a member, so it stays whole — `rename from` line included.
189
- const named = sides.filter((p) => p !== null);
190
- if (!named.length || !named.every((p) => CAPTURE_MANIFEST.has(p)))
191
- return section;
192
- // Clause 4 — both halves of a content-identical kind change are left exactly as git wrote them. A
193
- // kind change that DID move bytes is an ordinary content change and is set aside like any other.
194
- if (named.some((p) => kindOnly.has(p)))
195
- return section;
196
- const withheld = lines.slice(minus).join("\n");
197
- // Clause 6 — the claimed size is the UTF-8 BYTE length of what was withheld (the file headers and
198
- // hunks this receipt replaces), never a JavaScript string length: box-drawing frames make the two
199
- // disagree. And the receipt itself is part of the measured artifact, so N set-aside sections can never
200
- // measure as nothing while producing an arbitrarily large reader payload.
201
- const receipt = `set aside: regenerable capture ${named.at(-1)} — ${Buffer.byteLength(withheld, "utf8")} bytes withheld (regenerable frame corpus: asserted byte-for-byte by the corpus tests, never read)`;
202
- // Clause 4 — everything git said happened survives verbatim: old/new mode, new file mode, deleted file
203
- // mode, similarity index, rename from/to, index. Only content is replaced, so a deletion is never
204
- // presented as a file that still exists and an addition is never presented as a modification.
205
- return `${lines.slice(0, minus).join("\n")}\n${receipt}\n`;
206
- }
207
- /** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
208
- export function setAsideRegenerableCaptures(diff) {
209
- if (!diff.includes("diff --git "))
210
- return diff;
211
- const sections = diff.split(/(?=^diff --git )/m);
212
- const kindOnly = kindOnlyPaths(sections);
213
- return sections.map((s) => setAsideSection(s, kindOnly)).join("");
214
- }
215
74
  /**
216
75
  * The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
217
76
  * never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
@@ -232,7 +91,7 @@ const VERSION_FIELD_LINE_RE = /^[+-]\s*"version":\s*"[^"]*",?\s*$/;
232
91
  export async function mirrorsVersionOnly(worktree, baseRef, path) {
233
92
  let diff;
234
93
  try {
235
- diff = await shOk(`git diff -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
94
+ diff = await shOk(`git diff --full-index -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
236
95
  }
237
96
  catch {
238
97
  return false;
@@ -241,9 +100,23 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
241
100
  return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
242
101
  }
243
102
  export async function fetchTaskDiff(worktree, baseRef) {
244
- const full = setAsideRegenerableCaptures(await shOk(`git diff '${baseRef}..HEAD'`, worktree));
245
- const forCap = setAsideRegenerableCaptures(await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree));
246
- return { full, forCap };
103
+ // --full-index: abbreviated index lines vary with object-store density, so two measurements of
104
+ // the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
105
+ const [rawFull, rawForCap] = await Promise.all([
106
+ shOk(`git diff --full-index '${baseRef}..HEAD'`, worktree),
107
+ shOk(`git diff --full-index -U0 '${baseRef}..HEAD'`, worktree),
108
+ ]);
109
+ const fullMeasurement = measureArtifactDiff(rawFull);
110
+ const capMeasurement = measureArtifactDiff(rawForCap);
111
+ return {
112
+ full: fullMeasurement.rendered,
113
+ forCap: capMeasurement.rendered,
114
+ logicBytes: Buffer.byteLength(reviewableLogicDiff(capMeasurement.rendered), "utf8"),
115
+ captureBytes: capMeasurement.captureBytes,
116
+ classifications: capMeasurement.sections,
117
+ fullMeasurement,
118
+ capMeasurement,
119
+ };
247
120
  }
248
121
  export function checkDiffCap(gate, measured, cap, prefix = "") {
249
122
  if (measured <= cap)
@@ -256,8 +129,26 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
256
129
  meta: { park: "human" },
257
130
  };
258
131
  }
132
+ /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
133
+ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
134
+ const logicFail = checkDiffCap(gate, measured.logicBytes, logicCap, prefix);
135
+ if (logicFail)
136
+ return logicFail;
137
+ const captureCap = captureDiffCapFor(logicCap);
138
+ if (measured.captureBytes <= captureCap)
139
+ return null;
140
+ return {
141
+ gate,
142
+ pass: false,
143
+ details: prefix
144
+ + `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
145
+ meta: { park: "human" },
146
+ };
147
+ }
259
148
  export function isDiffCapPark(result) {
260
- return result.pass === false && result.meta?.park === "human" && /diff exceeds verifiable cap/i.test(result.details);
149
+ return result.pass === false
150
+ && result.meta?.park === "human"
151
+ && /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
261
152
  }
262
153
  // ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
263
154
  export function diffCapParkReason(results) {
@@ -374,9 +265,12 @@ artifactDir) {
374
265
  ? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
375
266
  : { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
376
267
  }
377
- const { full: diff, forCap } = await fetchTaskDiff(worktree, baseRef);
268
+ const measuredDiff = await fetchTaskDiff(worktree, baseRef);
269
+ // Keep the reader payload identical to the text charged to the strict cap:
270
+ // whole-file source deletions are represented by their citable operation fact.
271
+ const diff = reviewableLogicDiff(measuredDiff.full);
378
272
  const diffCap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP;
379
- const capFail = checkDiffCap("review", forCap.length, diffCap);
273
+ const capFail = checkTaskDiffCaps("review", measuredDiff, diffCap);
380
274
  if (capFail)
381
275
  return capFail;
382
276
  const nonce = generateVerdictNonce();