tickmarkr 1.87.0 → 1.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/adapters/catalog.d.ts +18 -1
  2. package/dist/adapters/catalog.js +44 -1
  3. package/dist/adapters/fake.d.ts +2 -1
  4. package/dist/adapters/fake.js +7 -0
  5. package/dist/adapters/grok.js +11 -0
  6. package/dist/adapters/kimi.d.ts +2 -1
  7. package/dist/adapters/kimi.js +36 -0
  8. package/dist/adapters/opencode.js +17 -0
  9. package/dist/adapters/pi.js +11 -0
  10. package/dist/adapters/prompt.js +8 -1
  11. package/dist/adapters/registry.js +76 -57
  12. package/dist/adapters/types.d.ts +34 -3
  13. package/dist/adapters/types.js +99 -1
  14. package/dist/cli/commands/approve.d.ts +2 -0
  15. package/dist/cli/commands/approve.js +104 -84
  16. package/dist/cli/commands/compile.d.ts +1 -1
  17. package/dist/cli/commands/compile.js +29 -12
  18. package/dist/cli/commands/init.js +1 -1
  19. package/dist/cli/commands/plan.d.ts +1 -1
  20. package/dist/cli/commands/plan.js +10 -1
  21. package/dist/cli/commands/report.js +49 -0
  22. package/dist/cli/commands/status.js +298 -96
  23. package/dist/cli/harness.d.ts +13 -0
  24. package/dist/cli/harness.js +50 -0
  25. package/dist/compile/collateral.js +4 -4
  26. package/dist/compile/index.d.ts +14 -3
  27. package/dist/compile/index.js +36 -10
  28. package/dist/compile/native.js +101 -25
  29. package/dist/drivers/subprocess.d.ts +6 -1
  30. package/dist/drivers/subprocess.js +9 -4
  31. package/dist/gates/acceptance.d.ts +21 -1
  32. package/dist/gates/acceptance.js +67 -22
  33. package/dist/gates/artifact-manifest.d.ts +119 -0
  34. package/dist/gates/artifact-manifest.js +357 -0
  35. package/dist/gates/baseline.d.ts +6 -0
  36. package/dist/gates/baseline.js +52 -7
  37. package/dist/gates/llm.js +37 -26
  38. package/dist/gates/review.d.ts +16 -11
  39. package/dist/gates/review.js +44 -150
  40. package/dist/gates/run-gates.js +124 -7
  41. package/dist/graph/schema.d.ts +3 -1
  42. package/dist/graph/schema.js +4 -1
  43. package/dist/run/daemon.d.ts +42 -0
  44. package/dist/run/daemon.js +2321 -1967
  45. package/dist/run/git.d.ts +50 -0
  46. package/dist/run/git.js +113 -2
  47. package/dist/run/interactive-seed.d.ts +6 -2
  48. package/dist/run/interactive-seed.js +72 -5
  49. package/dist/run/journal.d.ts +9 -1
  50. package/dist/run/journal.js +99 -9
  51. package/dist/run/lock.d.ts +11 -0
  52. package/dist/run/lock.js +97 -6
  53. package/dist/run/outcome.d.ts +50 -0
  54. package/dist/run/outcome.js +152 -0
  55. package/dist/run/protocol.d.ts +460 -0
  56. package/dist/run/protocol.js +433 -0
  57. package/dist/run/supervision.d.ts +29 -0
  58. package/dist/run/supervision.js +189 -0
  59. package/fixtures/wrapped-acceptance.native.md +29 -0
  60. package/package.json +1 -1
  61. package/schema/rungraph.schema.json +21 -2
  62. package/skills/tickmarkr-overseer/SKILL.md +257 -5
  63. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
  64. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
  65. package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
  66. package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
  67. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
@@ -9,6 +9,8 @@ import { redactSecrets } from "../run/redact.js";
9
9
  import { marginalCostRank } from "../route/router.js";
10
10
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
11
11
  import { classifyVerdictCause } from "./verdict-cause.js";
12
+ import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
13
+ export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
12
14
  // legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
13
15
  function classifyReviewIssues(approve, issues) {
14
16
  const inconsistencies = [];
@@ -69,149 +71,6 @@ function classifyReviewFindings(findings) {
69
71
  // OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
70
72
  // one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
71
73
  const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
72
- // v1.82 T1 — the cap bounds what a READER MUST READ, not what a run must write. A regeneration of the
73
- // frame corpora is ~134KB of `-U0` measurement before a source line changes, and nobody reads it: those
74
- // frames are asserted byte-for-byte by the corpus tests. Counting them is the category error this
75
- // removes. The two artifacts stay two (OBS-48: the cap measures -U0, the reader receives the full diff);
76
- // the exclusion is applied to both, identically, right here so BOTH measuring gates inherit it.
77
- //
78
- // Clause 1 — membership is an EXACT PATH match against the shipped capture manifest, the same lists the
79
- // regeneration path itself uses. Location, directory depth and file extension confer nothing: an
80
- // unmanifested file sitting beside real frames is measured and shown in full. (The anchors deliberately
81
- // share basenames with the frames; only the full path separates the oracle from its regenerable twin.)
82
- //
83
- // The members are LISTED here rather than imported from the manifest module, for one measured reason:
84
- // that module is the Ink/React renderer, and importing it puts the whole TUI in every gate's module
85
- // graph — which also memoises chalk's colour level at import time and turns the fleet suite red. A
86
- // drift test in tests/gates/diff-cap.test.ts asserts this list is exactly GOLDEN_FRAME_CASES +
87
- // COLOUR_FRAME_CASES and that every entry exists on disk, so it is a copy that cannot drift rather
88
- // than a second source of truth: add or rename a frame case and that test goes red until this matches.
89
- export const REGENERABLE_CAPTURE_PATHS = [
90
- "tests/fixtures/cockpit/frames/run.width-stacked.80x24.txt",
91
- "tests/fixtures/cockpit/frames/run.width-folded-keys.100x24.txt",
92
- "tests/fixtures/cockpit/frames/run.width-three-column.140x24.txt",
93
- "tests/fixtures/cockpit/frames/run.height-14.140x14.txt",
94
- "tests/fixtures/cockpit/frames/run.height-18.140x18.txt",
95
- "tests/fixtures/cockpit/frames/run.height-24.140x24.txt",
96
- "tests/fixtures/cockpit/frames/run.height-40.140x40.txt",
97
- "tests/fixtures/cockpit/frames/run.no-colour.140x24.txt",
98
- "tests/fixtures/cockpit/frames/run.non-tty.140x24.txt",
99
- "tests/fixtures/cockpit/frames/run.ci.140x24.txt",
100
- "tests/fixtures/cockpit/frames/setup.width-stacked.80x24.txt",
101
- "tests/fixtures/cockpit/frames/setup.width-folded-keys.100x24.txt",
102
- "tests/fixtures/cockpit/frames/setup.width-three-column.140x24.txt",
103
- "tests/fixtures/cockpit/frames/setup.height-14.140x14.txt",
104
- "tests/fixtures/cockpit/frames/setup.height-18.140x18.txt",
105
- "tests/fixtures/cockpit/frames/setup.height-24.140x24.txt",
106
- "tests/fixtures/cockpit/frames/setup.height-40.140x40.txt",
107
- "tests/fixtures/cockpit/frames/setup.no-colour.140x24.txt",
108
- "tests/fixtures/cockpit/frames/setup.non-tty.140x24.txt",
109
- "tests/fixtures/cockpit/frames/setup.ci.140x24.txt",
110
- "tests/fixtures/cockpit/colour/run-20260718-000943.colour.140x24.txt",
111
- "tests/fixtures/cockpit/colour/run-20260718-000943.no-colour.140x24.txt",
112
- "tests/fixtures/cockpit/colour/run-20260725-025004.interrupted.colour.140x24.txt",
113
- ];
114
- const CAPTURE_MANIFEST = new Set(REGENERABLE_CAPTURE_PATHS);
115
- // Clause 2 — the frozen appearance anchors and the captured engagement journals are NEVER set aside and
116
- // are exempt from every other reduction too (clause 5): they are reviewed, immutable evidence. The
117
- // anchors are the oracle this milestone declares, and the fixture law bans editing a captured journal to
118
- // satisfy an assertion — so both keep counting toward the cap and keep reaching a reader verbatim.
119
- export const PROTECTED_EVIDENCE_PREFIXES = [
120
- "tests/fixtures/cockpit/anchors/",
121
- "tests/fixtures/cockpit/sources/",
122
- "tests/fixtures/cockpit/colour/sources/",
123
- ];
124
- export function isProtectedEvidence(path) {
125
- return PROTECTED_EVIDENCE_PREFIXES.some((prefix) => path.startsWith(prefix));
126
- }
127
- const SET_ASIDE_RECEIPT = /^set aside: regenerable capture (.+?) — \d+ bytes withheld\b/m;
128
- /** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
129
- export function setAsideReceiptPath(section) {
130
- return SET_ASIDE_RECEIPT.exec(section)?.[1] ?? null;
131
- }
132
- // `--- a/x` / `+++ b/x` → "x"; "/dev/null" → null, which is the ABSENCE of a side, not a membership
133
- // failure (clause 3). git quotes paths carrying specials, so unquote before stripping the a/ b/ prefix.
134
- function diffSidePath(raw) {
135
- const v = raw.trim();
136
- if (v === "/dev/null")
137
- return null;
138
- const unquoted = v.startsWith('"') && v.endsWith('"') ? v.slice(1, -1) : v;
139
- return unquoted.replace(/^[ab]\//, "");
140
- }
141
- // The `--- a/x` / `+++ b/x` header pair and the hunk lines below it. Clause 4 — null when the section
142
- // carries NO content hunk at all (a mode-only change, a pure rename, a binary marker): such a section is
143
- // left exactly as git wrote it rather than handed a manufactured receipt.
144
- function parseSection(section) {
145
- const lines = section.split("\n");
146
- const minus = lines.findIndex((l) => l.startsWith("--- "));
147
- if (minus === -1 || !lines[minus + 1]?.startsWith("+++ "))
148
- return null;
149
- const body = lines.slice(minus + 2);
150
- if (!body.some((l) => l.startsWith("@@ ")))
151
- return null;
152
- return { lines, minus, sides: [diffSidePath(lines[minus].slice(4)), diffSidePath(lines[minus + 1].slice(4))], body };
153
- }
154
- // The content a one-sided section carries: its hunk lines with the sign stripped, keeping git's
155
- // `` markers so a trailing-newline difference still reads as a content
156
- // difference. Hunk headers are dropped — a delete and an add of the same bytes never share them.
157
- function hunkPayload(body, sign) {
158
- return body.filter((l) => l.startsWith(sign) || l.startsWith("\\")).map((l) => (l[0] === sign ? l.slice(1) : l)).join("\n");
159
- }
160
- // Clause 4 — a hunk is NOT proof of a content change. Git spells a KIND change (regular file ⇄ symlink)
161
- // as a delete plus an add of the SAME path, each carrying a hunk, so when the regular file's bytes are
162
- // exactly the link target both halves carry identical payloads: the kind changed and the content did
163
- // not. Nothing was withheld, so neither half earns a receipt — and neither half can see the other, so
164
- // the pairing is found across sections before any one of them is set aside.
165
- function kindOnlyPaths(sections) {
166
- const removed = new Map();
167
- const added = new Map();
168
- for (const section of sections) {
169
- const parsed = parseSection(section);
170
- if (!parsed)
171
- continue;
172
- const [a, b] = parsed.sides;
173
- if (a && !b)
174
- removed.set(a, hunkPayload(parsed.body, "-"));
175
- else if (b && !a)
176
- added.set(b, hunkPayload(parsed.body, "+"));
177
- }
178
- return new Set([...removed].filter(([path, payload]) => added.get(path) === payload).map(([path]) => path));
179
- }
180
- function setAsideSection(section, kindOnly) {
181
- const parsed = parseSection(section);
182
- if (!parsed)
183
- return section;
184
- const { lines, minus, sides } = parsed;
185
- // Clause 3 — the test is over the sides that name a real file. Requiring BOTH sides to be members
186
- // makes every corpus addition (a-side /dev/null) and deletion (b-side /dev/null) ineligible, which is
187
- // exactly the frames this milestone adds. A rename crossing the boundary in either direction has two
188
- // real sides and one of them is not a member, so it stays whole — `rename from` line included.
189
- const named = sides.filter((p) => p !== null);
190
- if (!named.length || !named.every((p) => CAPTURE_MANIFEST.has(p)))
191
- return section;
192
- // Clause 4 — both halves of a content-identical kind change are left exactly as git wrote them. A
193
- // kind change that DID move bytes is an ordinary content change and is set aside like any other.
194
- if (named.some((p) => kindOnly.has(p)))
195
- return section;
196
- const withheld = lines.slice(minus).join("\n");
197
- // Clause 6 — the claimed size is the UTF-8 BYTE length of what was withheld (the file headers and
198
- // hunks this receipt replaces), never a JavaScript string length: box-drawing frames make the two
199
- // disagree. And the receipt itself is part of the measured artifact, so N set-aside sections can never
200
- // measure as nothing while producing an arbitrarily large reader payload.
201
- const receipt = `set aside: regenerable capture ${named.at(-1)} — ${Buffer.byteLength(withheld, "utf8")} bytes withheld (regenerable frame corpus: asserted byte-for-byte by the corpus tests, never read)`;
202
- // Clause 4 — everything git said happened survives verbatim: old/new mode, new file mode, deleted file
203
- // mode, similarity index, rename from/to, index. Only content is replaced, so a deletion is never
204
- // presented as a file that still exists and an addition is never presented as a modification.
205
- return `${lines.slice(0, minus).join("\n")}\n${receipt}\n`;
206
- }
207
- /** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
208
- export function setAsideRegenerableCaptures(diff) {
209
- if (!diff.includes("diff --git "))
210
- return diff;
211
- const sections = diff.split(/(?=^diff --git )/m);
212
- const kindOnly = kindOnlyPaths(sections);
213
- return sections.map((s) => setAsideSection(s, kindOnly)).join("");
214
- }
215
74
  /**
216
75
  * The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
217
76
  * never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
@@ -232,7 +91,7 @@ const VERSION_FIELD_LINE_RE = /^[+-]\s*"version":\s*"[^"]*",?\s*$/;
232
91
  export async function mirrorsVersionOnly(worktree, baseRef, path) {
233
92
  let diff;
234
93
  try {
235
- diff = await shOk(`git diff -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
94
+ diff = await shOk(`git diff --full-index -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
236
95
  }
237
96
  catch {
238
97
  return false;
@@ -241,9 +100,23 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
241
100
  return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
242
101
  }
243
102
  export async function fetchTaskDiff(worktree, baseRef) {
244
- const full = setAsideRegenerableCaptures(await shOk(`git diff '${baseRef}..HEAD'`, worktree));
245
- const forCap = setAsideRegenerableCaptures(await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree));
246
- return { full, forCap };
103
+ // --full-index: abbreviated index lines vary with object-store density, so two measurements of
104
+ // the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
105
+ const [rawFull, rawForCap] = await Promise.all([
106
+ shOk(`git diff --full-index '${baseRef}..HEAD'`, worktree),
107
+ shOk(`git diff --full-index -U0 '${baseRef}..HEAD'`, worktree),
108
+ ]);
109
+ const fullMeasurement = measureArtifactDiff(rawFull);
110
+ const capMeasurement = measureArtifactDiff(rawForCap);
111
+ return {
112
+ full: fullMeasurement.rendered,
113
+ forCap: capMeasurement.rendered,
114
+ logicBytes: Buffer.byteLength(reviewableLogicDiff(capMeasurement.rendered), "utf8"),
115
+ captureBytes: capMeasurement.captureBytes,
116
+ classifications: capMeasurement.sections,
117
+ fullMeasurement,
118
+ capMeasurement,
119
+ };
247
120
  }
248
121
  export function checkDiffCap(gate, measured, cap, prefix = "") {
249
122
  if (measured <= cap)
@@ -256,8 +129,26 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
256
129
  meta: { park: "human" },
257
130
  };
258
131
  }
132
+ /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
133
+ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
134
+ const logicFail = checkDiffCap(gate, measured.logicBytes, logicCap, prefix);
135
+ if (logicFail)
136
+ return logicFail;
137
+ const captureCap = captureDiffCapFor(logicCap);
138
+ if (measured.captureBytes <= captureCap)
139
+ return null;
140
+ return {
141
+ gate,
142
+ pass: false,
143
+ details: prefix
144
+ + `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
145
+ meta: { park: "human" },
146
+ };
147
+ }
259
148
  export function isDiffCapPark(result) {
260
- return result.pass === false && result.meta?.park === "human" && /diff exceeds verifiable cap/i.test(result.details);
149
+ return result.pass === false
150
+ && result.meta?.park === "human"
151
+ && /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
261
152
  }
262
153
  // ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
263
154
  export function diffCapParkReason(results) {
@@ -374,9 +265,12 @@ artifactDir) {
374
265
  ? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
375
266
  : { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
376
267
  }
377
- const { full: diff, forCap } = await fetchTaskDiff(worktree, baseRef);
268
+ const measuredDiff = await fetchTaskDiff(worktree, baseRef);
269
+ // Keep the reader payload identical to the text charged to the strict cap:
270
+ // whole-file source deletions are represented by their citable operation fact.
271
+ const diff = reviewableLogicDiff(measuredDiff.full);
378
272
  const diffCap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP;
379
- const capFail = checkDiffCap("review", forCap.length, diffCap);
273
+ const capFail = checkTaskDiffCaps("review", measuredDiff, diffCap);
380
274
  if (capFail)
381
275
  return capFail;
382
276
  const nonce = generateVerdictNonce();
@@ -143,12 +143,76 @@ export async function runGates(task, ctx) {
143
143
  heldTest = undefined;
144
144
  await ctx.onGate?.({ phase: "end", gate: "test", result: held });
145
145
  }
146
- return {
147
- results: [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate)),
148
- commits,
149
- };
146
+ const sorted = [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate));
147
+ // v1.87 T5: no round returns a MERGEABLE GREEN on a dirty tree. The battery is not the only gate
148
+ // that executes shell in this worktree — the acceptance gate runs command and named-test oracles
149
+ // (acceptance.ts:264,275) and both verdict gates dispatch a vendor CLI here — so the last word on
150
+ // cleanliness has to be the round's last act rather than the battery's. `results` is what the
151
+ // daemon merges on (daemon.ts `results.every(gateSatisfied)`), so the withdrawal lands there; the
152
+ // journal keeps both the green and its retraction, the honest record of a verdict that did not
153
+ // survive its own round.
154
+ const last = sorted[sorted.length - 1];
155
+ if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
156
+ const dirt = await dirtyWorktree();
157
+ if (dirt) {
158
+ const refusal = dirtyRoundRefusal(last.gate, dirt);
159
+ results[results.indexOf(last)] = refusal;
160
+ sorted[sorted.length - 1] = refusal;
161
+ await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
162
+ }
163
+ }
164
+ return { results: sorted, commits };
150
165
  };
151
166
  const toolGates = ["build", "test", "lint"].filter(enabled);
167
+ /**
168
+ * v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
169
+ * the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
170
+ * build/test/lint and invisible to everything that decides what ships — a green battery on a dirty
171
+ * tree certifies a tree nobody will ever merge, and the committed diff it stands for was never run.
172
+ *
173
+ * That is not gatable-with-a-caveat, so the battery refuses it rather than gating it and hoping.
174
+ * An unreadable `git status` is refused on the same rule: a tree that cannot be proven clean is not
175
+ * proven clean. Returns the dirt (porcelain lines) to name in the refusal, or undefined when clean.
176
+ *
177
+ * The one exemption is tickmarkr's OWN droppings — root-level `.tickmarkr-*` (the adapters' usage
178
+ * record). The harness wrote those, not the worker; they are not work anyone meant to merge, and
179
+ * refusing a tree for the harness's own litter would fail every metered run. Nothing else is
180
+ * exempt: an untracked source file is uncommitted work by every reading git offers.
181
+ */
182
+ const dirtyWorktree = async () => {
183
+ const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
184
+ if (r.code !== 0)
185
+ return `git status failed (exit ${r.code}) — the worktree cannot be proven clean`;
186
+ const entries = r.stdout
187
+ .split("\n")
188
+ .map((l) => l.trimEnd())
189
+ .filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l));
190
+ return entries.length ? entries.join("\n") : undefined;
191
+ };
192
+ const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
193
+ + `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
194
+ + `and never merged (and the committed diff would never be run)`;
195
+ // `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
196
+ const dirtyRefusal = (gate, dirt, left) => ({
197
+ gate,
198
+ pass: false,
199
+ details: DIRTY_WHY
200
+ + (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
201
+ + dirt,
202
+ meta: { dirtyWorktree: true, ...(left ? { dirtiedBy: gate } : {}) },
203
+ });
204
+ // The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
205
+ // the last cleanliness check, and naming a culprit this function cannot identify would be a worse
206
+ // record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
207
+ const dirtyRoundRefusal = (gate, dirt) => ({
208
+ gate,
209
+ pass: false,
210
+ details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
211
+ + `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
212
+ + `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
213
+ + `Uncommitted at round end:\n${dirt}`,
214
+ meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
215
+ });
152
216
  // build/test/lint vs the shared baseline
153
217
  const runBattery = async (commands, selected) => {
154
218
  if (!toolGates.length)
@@ -158,9 +222,15 @@ export async function runGates(task, ctx) {
158
222
  // not at true execution start. They are collectively sub-second (measured), so the debounce
159
223
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
160
224
  const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
225
+ // The same refusal AFTER the commands, because a green command can dirty the tree the check
226
+ // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
227
+ // lands on the last gate that had one — the round dies there either way. A red battery is
228
+ // reported as the red it is: the round already ends, and the command output is the better lead.
229
+ const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
230
+ const blame = dirt ? [...toolGates].reverse().find((g) => commands[g]) : undefined;
161
231
  for (const r of toolResults) {
162
232
  await emitStart(r.gate);
163
- await record(r);
233
+ await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
164
234
  }
165
235
  return;
166
236
  }
@@ -169,6 +239,18 @@ export async function runGates(task, ctx) {
169
239
  for (const g of toolGates) {
170
240
  await emitStart(g);
171
241
  const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
242
+ // The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
243
+ // tracked file makes it dirty again, and every gate after it — including the next shell gate,
244
+ // which would then run against bytes HEAD does not hold — inherits that. So re-check after each
245
+ // command, the last one included, and fail the gate whose command did it. (A red command needs
246
+ // no check: it already ends the round, and its own output is the truer verdict.)
247
+ if (r.pass && commands[g]) {
248
+ const dirt = await dirtyWorktree();
249
+ if (dirt) {
250
+ await record(dirtyRefusal(g, dirt, commands[g]));
251
+ return;
252
+ }
253
+ }
172
254
  if (g === "test" && selected) {
173
255
  const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
174
256
  // green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
@@ -194,7 +276,23 @@ export async function runGates(task, ctx) {
194
276
  commits = e.commits;
195
277
  return { gate: e.gate, pass: e.pass, details: e.details };
196
278
  };
197
- const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, ctx.cfg.scope?.allowDeviations ?? []);
279
+ /**
280
+ * v1.87 T5: the allowlist is read ONCE — at entry, before any gate of this round runs — copied out
281
+ * of `ctx.cfg` and frozen. Both halves are the enforcement: the copy means a later write to the
282
+ * daemon's live config object cannot reach the gate mid-round (the screen at line ~296 and the
283
+ * canonical scope gate below are two separate reads of it), and the freeze means nothing this file
284
+ * hands to `scopeGate` can be widened in flight either. Nothing anywhere writes it back.
285
+ *
286
+ * The boundary, stated rather than implied: this binds the allowlist for the lifetime of a round,
287
+ * which is the largest unit this function owns. It cannot speak for a `tickmarkr resume`, which is
288
+ * a new process whose config the daemon resolves afresh (src/run/daemon.ts) — binding an allowlist
289
+ * across a restart would have to live there, is outside this task's file scope, and is claimed
290
+ * neither here nor in the worker prompt. What the prompt does claim is what holds: a WORKER has no
291
+ * way to change this list, mid-round or otherwise.
292
+ */
293
+ const allowDeviations = [...(ctx.cfg.scope?.allowDeviations ?? [])];
294
+ Object.freeze(allowDeviations);
295
+ const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations);
198
296
  const runGate = async (gate, compute) => {
199
297
  await emitStart(gate);
200
298
  await record(await compute());
@@ -337,6 +435,19 @@ export async function runGates(task, ctx) {
337
435
  }
338
436
  return rv;
339
437
  };
438
+ // v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
439
+ // Guarding only the configured build/test/lint commands left the hole this repairs: the battery is
440
+ // not the only gate that executes shell in this worktree — the acceptance gate runs command and
441
+ // named-test oracles — so a task with NO tool command configured skipped the check entirely and its
442
+ // oracles judged uncommitted state, on a commit whose diff nobody had run. One check at the top, and
443
+ // no shell-executing gate path is reachable on a dirty tree. It lands on the first gate of this
444
+ // round's sequence: the round dies there, exactly as it does on a red command.
445
+ const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
446
+ if (entryDirt) {
447
+ await emitStart(sequence[0]);
448
+ await record(dirtyRefusal(sequence[0], entryDirt));
449
+ return done();
450
+ }
340
451
  if (v185 && await screenBlocks())
341
452
  return done();
342
453
  // A non-final round may run only the tests covering its own diff; the merge-candidate round below
@@ -404,8 +515,14 @@ export async function runGates(task, ctx) {
404
515
  // stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
405
516
  if (selected) {
406
517
  await emitStart("test");
518
+ // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
519
+ // may have run one before it, and every gate between the battery and here reads commits only, so
520
+ // a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
407
521
  const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
408
- const merged = { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
522
+ const dirt = full.pass ? await dirtyWorktree() : undefined;
523
+ const merged = dirt
524
+ ? dirtyRefusal("test", dirt, ctx.commands.test)
525
+ : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
409
526
  results[results.findIndex((r) => r.gate === "test")] = merged;
410
527
  heldTest = undefined;
411
528
  await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
@@ -4,11 +4,13 @@ export declare const GRAPH_ROUTING_MODES: readonly ["partner-led", "risk-based",
4
4
  export declare const STATUSES: readonly ["pending", "running", "gated", "failed", "done", "human"];
5
5
  export declare const GATE_NAMES: readonly ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
6
6
  export declare const TIERS: readonly ["cheap", "mid", "frontier"];
7
+ export declare const SPEC_SOURCES: readonly ["speckit", "gsd", "prd", "native"];
7
8
  export declare const ORACLES: readonly ["command", "test", "judge"];
8
9
  export type Shape = (typeof SHAPES)[number];
9
10
  export type TaskStatus = (typeof STATUSES)[number];
10
11
  export type GateName = (typeof GATE_NAMES)[number];
11
12
  export type Oracle = (typeof ORACLES)[number];
13
+ export type SpecSource = (typeof SPEC_SOURCES)[number];
12
14
  export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
13
15
  oracle: z.ZodLiteral<"command">;
14
16
  command: z.ZodString;
@@ -110,10 +112,10 @@ export declare const RunGraphSchema: z.ZodObject<{
110
112
  gsd: "gsd";
111
113
  prd: "prd";
112
114
  native: "native";
113
- taskmaster: "taskmaster";
114
115
  }>;
115
116
  paths: z.ZodArray<z.ZodString>;
116
117
  hash: z.ZodString;
118
+ base: z.ZodOptional<z.ZodString>;
117
119
  }, z.core.$strip>;
118
120
  tasks: z.ZodArray<z.ZodObject<{
119
121
  id: z.ZodString;
@@ -7,6 +7,7 @@ export const STATUSES = ["pending", "running", "gated", "failed", "done", "human
7
7
  export const GATE_NAMES = ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
8
8
  const MANDATORY_GATES = ["build", "test", "lint", "evidence", "scope"];
9
9
  export const TIERS = ["cheap", "mid", "frontier"];
10
+ export const SPEC_SOURCES = ["speckit", "gsd", "prd", "native"];
10
11
  // v1.19 acceptance oracles: command (exit code), test (named test), judge (LLM, free-text rubric).
11
12
  // A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
12
13
  export const ORACLES = ["command", "test", "judge"];
@@ -98,9 +99,11 @@ export const RunGraphSchema = z
98
99
  // precedence: run flag > this > repo config > global config > default (risk-based).
99
100
  mode: z.enum(GRAPH_ROUTING_MODES).optional(),
100
101
  spec: z.object({
101
- source: z.enum(["speckit", "gsd", "prd", "native", "taskmaster"]),
102
+ source: z.enum(SPEC_SOURCES),
102
103
  paths: z.array(z.string()),
103
104
  hash: z.string(),
105
+ // Q11 compile half: an author-declared ref only. Runtime Git resolution/enforcement is later.
106
+ base: z.string().min(1).optional(),
104
107
  }),
105
108
  tasks: z.array(TaskSchema).min(1),
106
109
  })
@@ -1,6 +1,7 @@
1
1
  import { type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
3
3
  import { type ExecutorDriver } from "../drivers/types.js";
4
+ import type { GateResult } from "../gates/types.js";
4
5
  import { Journal, type JournalEvent } from "./journal.js";
5
6
  export interface RunOptions {
6
7
  runId?: string;
@@ -15,6 +16,7 @@ export interface RunOptions {
15
16
  mode?: RoutingMode;
16
17
  narrate?: (event: JournalEvent) => void;
17
18
  exit?: (code: number) => void;
19
+ supervise?: boolean;
18
20
  }
19
21
  export type ModeSource = "run flag" | "spec" | "repo config" | "global config" | "default";
20
22
  export interface ResolvedRunMode {
@@ -40,8 +42,45 @@ export interface RunSummary {
40
42
  blocked: string[];
41
43
  tipVerify?: "passed" | "failed";
42
44
  lastMergedTask?: string;
45
+ /** T14: did every approval this run accepted actually get enacted, or did the run end over one? */
46
+ approvalDisposition?: "complete" | "outstanding";
47
+ /** the accepted approvals that never reached a dispatch — named, never left to the park buckets */
48
+ outstandingApprovals?: string[];
43
49
  }
50
+ /**
51
+ * T14: approvals the run accepted and never acted on. `approved` above is built ONCE at startup —
52
+ * deliberately, replay determinism depends on it — so an approval written while the daemon is live is
53
+ * inert for that run. Without this the run-end record stated only buckets and tipVerify, both
54
+ * accurate, over a milestone that was silently incomplete: run …230 ended tipVerify "passed" with two
55
+ * upheld approvals and zero subsequent dispatches. Scored per task on its NEWEST approval: a later
56
+ * approval is the live decision, and the events that answer it are the ones after it.
57
+ */
58
+ export declare function outstandingApprovals(events: JournalEvent[]): string[];
44
59
  export declare function formatSummary(s: RunSummary): string;
60
+ /**
61
+ * R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
62
+ * no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
63
+ * the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
64
+ * the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
65
+ * it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
66
+ * verdict from a review that actually ran still fails here, exactly as before.
67
+ *
68
+ * ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
69
+ * was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
70
+ * file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
71
+ * ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
72
+ * verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
73
+ * write below is the seam every OUT-of-file fold shares.
74
+ */
75
+ /**
76
+ * T9: `meta.infra === true` overrides BOTH clauses above. A runner that died on the machine
77
+ * (spawn EAGAIN, OOM) without completing a suite answered nothing about the work, so the honest
78
+ * report of that fact must not double as authorization to merge — and it is the merge predicate,
79
+ * not the gate, that has to say so: classifying the failure into infra metadata while still
80
+ * reporting `pass: true` is exactly how a run that never verified anything gets merged. A declared
81
+ * skip stays satisfied; a gate that ran and passed stays satisfied.
82
+ */
83
+ export declare const gateSatisfied: (g: GateResult) => boolean;
45
84
  /**
46
85
  * T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
47
86
  * review are now launched together, so a round can journal a failed review that the serial walk would
@@ -97,4 +136,7 @@ export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
97
136
  ms: number;
98
137
  resolutionMs: number;
99
138
  } | undefined>;
139
+ /** Test seam — exercise the production observer's total read bound with a small real tree. */
140
+ export declare function setObserveBudgetBytesForTests(bytes: number): void;
141
+ export declare function resetObserveBudgetBytesForTests(): void;
100
142
  export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;