tickmarkr 1.87.0 → 1.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog.d.ts +18 -1
- package/dist/adapters/catalog.js +44 -1
- package/dist/adapters/fake.d.ts +2 -1
- package/dist/adapters/fake.js +7 -0
- package/dist/adapters/grok.js +11 -0
- package/dist/adapters/kimi.d.ts +2 -1
- package/dist/adapters/kimi.js +36 -0
- package/dist/adapters/opencode.js +17 -0
- package/dist/adapters/pi.js +11 -0
- package/dist/adapters/prompt.js +8 -1
- package/dist/adapters/registry.js +76 -57
- package/dist/adapters/types.d.ts +34 -3
- package/dist/adapters/types.js +99 -1
- package/dist/cli/commands/approve.d.ts +2 -0
- package/dist/cli/commands/approve.js +104 -84
- package/dist/cli/commands/compile.d.ts +1 -1
- package/dist/cli/commands/compile.js +29 -12
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/plan.d.ts +1 -1
- package/dist/cli/commands/plan.js +10 -1
- package/dist/cli/commands/report.js +49 -0
- package/dist/cli/commands/status.js +298 -96
- package/dist/cli/harness.d.ts +13 -0
- package/dist/cli/harness.js +50 -0
- package/dist/compile/collateral.js +4 -4
- package/dist/compile/index.d.ts +14 -3
- package/dist/compile/index.js +36 -10
- package/dist/compile/native.js +101 -25
- package/dist/drivers/subprocess.d.ts +6 -1
- package/dist/drivers/subprocess.js +9 -4
- package/dist/gates/acceptance.d.ts +21 -1
- package/dist/gates/acceptance.js +67 -22
- package/dist/gates/artifact-manifest.d.ts +119 -0
- package/dist/gates/artifact-manifest.js +357 -0
- package/dist/gates/baseline.d.ts +6 -0
- package/dist/gates/baseline.js +52 -7
- package/dist/gates/llm.js +37 -26
- package/dist/gates/review.d.ts +16 -11
- package/dist/gates/review.js +44 -150
- package/dist/gates/run-gates.js +124 -7
- package/dist/graph/schema.d.ts +3 -1
- package/dist/graph/schema.js +4 -1
- package/dist/run/daemon.d.ts +42 -0
- package/dist/run/daemon.js +2321 -1967
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +113 -2
- package/dist/run/interactive-seed.d.ts +6 -2
- package/dist/run/interactive-seed.js +72 -5
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +99 -9
- package/dist/run/lock.d.ts +11 -0
- package/dist/run/lock.js +97 -6
- package/dist/run/outcome.d.ts +50 -0
- package/dist/run/outcome.js +152 -0
- package/dist/run/protocol.d.ts +460 -0
- package/dist/run/protocol.js +433 -0
- package/dist/run/supervision.d.ts +29 -0
- package/dist/run/supervision.js +189 -0
- package/fixtures/wrapped-acceptance.native.md +29 -0
- package/package.json +1 -1
- package/schema/rungraph.schema.json +21 -2
- package/skills/tickmarkr-overseer/SKILL.md +257 -5
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
package/dist/gates/review.js
CHANGED
|
@@ -9,6 +9,8 @@ import { redactSecrets } from "../run/redact.js";
|
|
|
9
9
|
import { marginalCostRank } from "../route/router.js";
|
|
10
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
11
11
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
12
|
+
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
13
|
+
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
12
14
|
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
13
15
|
function classifyReviewIssues(approve, issues) {
|
|
14
16
|
const inconsistencies = [];
|
|
@@ -69,149 +71,6 @@ function classifyReviewFindings(findings) {
|
|
|
69
71
|
// OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
|
|
70
72
|
// one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
|
|
71
73
|
const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
|
|
72
|
-
// v1.82 T1 — the cap bounds what a READER MUST READ, not what a run must write. A regeneration of the
|
|
73
|
-
// frame corpora is ~134KB of `-U0` measurement before a source line changes, and nobody reads it: those
|
|
74
|
-
// frames are asserted byte-for-byte by the corpus tests. Counting them is the category error this
|
|
75
|
-
// removes. The two artifacts stay two (OBS-48: the cap measures -U0, the reader receives the full diff);
|
|
76
|
-
// the exclusion is applied to both, identically, right here so BOTH measuring gates inherit it.
|
|
77
|
-
//
|
|
78
|
-
// Clause 1 — membership is an EXACT PATH match against the shipped capture manifest, the same lists the
|
|
79
|
-
// regeneration path itself uses. Location, directory depth and file extension confer nothing: an
|
|
80
|
-
// unmanifested file sitting beside real frames is measured and shown in full. (The anchors deliberately
|
|
81
|
-
// share basenames with the frames; only the full path separates the oracle from its regenerable twin.)
|
|
82
|
-
//
|
|
83
|
-
// The members are LISTED here rather than imported from the manifest module, for one measured reason:
|
|
84
|
-
// that module is the Ink/React renderer, and importing it puts the whole TUI in every gate's module
|
|
85
|
-
// graph — which also memoises chalk's colour level at import time and turns the fleet suite red. A
|
|
86
|
-
// drift test in tests/gates/diff-cap.test.ts asserts this list is exactly GOLDEN_FRAME_CASES +
|
|
87
|
-
// COLOUR_FRAME_CASES and that every entry exists on disk, so it is a copy that cannot drift rather
|
|
88
|
-
// than a second source of truth: add or rename a frame case and that test goes red until this matches.
|
|
89
|
-
export const REGENERABLE_CAPTURE_PATHS = [
|
|
90
|
-
"tests/fixtures/cockpit/frames/run.width-stacked.80x24.txt",
|
|
91
|
-
"tests/fixtures/cockpit/frames/run.width-folded-keys.100x24.txt",
|
|
92
|
-
"tests/fixtures/cockpit/frames/run.width-three-column.140x24.txt",
|
|
93
|
-
"tests/fixtures/cockpit/frames/run.height-14.140x14.txt",
|
|
94
|
-
"tests/fixtures/cockpit/frames/run.height-18.140x18.txt",
|
|
95
|
-
"tests/fixtures/cockpit/frames/run.height-24.140x24.txt",
|
|
96
|
-
"tests/fixtures/cockpit/frames/run.height-40.140x40.txt",
|
|
97
|
-
"tests/fixtures/cockpit/frames/run.no-colour.140x24.txt",
|
|
98
|
-
"tests/fixtures/cockpit/frames/run.non-tty.140x24.txt",
|
|
99
|
-
"tests/fixtures/cockpit/frames/run.ci.140x24.txt",
|
|
100
|
-
"tests/fixtures/cockpit/frames/setup.width-stacked.80x24.txt",
|
|
101
|
-
"tests/fixtures/cockpit/frames/setup.width-folded-keys.100x24.txt",
|
|
102
|
-
"tests/fixtures/cockpit/frames/setup.width-three-column.140x24.txt",
|
|
103
|
-
"tests/fixtures/cockpit/frames/setup.height-14.140x14.txt",
|
|
104
|
-
"tests/fixtures/cockpit/frames/setup.height-18.140x18.txt",
|
|
105
|
-
"tests/fixtures/cockpit/frames/setup.height-24.140x24.txt",
|
|
106
|
-
"tests/fixtures/cockpit/frames/setup.height-40.140x40.txt",
|
|
107
|
-
"tests/fixtures/cockpit/frames/setup.no-colour.140x24.txt",
|
|
108
|
-
"tests/fixtures/cockpit/frames/setup.non-tty.140x24.txt",
|
|
109
|
-
"tests/fixtures/cockpit/frames/setup.ci.140x24.txt",
|
|
110
|
-
"tests/fixtures/cockpit/colour/run-20260718-000943.colour.140x24.txt",
|
|
111
|
-
"tests/fixtures/cockpit/colour/run-20260718-000943.no-colour.140x24.txt",
|
|
112
|
-
"tests/fixtures/cockpit/colour/run-20260725-025004.interrupted.colour.140x24.txt",
|
|
113
|
-
];
|
|
114
|
-
const CAPTURE_MANIFEST = new Set(REGENERABLE_CAPTURE_PATHS);
|
|
115
|
-
// Clause 2 — the frozen appearance anchors and the captured engagement journals are NEVER set aside and
|
|
116
|
-
// are exempt from every other reduction too (clause 5): they are reviewed, immutable evidence. The
|
|
117
|
-
// anchors are the oracle this milestone declares, and the fixture law bans editing a captured journal to
|
|
118
|
-
// satisfy an assertion — so both keep counting toward the cap and keep reaching a reader verbatim.
|
|
119
|
-
export const PROTECTED_EVIDENCE_PREFIXES = [
|
|
120
|
-
"tests/fixtures/cockpit/anchors/",
|
|
121
|
-
"tests/fixtures/cockpit/sources/",
|
|
122
|
-
"tests/fixtures/cockpit/colour/sources/",
|
|
123
|
-
];
|
|
124
|
-
export function isProtectedEvidence(path) {
|
|
125
|
-
return PROTECTED_EVIDENCE_PREFIXES.some((prefix) => path.startsWith(prefix));
|
|
126
|
-
}
|
|
127
|
-
const SET_ASIDE_RECEIPT = /^set aside: regenerable capture (.+?) — \d+ bytes withheld\b/m;
|
|
128
|
-
/** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
|
|
129
|
-
export function setAsideReceiptPath(section) {
|
|
130
|
-
return SET_ASIDE_RECEIPT.exec(section)?.[1] ?? null;
|
|
131
|
-
}
|
|
132
|
-
// `--- a/x` / `+++ b/x` → "x"; "/dev/null" → null, which is the ABSENCE of a side, not a membership
|
|
133
|
-
// failure (clause 3). git quotes paths carrying specials, so unquote before stripping the a/ b/ prefix.
|
|
134
|
-
function diffSidePath(raw) {
|
|
135
|
-
const v = raw.trim();
|
|
136
|
-
if (v === "/dev/null")
|
|
137
|
-
return null;
|
|
138
|
-
const unquoted = v.startsWith('"') && v.endsWith('"') ? v.slice(1, -1) : v;
|
|
139
|
-
return unquoted.replace(/^[ab]\//, "");
|
|
140
|
-
}
|
|
141
|
-
// The `--- a/x` / `+++ b/x` header pair and the hunk lines below it. Clause 4 — null when the section
|
|
142
|
-
// carries NO content hunk at all (a mode-only change, a pure rename, a binary marker): such a section is
|
|
143
|
-
// left exactly as git wrote it rather than handed a manufactured receipt.
|
|
144
|
-
function parseSection(section) {
|
|
145
|
-
const lines = section.split("\n");
|
|
146
|
-
const minus = lines.findIndex((l) => l.startsWith("--- "));
|
|
147
|
-
if (minus === -1 || !lines[minus + 1]?.startsWith("+++ "))
|
|
148
|
-
return null;
|
|
149
|
-
const body = lines.slice(minus + 2);
|
|
150
|
-
if (!body.some((l) => l.startsWith("@@ ")))
|
|
151
|
-
return null;
|
|
152
|
-
return { lines, minus, sides: [diffSidePath(lines[minus].slice(4)), diffSidePath(lines[minus + 1].slice(4))], body };
|
|
153
|
-
}
|
|
154
|
-
// The content a one-sided section carries: its hunk lines with the sign stripped, keeping git's
|
|
155
|
-
// `` markers so a trailing-newline difference still reads as a content
|
|
156
|
-
// difference. Hunk headers are dropped — a delete and an add of the same bytes never share them.
|
|
157
|
-
function hunkPayload(body, sign) {
|
|
158
|
-
return body.filter((l) => l.startsWith(sign) || l.startsWith("\\")).map((l) => (l[0] === sign ? l.slice(1) : l)).join("\n");
|
|
159
|
-
}
|
|
160
|
-
// Clause 4 — a hunk is NOT proof of a content change. Git spells a KIND change (regular file ⇄ symlink)
|
|
161
|
-
// as a delete plus an add of the SAME path, each carrying a hunk, so when the regular file's bytes are
|
|
162
|
-
// exactly the link target both halves carry identical payloads: the kind changed and the content did
|
|
163
|
-
// not. Nothing was withheld, so neither half earns a receipt — and neither half can see the other, so
|
|
164
|
-
// the pairing is found across sections before any one of them is set aside.
|
|
165
|
-
function kindOnlyPaths(sections) {
|
|
166
|
-
const removed = new Map();
|
|
167
|
-
const added = new Map();
|
|
168
|
-
for (const section of sections) {
|
|
169
|
-
const parsed = parseSection(section);
|
|
170
|
-
if (!parsed)
|
|
171
|
-
continue;
|
|
172
|
-
const [a, b] = parsed.sides;
|
|
173
|
-
if (a && !b)
|
|
174
|
-
removed.set(a, hunkPayload(parsed.body, "-"));
|
|
175
|
-
else if (b && !a)
|
|
176
|
-
added.set(b, hunkPayload(parsed.body, "+"));
|
|
177
|
-
}
|
|
178
|
-
return new Set([...removed].filter(([path, payload]) => added.get(path) === payload).map(([path]) => path));
|
|
179
|
-
}
|
|
180
|
-
function setAsideSection(section, kindOnly) {
|
|
181
|
-
const parsed = parseSection(section);
|
|
182
|
-
if (!parsed)
|
|
183
|
-
return section;
|
|
184
|
-
const { lines, minus, sides } = parsed;
|
|
185
|
-
// Clause 3 — the test is over the sides that name a real file. Requiring BOTH sides to be members
|
|
186
|
-
// makes every corpus addition (a-side /dev/null) and deletion (b-side /dev/null) ineligible, which is
|
|
187
|
-
// exactly the frames this milestone adds. A rename crossing the boundary in either direction has two
|
|
188
|
-
// real sides and one of them is not a member, so it stays whole — `rename from` line included.
|
|
189
|
-
const named = sides.filter((p) => p !== null);
|
|
190
|
-
if (!named.length || !named.every((p) => CAPTURE_MANIFEST.has(p)))
|
|
191
|
-
return section;
|
|
192
|
-
// Clause 4 — both halves of a content-identical kind change are left exactly as git wrote them. A
|
|
193
|
-
// kind change that DID move bytes is an ordinary content change and is set aside like any other.
|
|
194
|
-
if (named.some((p) => kindOnly.has(p)))
|
|
195
|
-
return section;
|
|
196
|
-
const withheld = lines.slice(minus).join("\n");
|
|
197
|
-
// Clause 6 — the claimed size is the UTF-8 BYTE length of what was withheld (the file headers and
|
|
198
|
-
// hunks this receipt replaces), never a JavaScript string length: box-drawing frames make the two
|
|
199
|
-
// disagree. And the receipt itself is part of the measured artifact, so N set-aside sections can never
|
|
200
|
-
// measure as nothing while producing an arbitrarily large reader payload.
|
|
201
|
-
const receipt = `set aside: regenerable capture ${named.at(-1)} — ${Buffer.byteLength(withheld, "utf8")} bytes withheld (regenerable frame corpus: asserted byte-for-byte by the corpus tests, never read)`;
|
|
202
|
-
// Clause 4 — everything git said happened survives verbatim: old/new mode, new file mode, deleted file
|
|
203
|
-
// mode, similarity index, rename from/to, index. Only content is replaced, so a deletion is never
|
|
204
|
-
// presented as a file that still exists and an addition is never presented as a modification.
|
|
205
|
-
return `${lines.slice(0, minus).join("\n")}\n${receipt}\n`;
|
|
206
|
-
}
|
|
207
|
-
/** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
|
|
208
|
-
export function setAsideRegenerableCaptures(diff) {
|
|
209
|
-
if (!diff.includes("diff --git "))
|
|
210
|
-
return diff;
|
|
211
|
-
const sections = diff.split(/(?=^diff --git )/m);
|
|
212
|
-
const kindOnly = kindOnlyPaths(sections);
|
|
213
|
-
return sections.map((s) => setAsideSection(s, kindOnly)).join("");
|
|
214
|
-
}
|
|
215
74
|
/**
|
|
216
75
|
* The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
|
|
217
76
|
* never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
|
|
@@ -232,7 +91,7 @@ const VERSION_FIELD_LINE_RE = /^[+-]\s*"version":\s*"[^"]*",?\s*$/;
|
|
|
232
91
|
export async function mirrorsVersionOnly(worktree, baseRef, path) {
|
|
233
92
|
let diff;
|
|
234
93
|
try {
|
|
235
|
-
diff = await shOk(`git diff -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
|
|
94
|
+
diff = await shOk(`git diff --full-index -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
|
|
236
95
|
}
|
|
237
96
|
catch {
|
|
238
97
|
return false;
|
|
@@ -241,9 +100,23 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
|
|
|
241
100
|
return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
|
|
242
101
|
}
|
|
243
102
|
export async function fetchTaskDiff(worktree, baseRef) {
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
103
|
+
// --full-index: abbreviated index lines vary with object-store density, so two measurements of
|
|
104
|
+
// the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
|
|
105
|
+
const [rawFull, rawForCap] = await Promise.all([
|
|
106
|
+
shOk(`git diff --full-index '${baseRef}..HEAD'`, worktree),
|
|
107
|
+
shOk(`git diff --full-index -U0 '${baseRef}..HEAD'`, worktree),
|
|
108
|
+
]);
|
|
109
|
+
const fullMeasurement = measureArtifactDiff(rawFull);
|
|
110
|
+
const capMeasurement = measureArtifactDiff(rawForCap);
|
|
111
|
+
return {
|
|
112
|
+
full: fullMeasurement.rendered,
|
|
113
|
+
forCap: capMeasurement.rendered,
|
|
114
|
+
logicBytes: Buffer.byteLength(reviewableLogicDiff(capMeasurement.rendered), "utf8"),
|
|
115
|
+
captureBytes: capMeasurement.captureBytes,
|
|
116
|
+
classifications: capMeasurement.sections,
|
|
117
|
+
fullMeasurement,
|
|
118
|
+
capMeasurement,
|
|
119
|
+
};
|
|
247
120
|
}
|
|
248
121
|
export function checkDiffCap(gate, measured, cap, prefix = "") {
|
|
249
122
|
if (measured <= cap)
|
|
@@ -256,8 +129,26 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
|
|
|
256
129
|
meta: { park: "human" },
|
|
257
130
|
};
|
|
258
131
|
}
|
|
132
|
+
/** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
|
|
133
|
+
export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
|
|
134
|
+
const logicFail = checkDiffCap(gate, measured.logicBytes, logicCap, prefix);
|
|
135
|
+
if (logicFail)
|
|
136
|
+
return logicFail;
|
|
137
|
+
const captureCap = captureDiffCapFor(logicCap);
|
|
138
|
+
if (measured.captureBytes <= captureCap)
|
|
139
|
+
return null;
|
|
140
|
+
return {
|
|
141
|
+
gate,
|
|
142
|
+
pass: false,
|
|
143
|
+
details: prefix
|
|
144
|
+
+ `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
|
|
145
|
+
meta: { park: "human" },
|
|
146
|
+
};
|
|
147
|
+
}
|
|
259
148
|
export function isDiffCapPark(result) {
|
|
260
|
-
return result.pass === false
|
|
149
|
+
return result.pass === false
|
|
150
|
+
&& result.meta?.park === "human"
|
|
151
|
+
&& /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
|
|
261
152
|
}
|
|
262
153
|
// ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
|
|
263
154
|
export function diffCapParkReason(results) {
|
|
@@ -374,9 +265,12 @@ artifactDir) {
|
|
|
374
265
|
? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
|
|
375
266
|
: { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
|
|
376
267
|
}
|
|
377
|
-
const
|
|
268
|
+
const measuredDiff = await fetchTaskDiff(worktree, baseRef);
|
|
269
|
+
// Keep the reader payload identical to the text charged to the strict cap:
|
|
270
|
+
// whole-file source deletions are represented by their citable operation fact.
|
|
271
|
+
const diff = reviewableLogicDiff(measuredDiff.full);
|
|
378
272
|
const diffCap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP;
|
|
379
|
-
const capFail =
|
|
273
|
+
const capFail = checkTaskDiffCaps("review", measuredDiff, diffCap);
|
|
380
274
|
if (capFail)
|
|
381
275
|
return capFail;
|
|
382
276
|
const nonce = generateVerdictNonce();
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -143,12 +143,76 @@ export async function runGates(task, ctx) {
|
|
|
143
143
|
heldTest = undefined;
|
|
144
144
|
await ctx.onGate?.({ phase: "end", gate: "test", result: held });
|
|
145
145
|
}
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
146
|
+
const sorted = [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate));
|
|
147
|
+
// v1.87 T5: no round returns a MERGEABLE GREEN on a dirty tree. The battery is not the only gate
|
|
148
|
+
// that executes shell in this worktree — the acceptance gate runs command and named-test oracles
|
|
149
|
+
// (acceptance.ts:264,275) and both verdict gates dispatch a vendor CLI here — so the last word on
|
|
150
|
+
// cleanliness has to be the round's last act rather than the battery's. `results` is what the
|
|
151
|
+
// daemon merges on (daemon.ts `results.every(gateSatisfied)`), so the withdrawal lands there; the
|
|
152
|
+
// journal keeps both the green and its retraction, the honest record of a verdict that did not
|
|
153
|
+
// survive its own round.
|
|
154
|
+
const last = sorted[sorted.length - 1];
|
|
155
|
+
if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
|
|
156
|
+
const dirt = await dirtyWorktree();
|
|
157
|
+
if (dirt) {
|
|
158
|
+
const refusal = dirtyRoundRefusal(last.gate, dirt);
|
|
159
|
+
results[results.indexOf(last)] = refusal;
|
|
160
|
+
sorted[sorted.length - 1] = refusal;
|
|
161
|
+
await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return { results: sorted, commits };
|
|
150
165
|
};
|
|
151
166
|
const toolGates = ["build", "test", "lint"].filter(enabled);
|
|
167
|
+
/**
|
|
168
|
+
* v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
|
|
169
|
+
* the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
|
|
170
|
+
* build/test/lint and invisible to everything that decides what ships — a green battery on a dirty
|
|
171
|
+
* tree certifies a tree nobody will ever merge, and the committed diff it stands for was never run.
|
|
172
|
+
*
|
|
173
|
+
* That is not gatable-with-a-caveat, so the battery refuses it rather than gating it and hoping.
|
|
174
|
+
* An unreadable `git status` is refused on the same rule: a tree that cannot be proven clean is not
|
|
175
|
+
* proven clean. Returns the dirt (porcelain lines) to name in the refusal, or undefined when clean.
|
|
176
|
+
*
|
|
177
|
+
* The one exemption is tickmarkr's OWN droppings — root-level `.tickmarkr-*` (the adapters' usage
|
|
178
|
+
* record). The harness wrote those, not the worker; they are not work anyone meant to merge, and
|
|
179
|
+
* refusing a tree for the harness's own litter would fail every metered run. Nothing else is
|
|
180
|
+
* exempt: an untracked source file is uncommitted work by every reading git offers.
|
|
181
|
+
*/
|
|
182
|
+
const dirtyWorktree = async () => {
|
|
183
|
+
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
|
|
184
|
+
if (r.code !== 0)
|
|
185
|
+
return `git status failed (exit ${r.code}) — the worktree cannot be proven clean`;
|
|
186
|
+
const entries = r.stdout
|
|
187
|
+
.split("\n")
|
|
188
|
+
.map((l) => l.trimEnd())
|
|
189
|
+
.filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l));
|
|
190
|
+
return entries.length ? entries.join("\n") : undefined;
|
|
191
|
+
};
|
|
192
|
+
const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
|
|
193
|
+
+ `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
|
|
194
|
+
+ `and never merged (and the committed diff would never be run)`;
|
|
195
|
+
// `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
|
|
196
|
+
const dirtyRefusal = (gate, dirt, left) => ({
|
|
197
|
+
gate,
|
|
198
|
+
pass: false,
|
|
199
|
+
details: DIRTY_WHY
|
|
200
|
+
+ (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
|
|
201
|
+
+ dirt,
|
|
202
|
+
meta: { dirtyWorktree: true, ...(left ? { dirtiedBy: gate } : {}) },
|
|
203
|
+
});
|
|
204
|
+
// The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
|
|
205
|
+
// the last cleanliness check, and naming a culprit this function cannot identify would be a worse
|
|
206
|
+
// record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
|
|
207
|
+
const dirtyRoundRefusal = (gate, dirt) => ({
|
|
208
|
+
gate,
|
|
209
|
+
pass: false,
|
|
210
|
+
details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
|
|
211
|
+
+ `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
|
|
212
|
+
+ `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
|
|
213
|
+
+ `Uncommitted at round end:\n${dirt}`,
|
|
214
|
+
meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
|
|
215
|
+
});
|
|
152
216
|
// build/test/lint vs the shared baseline
|
|
153
217
|
const runBattery = async (commands, selected) => {
|
|
154
218
|
if (!toolGates.length)
|
|
@@ -158,9 +222,15 @@ export async function runGates(task, ctx) {
|
|
|
158
222
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
159
223
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
160
224
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
|
|
225
|
+
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
226
|
+
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
227
|
+
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
228
|
+
// reported as the red it is: the round already ends, and the command output is the better lead.
|
|
229
|
+
const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
|
|
230
|
+
const blame = dirt ? [...toolGates].reverse().find((g) => commands[g]) : undefined;
|
|
161
231
|
for (const r of toolResults) {
|
|
162
232
|
await emitStart(r.gate);
|
|
163
|
-
await record(r);
|
|
233
|
+
await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
|
|
164
234
|
}
|
|
165
235
|
return;
|
|
166
236
|
}
|
|
@@ -169,6 +239,18 @@ export async function runGates(task, ctx) {
|
|
|
169
239
|
for (const g of toolGates) {
|
|
170
240
|
await emitStart(g);
|
|
171
241
|
const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
|
|
242
|
+
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
243
|
+
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
244
|
+
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
245
|
+
// command, the last one included, and fail the gate whose command did it. (A red command needs
|
|
246
|
+
// no check: it already ends the round, and its own output is the truer verdict.)
|
|
247
|
+
if (r.pass && commands[g]) {
|
|
248
|
+
const dirt = await dirtyWorktree();
|
|
249
|
+
if (dirt) {
|
|
250
|
+
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
251
|
+
return;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
172
254
|
if (g === "test" && selected) {
|
|
173
255
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
174
256
|
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
@@ -194,7 +276,23 @@ export async function runGates(task, ctx) {
|
|
|
194
276
|
commits = e.commits;
|
|
195
277
|
return { gate: e.gate, pass: e.pass, details: e.details };
|
|
196
278
|
};
|
|
197
|
-
|
|
279
|
+
/**
|
|
280
|
+
* v1.87 T5: the allowlist is read ONCE — at entry, before any gate of this round runs — copied out
|
|
281
|
+
* of `ctx.cfg` and frozen. Both halves are the enforcement: the copy means a later write to the
|
|
282
|
+
* daemon's live config object cannot reach the gate mid-round (the screen at line ~296 and the
|
|
283
|
+
* canonical scope gate below are two separate reads of it), and the freeze means nothing this file
|
|
284
|
+
* hands to `scopeGate` can be widened in flight either. Nothing anywhere writes it back.
|
|
285
|
+
*
|
|
286
|
+
* The boundary, stated rather than implied: this binds the allowlist for the lifetime of a round,
|
|
287
|
+
* which is the largest unit this function owns. It cannot speak for a `tickmarkr resume`, which is
|
|
288
|
+
* a new process whose config the daemon resolves afresh (src/run/daemon.ts) — binding an allowlist
|
|
289
|
+
* across a restart would have to live there, is outside this task's file scope, and is claimed
|
|
290
|
+
* neither here nor in the worker prompt. What the prompt does claim is what holds: a WORKER has no
|
|
291
|
+
* way to change this list, mid-round or otherwise.
|
|
292
|
+
*/
|
|
293
|
+
const allowDeviations = [...(ctx.cfg.scope?.allowDeviations ?? [])];
|
|
294
|
+
Object.freeze(allowDeviations);
|
|
295
|
+
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations);
|
|
198
296
|
const runGate = async (gate, compute) => {
|
|
199
297
|
await emitStart(gate);
|
|
200
298
|
await record(await compute());
|
|
@@ -337,6 +435,19 @@ export async function runGates(task, ctx) {
|
|
|
337
435
|
}
|
|
338
436
|
return rv;
|
|
339
437
|
};
|
|
438
|
+
// v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
|
|
439
|
+
// Guarding only the configured build/test/lint commands left the hole this repairs: the battery is
|
|
440
|
+
// not the only gate that executes shell in this worktree — the acceptance gate runs command and
|
|
441
|
+
// named-test oracles — so a task with NO tool command configured skipped the check entirely and its
|
|
442
|
+
// oracles judged uncommitted state, on a commit whose diff nobody had run. One check at the top, and
|
|
443
|
+
// no shell-executing gate path is reachable on a dirty tree. It lands on the first gate of this
|
|
444
|
+
// round's sequence: the round dies there, exactly as it does on a red command.
|
|
445
|
+
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
446
|
+
if (entryDirt) {
|
|
447
|
+
await emitStart(sequence[0]);
|
|
448
|
+
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
449
|
+
return done();
|
|
450
|
+
}
|
|
340
451
|
if (v185 && await screenBlocks())
|
|
341
452
|
return done();
|
|
342
453
|
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
@@ -404,8 +515,14 @@ export async function runGates(task, ctx) {
|
|
|
404
515
|
// stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
|
|
405
516
|
if (selected) {
|
|
406
517
|
await emitStart("test");
|
|
518
|
+
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
519
|
+
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
520
|
+
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
407
521
|
const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
|
|
408
|
-
const
|
|
522
|
+
const dirt = full.pass ? await dirtyWorktree() : undefined;
|
|
523
|
+
const merged = dirt
|
|
524
|
+
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
525
|
+
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
|
|
409
526
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
410
527
|
heldTest = undefined;
|
|
411
528
|
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
package/dist/graph/schema.d.ts
CHANGED
|
@@ -4,11 +4,13 @@ export declare const GRAPH_ROUTING_MODES: readonly ["partner-led", "risk-based",
|
|
|
4
4
|
export declare const STATUSES: readonly ["pending", "running", "gated", "failed", "done", "human"];
|
|
5
5
|
export declare const GATE_NAMES: readonly ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
|
|
6
6
|
export declare const TIERS: readonly ["cheap", "mid", "frontier"];
|
|
7
|
+
export declare const SPEC_SOURCES: readonly ["speckit", "gsd", "prd", "native"];
|
|
7
8
|
export declare const ORACLES: readonly ["command", "test", "judge"];
|
|
8
9
|
export type Shape = (typeof SHAPES)[number];
|
|
9
10
|
export type TaskStatus = (typeof STATUSES)[number];
|
|
10
11
|
export type GateName = (typeof GATE_NAMES)[number];
|
|
11
12
|
export type Oracle = (typeof ORACLES)[number];
|
|
13
|
+
export type SpecSource = (typeof SPEC_SOURCES)[number];
|
|
12
14
|
export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
13
15
|
oracle: z.ZodLiteral<"command">;
|
|
14
16
|
command: z.ZodString;
|
|
@@ -110,10 +112,10 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
110
112
|
gsd: "gsd";
|
|
111
113
|
prd: "prd";
|
|
112
114
|
native: "native";
|
|
113
|
-
taskmaster: "taskmaster";
|
|
114
115
|
}>;
|
|
115
116
|
paths: z.ZodArray<z.ZodString>;
|
|
116
117
|
hash: z.ZodString;
|
|
118
|
+
base: z.ZodOptional<z.ZodString>;
|
|
117
119
|
}, z.core.$strip>;
|
|
118
120
|
tasks: z.ZodArray<z.ZodObject<{
|
|
119
121
|
id: z.ZodString;
|
package/dist/graph/schema.js
CHANGED
|
@@ -7,6 +7,7 @@ export const STATUSES = ["pending", "running", "gated", "failed", "done", "human
|
|
|
7
7
|
export const GATE_NAMES = ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
|
|
8
8
|
const MANDATORY_GATES = ["build", "test", "lint", "evidence", "scope"];
|
|
9
9
|
export const TIERS = ["cheap", "mid", "frontier"];
|
|
10
|
+
export const SPEC_SOURCES = ["speckit", "gsd", "prd", "native"];
|
|
10
11
|
// v1.19 acceptance oracles: command (exit code), test (named test), judge (LLM, free-text rubric).
|
|
11
12
|
// A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
|
|
12
13
|
export const ORACLES = ["command", "test", "judge"];
|
|
@@ -98,9 +99,11 @@ export const RunGraphSchema = z
|
|
|
98
99
|
// precedence: run flag > this > repo config > global config > default (risk-based).
|
|
99
100
|
mode: z.enum(GRAPH_ROUTING_MODES).optional(),
|
|
100
101
|
spec: z.object({
|
|
101
|
-
source: z.enum(
|
|
102
|
+
source: z.enum(SPEC_SOURCES),
|
|
102
103
|
paths: z.array(z.string()),
|
|
103
104
|
hash: z.string(),
|
|
105
|
+
// Q11 compile half: an author-declared ref only. Runtime Git resolution/enforcement is later.
|
|
106
|
+
base: z.string().min(1).optional(),
|
|
104
107
|
}),
|
|
105
108
|
tasks: z.array(TaskSchema).min(1),
|
|
106
109
|
})
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import { type ExecutorDriver } from "../drivers/types.js";
|
|
4
|
+
import type { GateResult } from "../gates/types.js";
|
|
4
5
|
import { Journal, type JournalEvent } from "./journal.js";
|
|
5
6
|
export interface RunOptions {
|
|
6
7
|
runId?: string;
|
|
@@ -15,6 +16,7 @@ export interface RunOptions {
|
|
|
15
16
|
mode?: RoutingMode;
|
|
16
17
|
narrate?: (event: JournalEvent) => void;
|
|
17
18
|
exit?: (code: number) => void;
|
|
19
|
+
supervise?: boolean;
|
|
18
20
|
}
|
|
19
21
|
export type ModeSource = "run flag" | "spec" | "repo config" | "global config" | "default";
|
|
20
22
|
export interface ResolvedRunMode {
|
|
@@ -40,8 +42,45 @@ export interface RunSummary {
|
|
|
40
42
|
blocked: string[];
|
|
41
43
|
tipVerify?: "passed" | "failed";
|
|
42
44
|
lastMergedTask?: string;
|
|
45
|
+
/** T14: did every approval this run accepted actually get enacted, or did the run end over one? */
|
|
46
|
+
approvalDisposition?: "complete" | "outstanding";
|
|
47
|
+
/** the accepted approvals that never reached a dispatch — named, never left to the park buckets */
|
|
48
|
+
outstandingApprovals?: string[];
|
|
43
49
|
}
|
|
50
|
+
/**
|
|
51
|
+
* T14: approvals the run accepted and never acted on. `approved` above is built ONCE at startup —
|
|
52
|
+
* deliberately, replay determinism depends on it — so an approval written while the daemon is live is
|
|
53
|
+
* inert for that run. Without this the run-end record stated only buckets and tipVerify, both
|
|
54
|
+
* accurate, over a milestone that was silently incomplete: run …230 ended tipVerify "passed" with two
|
|
55
|
+
* upheld approvals and zero subsequent dispatches. Scored per task on its NEWEST approval: a later
|
|
56
|
+
* approval is the live decision, and the events that answer it are the ones after it.
|
|
57
|
+
*/
|
|
58
|
+
export declare function outstandingApprovals(events: JournalEvent[]): string[];
|
|
44
59
|
export declare function formatSummary(s: RunSummary): string;
|
|
60
|
+
/**
|
|
61
|
+
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
62
|
+
* no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
|
|
63
|
+
* the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
|
|
64
|
+
* the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
|
|
65
|
+
* it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
|
|
66
|
+
* verdict from a review that actually ran still fails here, exactly as before.
|
|
67
|
+
*
|
|
68
|
+
* ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
|
|
69
|
+
* was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
|
|
70
|
+
* file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
|
|
71
|
+
* ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
|
|
72
|
+
* verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
|
|
73
|
+
* write below is the seam every OUT-of-file fold shares.
|
|
74
|
+
*/
|
|
75
|
+
/**
|
|
76
|
+
* T9: `meta.infra === true` overrides BOTH clauses above. A runner that died on the machine
|
|
77
|
+
* (spawn EAGAIN, OOM) without completing a suite answered nothing about the work, so the honest
|
|
78
|
+
* report of that fact must not double as authorization to merge — and it is the merge predicate,
|
|
79
|
+
* not the gate, that has to say so: classifying the failure into infra metadata while still
|
|
80
|
+
* reporting `pass: true` is exactly how a run that never verified anything gets merged. A declared
|
|
81
|
+
* skip stays satisfied; a gate that ran and passed stays satisfied.
|
|
82
|
+
*/
|
|
83
|
+
export declare const gateSatisfied: (g: GateResult) => boolean;
|
|
45
84
|
/**
|
|
46
85
|
* T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
|
|
47
86
|
* review are now launched together, so a round can journal a failed review that the serial walk would
|
|
@@ -97,4 +136,7 @@ export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
|
|
|
97
136
|
ms: number;
|
|
98
137
|
resolutionMs: number;
|
|
99
138
|
} | undefined>;
|
|
139
|
+
/** Test seam — exercise the production observer's total read bound with a small real tree. */
|
|
140
|
+
export declare function setObserveBudgetBytesForTests(bytes: number): void;
|
|
141
|
+
export declare function resetObserveBudgetBytesForTests(): void;
|
|
100
142
|
export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;
|