tickmarkr 1.84.0 → 1.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/adapters/catalog-remote.d.ts +64 -0
- package/dist/adapters/catalog-remote.js +287 -0
- package/dist/adapters/catalog.d.ts +96 -0
- package/dist/adapters/catalog.js +176 -0
- package/dist/adapters/claude-code.d.ts +1 -0
- package/dist/adapters/claude-code.js +59 -1
- package/dist/adapters/fake.js +42 -4
- package/dist/adapters/model-lints.d.ts +25 -5
- package/dist/adapters/model-lints.js +184 -50
- package/dist/adapters/model-windows.d.ts +31 -0
- package/dist/adapters/model-windows.js +69 -0
- package/dist/adapters/prompt.d.ts +5 -1
- package/dist/adapters/prompt.js +13 -4
- package/dist/adapters/registry.d.ts +25 -26
- package/dist/adapters/registry.js +173 -110
- package/dist/adapters/types.d.ts +3 -0
- package/dist/adapters/types.js +36 -3
- package/dist/brand.d.ts +5 -1
- package/dist/brand.js +18 -2
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +43 -21
- package/dist/cli/commands/fleet.d.ts +7 -0
- package/dist/cli/commands/fleet.js +94 -74
- package/dist/cli/commands/init.js +118 -5
- package/dist/cli/commands/status.js +202 -46
- package/dist/compile/collateral.d.ts +86 -2
- package/dist/compile/collateral.js +294 -3
- package/dist/compile/gsd.d.ts +2 -1
- package/dist/compile/gsd.js +68 -2
- package/dist/compile/native.d.ts +14 -0
- package/dist/compile/native.js +161 -12
- package/dist/config/config.d.ts +82 -5
- package/dist/config/config.js +253 -66
- package/dist/config/fleet-overlay.d.ts +25 -20
- package/dist/config/fleet-overlay.js +195 -77
- package/dist/config/fleet-why.d.ts +23 -0
- package/dist/config/fleet-why.js +42 -0
- package/dist/drivers/herdr.d.ts +21 -3
- package/dist/drivers/herdr.js +344 -110
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +1 -0
- package/dist/gates/baseline.js +91 -13
- package/dist/gates/llm.d.ts +0 -1
- package/dist/gates/llm.js +5 -30
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +105 -10
- package/dist/gates/run-gates.d.ts +9 -0
- package/dist/gates/run-gates.js +285 -41
- package/dist/gates/verdict-cause.d.ts +4 -0
- package/dist/gates/verdict-cause.js +63 -0
- package/dist/graph/schema.d.ts +6 -0
- package/dist/graph/schema.js +8 -5
- package/dist/route/router.d.ts +0 -5
- package/dist/route/router.js +16 -20
- package/dist/run/consult.d.ts +6 -0
- package/dist/run/consult.js +35 -25
- package/dist/run/daemon.d.ts +48 -2
- package/dist/run/daemon.js +1488 -330
- package/dist/run/journal.d.ts +56 -3
- package/dist/run/journal.js +358 -4
- package/dist/run/stall.d.ts +35 -1
- package/dist/run/stall.js +118 -8
- package/dist/tui/cockpit/capture.d.ts +12 -0
- package/dist/tui/cockpit/capture.js +37 -1
- package/dist/tui/cockpit/components.d.ts +2 -0
- package/dist/tui/cockpit/components.js +8 -8
- package/dist/tui/cockpit/derive.d.ts +29 -2
- package/dist/tui/cockpit/derive.js +219 -23
- package/dist/tui/cockpit/run-cockpit.js +128 -27
- package/dist/tui/cockpit/theme.d.ts +32 -26
- package/dist/tui/cockpit/theme.js +11 -5
- package/dist/tui/ink/components.d.ts +0 -15
- package/dist/tui/ink/components.js +0 -17
- package/dist/tui/ink/fleet-app.d.ts +4 -1
- package/dist/tui/ink/fleet-app.js +134 -13
- package/fixtures/sample.native.md +1 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +354 -34
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
- package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
- package/dist/tui/ink/studio-app.d.ts +0 -59
- package/dist/tui/ink/studio-app.js +0 -320
- package/dist/tui/save.d.ts +0 -38
- package/dist/tui/save.js +0 -96
- package/dist/tui/staging.d.ts +0 -29
- package/dist/tui/staging.js +0 -78
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { posix } from "node:path";
|
|
3
|
+
import { channelKey, shq } from "../adapters/types.js";
|
|
2
4
|
import { TIER_RANK } from "../config/config.js";
|
|
3
5
|
import { getAdapter } from "../adapters/registry.js";
|
|
4
6
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
@@ -9,13 +11,118 @@ import { captureLlmOutput } from "./llm.js";
|
|
|
9
11
|
import { marginalCostRank } from "../route/router.js";
|
|
10
12
|
import { reviewGate } from "./review.js";
|
|
11
13
|
import { scopeGate } from "./scope.js";
|
|
14
|
+
import { shGit } from "../run/git.js";
|
|
12
15
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
16
|
+
const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
|
|
17
|
+
// relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
|
|
18
|
+
const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
|
|
19
|
+
const SELECTION_FILE_CAP = 3000;
|
|
20
|
+
/**
|
|
21
|
+
* T4: the tests covering this round's diff, or undefined when the diff cannot be attributed with
|
|
22
|
+
* certainty — a rename or delete (the old path's coverage is gone), or a changed file no test
|
|
23
|
+
* reaches. Coverage: a test file covers itself; a test covers every file reachable from it through
|
|
24
|
+
* relative imports, directly or transitively.
|
|
25
|
+
*
|
|
26
|
+
* ponytail: ceiling — relative specifiers only (no tsconfig paths, no bare aliases, no computed
|
|
27
|
+
* specifiers), and an import cycle contributes only what it had resolved when re-entered. So this
|
|
28
|
+
* CAN miss. The miss is bounded by construction, not by care: the merge-candidate round re-runs the
|
|
29
|
+
* full suite on the same commit, so a miss costs one round and can never merge. Teach it a resolver
|
|
30
|
+
* (tsconfig paths, package exports) if selection ever misses often enough to be worth a round.
|
|
31
|
+
*/
|
|
32
|
+
async function coveringTests(worktree, baseRef) {
|
|
33
|
+
const diff = await shGit(`git diff --name-status ${shq(baseRef)} HEAD`, worktree);
|
|
34
|
+
if (diff.code !== 0)
|
|
35
|
+
return undefined;
|
|
36
|
+
const changed = [];
|
|
37
|
+
for (const line of diff.stdout.split("\n")) {
|
|
38
|
+
if (!line.trim())
|
|
39
|
+
continue;
|
|
40
|
+
const parts = line.split("\t");
|
|
41
|
+
const status = parts[0] ?? "";
|
|
42
|
+
// R (rename) and D (delete): whatever used to cover the old path is unattributable now — full suite.
|
|
43
|
+
if (!status || status[0] === "R" || status[0] === "D" || parts.length < 2)
|
|
44
|
+
return undefined;
|
|
45
|
+
changed.push(parts[parts.length - 1]);
|
|
46
|
+
}
|
|
47
|
+
if (!changed.length)
|
|
48
|
+
return undefined;
|
|
49
|
+
const listed = await shGit("git ls-files", worktree);
|
|
50
|
+
if (listed.code !== 0)
|
|
51
|
+
return undefined;
|
|
52
|
+
const tracked = listed.stdout.split("\n").filter(Boolean);
|
|
53
|
+
if (tracked.length > SELECTION_FILE_CAP)
|
|
54
|
+
return undefined; // ponytail: a huge repo pays the full suite rather than a long scan
|
|
55
|
+
const trackedSet = new Set(tracked);
|
|
56
|
+
const tests = tracked.filter((p) => TEST_FILE_RE.test(p));
|
|
57
|
+
if (!tests.length)
|
|
58
|
+
return undefined;
|
|
59
|
+
const resolveSpec = (from, spec) => {
|
|
60
|
+
const base = posix.join(posix.dirname(from), spec);
|
|
61
|
+
// ESM-TS writes ".js" for a ".ts" source; a directory specifier means its index.
|
|
62
|
+
const candidates = [base, base.replace(/\.js$/, ".ts"), base.replace(/\.jsx$/, ".tsx"),
|
|
63
|
+
`${base}.ts`, `${base}.tsx`, `${base}.js`, `${base}/index.ts`, `${base}/index.js`];
|
|
64
|
+
return candidates.find((c) => trackedSet.has(c));
|
|
65
|
+
};
|
|
66
|
+
const reachCache = new Map();
|
|
67
|
+
const reachOf = (file) => {
|
|
68
|
+
const cached = reachCache.get(file);
|
|
69
|
+
if (cached)
|
|
70
|
+
return cached;
|
|
71
|
+
const out = new Set();
|
|
72
|
+
reachCache.set(file, out); // cycle guard: a re-entered file contributes what it has so far
|
|
73
|
+
let src;
|
|
74
|
+
try {
|
|
75
|
+
src = readFileSync(posix.join(worktree, file), "utf8");
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
return out;
|
|
79
|
+
}
|
|
80
|
+
for (const m of src.matchAll(IMPORT_RE)) {
|
|
81
|
+
const dep = m[1] ? resolveSpec(file, m[1]) : undefined;
|
|
82
|
+
if (!dep || out.has(dep))
|
|
83
|
+
continue;
|
|
84
|
+
out.add(dep);
|
|
85
|
+
for (const t of reachOf(dep))
|
|
86
|
+
out.add(t);
|
|
87
|
+
}
|
|
88
|
+
return out;
|
|
89
|
+
};
|
|
90
|
+
const selected = new Set();
|
|
91
|
+
for (const file of changed) {
|
|
92
|
+
if (TEST_FILE_RE.test(file)) {
|
|
93
|
+
selected.add(file);
|
|
94
|
+
continue;
|
|
95
|
+
}
|
|
96
|
+
const covering = tests.filter((t) => reachOf(t).has(file));
|
|
97
|
+
if (!covering.length)
|
|
98
|
+
return undefined; // nothing covers this file — only the full suite can speak for it
|
|
99
|
+
for (const t of covering)
|
|
100
|
+
selected.add(t);
|
|
101
|
+
}
|
|
102
|
+
return [...selected].sort();
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
|
|
106
|
+
* npm/yarn/pnpm/npx script wrappers need one `--` to forward positional filters to the underlying runner;
|
|
107
|
+
* a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
|
|
108
|
+
*/
|
|
109
|
+
export function testCommandForFiles(testCmd, files) {
|
|
110
|
+
const wrapped = /^\s*(?:npm|yarn|pnpm|npx)\b/.test(testCmd);
|
|
111
|
+
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
112
|
+
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
113
|
+
}
|
|
13
114
|
export async function runGates(task, ctx) {
|
|
14
115
|
const results = [];
|
|
15
116
|
let commits = [];
|
|
16
117
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
17
118
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
18
119
|
const failed = () => results.some((r) => !r.pass);
|
|
120
|
+
const v185 = ctx.pipeline === "v185";
|
|
121
|
+
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
122
|
+
// round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
|
|
123
|
+
// exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
|
|
124
|
+
// (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
|
|
125
|
+
let heldTest;
|
|
19
126
|
const sequence = GATE_NAMES.filter((g) => enabled(g));
|
|
20
127
|
const total = sequence.length;
|
|
21
128
|
const indexOf = (gate) => sequence.indexOf(gate) + 1;
|
|
@@ -23,42 +130,109 @@ export async function runGates(task, ctx) {
|
|
|
23
130
|
results.push(result);
|
|
24
131
|
await ctx.onGate?.({ phase: "end", gate: result.gate, result });
|
|
25
132
|
};
|
|
26
|
-
const emitStart = async (gate) => {
|
|
27
|
-
await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total });
|
|
133
|
+
const emitStart = async (gate, parentAt) => {
|
|
134
|
+
await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total, ...(parentAt === undefined ? {} : { parentAt }) });
|
|
135
|
+
};
|
|
136
|
+
// The returned array stays in GATE_NAMES order however few gates a short-circuiting round reached.
|
|
137
|
+
// A round that ends before its merge-candidate stage flushes the held screen on the way out, so a
|
|
138
|
+
// green subset run is still journaled exactly once — as a selected run, which is what it was.
|
|
139
|
+
const done = async () => {
|
|
140
|
+
if (heldTest) {
|
|
141
|
+
const held = heldTest;
|
|
142
|
+
heldTest = undefined;
|
|
143
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: held });
|
|
144
|
+
}
|
|
145
|
+
return {
|
|
146
|
+
results: [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate)),
|
|
147
|
+
commits,
|
|
148
|
+
};
|
|
28
149
|
};
|
|
29
|
-
// 1. build/test/lint vs shared baseline — deterministic and cheap, first
|
|
30
150
|
const toolGates = ["build", "test", "lint"].filter(enabled);
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
151
|
+
// build/test/lint vs the shared baseline
|
|
152
|
+
const runBattery = async (commands, selected) => {
|
|
153
|
+
if (!toolGates.length)
|
|
154
|
+
return;
|
|
155
|
+
if (!v185) {
|
|
156
|
+
// ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
|
|
157
|
+
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
158
|
+
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
159
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
|
|
160
|
+
for (const r of toolResults) {
|
|
161
|
+
await emitStart(r.gate);
|
|
162
|
+
await record(r);
|
|
163
|
+
}
|
|
164
|
+
return;
|
|
39
165
|
}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
166
|
+
// T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
|
|
167
|
+
// the full vitest suite before anyone reads its verdict.
|
|
168
|
+
for (const g of toolGates) {
|
|
169
|
+
await emitStart(g);
|
|
170
|
+
const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
|
|
171
|
+
if (g === "test" && selected) {
|
|
172
|
+
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
173
|
+
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
174
|
+
if (!screened.pass)
|
|
175
|
+
await record(screened);
|
|
176
|
+
else {
|
|
177
|
+
heldTest = screened;
|
|
178
|
+
results.push(screened);
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
else {
|
|
182
|
+
await record(r);
|
|
183
|
+
}
|
|
184
|
+
if (failed())
|
|
185
|
+
return;
|
|
186
|
+
}
|
|
187
|
+
};
|
|
188
|
+
// The two sub-second git checks, as pure verdicts: no journal, no results push. Both read committed
|
|
189
|
+
// state only (commits ahead of base, `git diff --name-only base..HEAD`), so neither can be moved by
|
|
190
|
+
// anything the battery does to the worktree — which is what lets the screen below trust them early.
|
|
191
|
+
const evidenceResult = async () => {
|
|
46
192
|
const e = await evidenceGate(ctx.worktree, ctx.baseRef);
|
|
47
193
|
commits = e.commits;
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
194
|
+
return { gate: e.gate, pass: e.pass, details: e.details };
|
|
195
|
+
};
|
|
196
|
+
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, ctx.cfg.scope?.allowDeviations ?? []);
|
|
197
|
+
const runGate = async (gate, compute) => {
|
|
198
|
+
await emitStart(gate);
|
|
199
|
+
await record(await compute());
|
|
200
|
+
};
|
|
201
|
+
/**
|
|
202
|
+
* T4 (OBS-265): the deterministic git checks run BEFORE the battery, as a screen — they answer
|
|
203
|
+
* "is this diff worth starting a ~3.7m tool battery for?" before the first command runs. A red
|
|
204
|
+
* screen IS the round's verdict: what it produced is journaled (in the order it ran) and the round
|
|
205
|
+
* ends there, so a drive-by out-of-scope edit costs <1s instead of the whole battery.
|
|
206
|
+
*
|
|
207
|
+
* A green screen changes nothing downstream. The recorded sequence stays GATE_NAMES order, so the
|
|
208
|
+
* journal, `tickmarkr report`, the surfaces, and resume's GATE_NAMES walk over already-satisfied
|
|
209
|
+
* gates all keep reading exactly one order.
|
|
210
|
+
*
|
|
211
|
+
* ponytail: the price of that is re-reading two git checks (~40ms) in their canonical positions
|
|
212
|
+
* rather than teaching every consumer of the gate stream a second order. Both reads see the same
|
|
213
|
+
* commits — the battery never moves HEAD — so the screen cannot disagree with the gate it screens
|
|
214
|
+
* for. Charge it only when there IS a battery command to protect.
|
|
215
|
+
*/
|
|
216
|
+
const screenBlocks = async () => {
|
|
217
|
+
if (!toolGates.some((g) => ctx.commands[g]))
|
|
218
|
+
return false;
|
|
219
|
+
const screened = [];
|
|
220
|
+
for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
|
|
221
|
+
if (!enabled(gate))
|
|
222
|
+
continue;
|
|
223
|
+
screened.push(await compute());
|
|
224
|
+
if (screened[screened.length - 1].pass)
|
|
225
|
+
continue;
|
|
226
|
+
for (const r of screened) {
|
|
227
|
+
await emitStart(r.gate);
|
|
228
|
+
await record(r);
|
|
229
|
+
}
|
|
230
|
+
return true;
|
|
231
|
+
}
|
|
232
|
+
return false;
|
|
233
|
+
};
|
|
234
|
+
// acceptance judge — LLM spend, so everything deterministic has already passed when this runs
|
|
235
|
+
const runAcceptance = async () => {
|
|
62
236
|
const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
|
|
63
237
|
const jvia = ctx.via
|
|
64
238
|
? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", judgeAdapter.id), label: ctx.via.labelFor("judge") }
|
|
@@ -121,13 +295,10 @@ export async function runGates(task, ctx) {
|
|
|
121
295
|
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
122
296
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
123
297
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
// 5. cross-vendor review
|
|
129
|
-
if (enabled("review")) {
|
|
130
|
-
await emitStart("review");
|
|
298
|
+
return { result: a, invocations };
|
|
299
|
+
};
|
|
300
|
+
// cross-vendor review
|
|
301
|
+
const runReview = async () => {
|
|
131
302
|
let rv = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir);
|
|
132
303
|
// OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
|
|
133
304
|
// never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
|
|
@@ -145,7 +316,80 @@ export async function runGates(task, ctx) {
|
|
|
145
316
|
rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
|
|
146
317
|
}
|
|
147
318
|
}
|
|
148
|
-
|
|
319
|
+
return rv;
|
|
320
|
+
};
|
|
321
|
+
if (v185 && await screenBlocks())
|
|
322
|
+
return done();
|
|
323
|
+
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
324
|
+
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
325
|
+
const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
|
|
326
|
+
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
327
|
+
: undefined;
|
|
328
|
+
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected);
|
|
329
|
+
if (failed())
|
|
330
|
+
return done();
|
|
331
|
+
if (enabled("evidence")) {
|
|
332
|
+
await runGate("evidence", evidenceResult);
|
|
333
|
+
if (failed())
|
|
334
|
+
return done();
|
|
335
|
+
}
|
|
336
|
+
if (enabled("scope")) {
|
|
337
|
+
await runGate("scope", scopeResult);
|
|
338
|
+
if (failed())
|
|
339
|
+
return done();
|
|
340
|
+
}
|
|
341
|
+
if (v185 && (enabled("acceptance") || enabled("review"))) {
|
|
342
|
+
// Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
|
|
343
|
+
// unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
|
|
344
|
+
// verdict: each gets the same commit and the same brief it always got, and neither promise is
|
|
345
|
+
// reachable from inside the other. Only the waiting is gone.
|
|
346
|
+
const parentAt = Date.now();
|
|
347
|
+
// Both starts are emitted before either gate is launched, so the stream's order is the round's
|
|
348
|
+
// order and not a race between two dispatches. BOTH ARE IN FLIGHT BEFORE EITHER IS AWAITED:
|
|
349
|
+
// whichever adapter is slower no longer decides when the other one runs.
|
|
350
|
+
if (enabled("acceptance"))
|
|
351
|
+
await emitStart("acceptance", parentAt);
|
|
352
|
+
if (enabled("review"))
|
|
353
|
+
await emitStart("review", parentAt);
|
|
354
|
+
const judging = enabled("acceptance") ? runAcceptance() : undefined;
|
|
355
|
+
const reviewing = enabled("review") ? runReview() : undefined;
|
|
356
|
+
// Attach BOTH publication handlers before awaiting either. Dispatch concurrency alone is not
|
|
357
|
+
// enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
|
|
358
|
+
// process death can lose that already-earned verdict. The returned result is still sorted into
|
|
359
|
+
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
360
|
+
const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
|
|
361
|
+
const reviewed = reviewing?.then((outcome) => record(outcome));
|
|
362
|
+
await Promise.all([judged, reviewed]);
|
|
363
|
+
if (failed())
|
|
364
|
+
return done();
|
|
365
|
+
}
|
|
366
|
+
else if (!v185) {
|
|
367
|
+
// Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
|
|
368
|
+
if (enabled("acceptance")) {
|
|
369
|
+
await emitStart("acceptance");
|
|
370
|
+
const judged = await runAcceptance();
|
|
371
|
+
await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
|
|
372
|
+
if (failed())
|
|
373
|
+
return done();
|
|
374
|
+
}
|
|
375
|
+
if (enabled("review")) {
|
|
376
|
+
await emitStart("review");
|
|
377
|
+
await record(await runReview());
|
|
378
|
+
}
|
|
379
|
+
return done();
|
|
380
|
+
}
|
|
381
|
+
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
382
|
+
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
383
|
+
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
|
|
384
|
+
// held screen rather than joining it: one `test` entry in the record, one `test` end event in the
|
|
385
|
+
// stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
|
|
386
|
+
if (selected) {
|
|
387
|
+
await emitStart("test");
|
|
388
|
+
const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
|
|
389
|
+
const merged = { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
|
|
390
|
+
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
391
|
+
heldTest = undefined;
|
|
392
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
|
149
393
|
}
|
|
150
|
-
return
|
|
394
|
+
return done();
|
|
151
395
|
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
export type VerdictDiscriminator = "approve" | "pass" | "ok" | "action";
|
|
2
|
+
export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
|
|
3
|
+
export declare function hasVerdictParticipationWitness(raw: string, nonce: string, discriminator: VerdictDiscriminator): boolean;
|
|
4
|
+
export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator): VerdictUnparseableCause;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
const MAX_WITNESS_BYTES = 4096;
|
|
2
|
+
const JSON_STRING_VALUE = String.raw `"(?:\\(?:["\\/bfnrt]|u[0-9a-fA-F]{4})|[^"\\\u0000-\u001F])*"`;
|
|
3
|
+
function escapeRegex(value) {
|
|
4
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
5
|
+
}
|
|
6
|
+
// Pane renderers hard-wrap at physical columns and prefix continuation lines with whitespace or box
|
|
7
|
+
// chrome. Joining those renderer lines is deliberately narrower than parsing: it restores split keys,
|
|
8
|
+
// values and delimiters, but it does not require the response to be complete or valid JSON.
|
|
9
|
+
function joinRendererLines(raw) {
|
|
10
|
+
return raw.replace(/\r\n?/g, "\n").split("\n")
|
|
11
|
+
.map((line) => line.replace(/^[\t │|]+/, "").replace(/[\t │|]+$/, ""))
|
|
12
|
+
.join("");
|
|
13
|
+
}
|
|
14
|
+
// Yield only the first structural prefix of each object candidate. Stopping at the next unquoted
|
|
15
|
+
// opening brace prevents a nonce from one object binding a discriminator from another; the prompts
|
|
16
|
+
// put their discriminator before any nested object, so no valid boundary shape is lost.
|
|
17
|
+
function objectPrefixes(raw) {
|
|
18
|
+
const joined = joinRendererLines(raw);
|
|
19
|
+
const prefixes = [];
|
|
20
|
+
for (let open = joined.indexOf("{"); open !== -1; open = joined.indexOf("{", open + 1)) {
|
|
21
|
+
const ceiling = Math.min(joined.length, open + MAX_WITNESS_BYTES);
|
|
22
|
+
let end = ceiling;
|
|
23
|
+
let quoted = false;
|
|
24
|
+
let escaped = false;
|
|
25
|
+
for (let i = open + 1; i < ceiling; i++) {
|
|
26
|
+
const char = joined[i];
|
|
27
|
+
if (quoted) {
|
|
28
|
+
if (escaped)
|
|
29
|
+
escaped = false;
|
|
30
|
+
else if (char === "\\")
|
|
31
|
+
escaped = true;
|
|
32
|
+
else if (char === '"')
|
|
33
|
+
quoted = false;
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
36
|
+
if (char === '"')
|
|
37
|
+
quoted = true;
|
|
38
|
+
else if (char === "{") {
|
|
39
|
+
end = i;
|
|
40
|
+
break;
|
|
41
|
+
}
|
|
42
|
+
else if (char === "}") {
|
|
43
|
+
end = i + 1;
|
|
44
|
+
break;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
prefixes.push(joined.slice(open, end));
|
|
48
|
+
}
|
|
49
|
+
return prefixes;
|
|
50
|
+
}
|
|
51
|
+
export function hasVerdictParticipationWitness(raw, nonce, discriminator) {
|
|
52
|
+
const noncePattern = new RegExp(String.raw `"nonce"\s*:\s*${escapeRegex(JSON.stringify(nonce))}\s*[,]`);
|
|
53
|
+
const valuePattern = discriminator === "action" ? JSON_STRING_VALUE : "(?:true|false)";
|
|
54
|
+
const discriminatorPattern = new RegExp(String.raw `"${discriminator}"\s*:\s*${valuePattern}\s*[,}]`);
|
|
55
|
+
return objectPrefixes(raw).some((prefix) => noncePattern.test(prefix) && discriminatorPattern.test(prefix));
|
|
56
|
+
}
|
|
57
|
+
export function classifyVerdictCause(raw, nonce, discriminator) {
|
|
58
|
+
if (raw.trim().length === 0)
|
|
59
|
+
return "empty-output";
|
|
60
|
+
return hasVerdictParticipationWitness(raw, nonce, discriminator)
|
|
61
|
+
? "malformed-verdict"
|
|
62
|
+
: "no-verdict";
|
|
63
|
+
}
|
package/dist/graph/schema.d.ts
CHANGED
|
@@ -12,9 +12,11 @@ export type Oracle = (typeof ORACLES)[number];
|
|
|
12
12
|
export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
13
13
|
oracle: z.ZodLiteral<"command">;
|
|
14
14
|
command: z.ZodString;
|
|
15
|
+
text: z.ZodOptional<z.ZodString>;
|
|
15
16
|
}, z.core.$strip>, z.ZodObject<{
|
|
16
17
|
oracle: z.ZodLiteral<"test">;
|
|
17
18
|
test: z.ZodString;
|
|
19
|
+
text: z.ZodOptional<z.ZodString>;
|
|
18
20
|
}, z.core.$strip>, z.ZodObject<{
|
|
19
21
|
oracle: z.ZodLiteral<"judge">;
|
|
20
22
|
text: z.ZodString;
|
|
@@ -43,9 +45,11 @@ export declare const TaskSchema: z.ZodObject<{
|
|
|
43
45
|
acceptance: z.ZodArray<z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
44
46
|
oracle: z.ZodLiteral<"command">;
|
|
45
47
|
command: z.ZodString;
|
|
48
|
+
text: z.ZodOptional<z.ZodString>;
|
|
46
49
|
}, z.core.$strip>, z.ZodObject<{
|
|
47
50
|
oracle: z.ZodLiteral<"test">;
|
|
48
51
|
test: z.ZodString;
|
|
52
|
+
text: z.ZodOptional<z.ZodString>;
|
|
49
53
|
}, z.core.$strip>, z.ZodObject<{
|
|
50
54
|
oracle: z.ZodLiteral<"judge">;
|
|
51
55
|
text: z.ZodString;
|
|
@@ -133,9 +137,11 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
133
137
|
acceptance: z.ZodArray<z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
134
138
|
oracle: z.ZodLiteral<"command">;
|
|
135
139
|
command: z.ZodString;
|
|
140
|
+
text: z.ZodOptional<z.ZodString>;
|
|
136
141
|
}, z.core.$strip>, z.ZodObject<{
|
|
137
142
|
oracle: z.ZodLiteral<"test">;
|
|
138
143
|
test: z.ZodString;
|
|
144
|
+
text: z.ZodOptional<z.ZodString>;
|
|
139
145
|
}, z.core.$strip>, z.ZodObject<{
|
|
140
146
|
oracle: z.ZodLiteral<"judge">;
|
|
141
147
|
text: z.ZodString;
|
package/dist/graph/schema.js
CHANGED
|
@@ -11,11 +11,14 @@ export const TIERS = ["cheap", "mid", "frontier"];
|
|
|
11
11
|
// A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
|
|
12
12
|
export const ORACLES = ["command", "test", "judge"];
|
|
13
13
|
// Typed acceptance oracle: command carries the thing to run, test the test name, judge free text.
|
|
14
|
-
//
|
|
14
|
+
// text is OPTIONAL non-empty on command/test (declared prose beside the oracle; judge requires it).
|
|
15
|
+
// Declaring it matters: z.object strips unknown keys, and loadGraph revalidates on every read —
|
|
16
|
+
// an undeclared text would be silently discarded at the next load. Anything else (a typed object
|
|
17
|
+
// naming an unknown oracle, or a text that is not a non-empty string) fails validation loudly here.
|
|
15
18
|
export const AcceptanceItemSchema = z.union([
|
|
16
19
|
z.string().min(1),
|
|
17
|
-
z.object({ oracle: z.literal("command"), command: z.string().min(1) }),
|
|
18
|
-
z.object({ oracle: z.literal("test"), test: z.string().min(1) }),
|
|
20
|
+
z.object({ oracle: z.literal("command"), command: z.string().min(1), text: z.string().min(1).optional() }),
|
|
21
|
+
z.object({ oracle: z.literal("test"), test: z.string().min(1), text: z.string().min(1).optional() }),
|
|
19
22
|
z.object({ oracle: z.literal("judge"), text: z.string().min(1) }),
|
|
20
23
|
]);
|
|
21
24
|
// Shared text rendering of one acceptance item — every consumer (worker prompt, acceptance gate,
|
|
@@ -24,9 +27,9 @@ export function renderAcceptanceItem(item) {
|
|
|
24
27
|
if (typeof item === "string")
|
|
25
28
|
return item;
|
|
26
29
|
if (item.oracle === "command")
|
|
27
|
-
return `$ ${item.command}`;
|
|
30
|
+
return item.text ?? `$ ${item.command}`;
|
|
28
31
|
if (item.oracle === "test")
|
|
29
|
-
return `test: ${item.test}`;
|
|
32
|
+
return item.text ?? `test: ${item.test}`;
|
|
30
33
|
return item.text; // judge — bare text, byte-identical to a plain-string judge criterion
|
|
31
34
|
}
|
|
32
35
|
export const TaskSchema = z.object({
|
package/dist/route/router.d.ts
CHANGED
|
@@ -19,11 +19,6 @@ export interface Route {
|
|
|
19
19
|
deviation?: RouteDeviation;
|
|
20
20
|
}
|
|
21
21
|
export interface RoutingPreferContext {
|
|
22
|
-
autoPrefer?: {
|
|
23
|
-
derivedAt: string;
|
|
24
|
-
[shape: string]: string[] | string;
|
|
25
|
-
};
|
|
26
|
-
doctorFresh: boolean;
|
|
27
22
|
overlayPreferShapes: ReadonlySet<string>;
|
|
28
23
|
}
|
|
29
24
|
export interface ExploreContext {
|
package/dist/route/router.js
CHANGED
|
@@ -28,19 +28,9 @@ const exploreOff = (task, cfg, exploreCtx) => {
|
|
|
28
28
|
return true;
|
|
29
29
|
return false;
|
|
30
30
|
};
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
};
|
|
35
|
-
const preferFromAuto = (shape, preferCtx) => !!preferCtx?.doctorFresh && !!preferCtx.autoPrefer && !preferCtx.overlayPreferShapes.has(shape) &&
|
|
36
|
-
autoPreferList(preferCtx.autoPrefer, shape) !== undefined;
|
|
37
|
-
const effectivePrefer = (shape, entry, preferCtx) => {
|
|
38
|
-
if (preferCtx?.overlayPreferShapes.has(shape))
|
|
39
|
-
return entry?.prefer;
|
|
40
|
-
if (preferCtx?.doctorFresh && preferCtx.autoPrefer)
|
|
41
|
-
return autoPreferList(preferCtx.autoPrefer, shape) ?? entry?.prefer;
|
|
42
|
-
return entry?.prefer;
|
|
43
|
-
};
|
|
31
|
+
// v1.86 T3: autoPrefer is deleted — prefer is operator-declared only. Nothing is ever derived
|
|
32
|
+
// from probe health, tier data, or a machine-built shape table.
|
|
33
|
+
const effectivePrefer = (entry) => entry?.prefer;
|
|
44
34
|
export class RoutingError extends Error {
|
|
45
35
|
constructor(msg) {
|
|
46
36
|
super(msg);
|
|
@@ -134,7 +124,7 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
134
124
|
// ponytail: sla is plan-time advisory only — never thread into learnedScore (would reroute warm rivals).
|
|
135
125
|
const scoreOpts = { availWeight: cfg.routing.learnedTuning?.availWeight };
|
|
136
126
|
const entry = cfg.routing.map[task.shape];
|
|
137
|
-
const prefer = effectivePrefer(
|
|
127
|
+
const prefer = effectivePrefer(entry);
|
|
138
128
|
const prefActive = !!(cfg.routing.allow || cfg.routing.deny);
|
|
139
129
|
const disallowedPin = (via, model, kind) => {
|
|
140
130
|
const d = disallowedBy({ adapter: via, model }, cfg.routing);
|
|
@@ -167,6 +157,14 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
167
157
|
}
|
|
168
158
|
};
|
|
169
159
|
const taskFloor = task.routingHints?.floor;
|
|
160
|
+
const mapPinFloor = taskFloor && (!advisoryFloor || TIER_RANK[taskFloor] >= TIER_RANK[advisoryFloor])
|
|
161
|
+
? { tier: taskFloor, source: "task" }
|
|
162
|
+
: advisoryFloor ? { tier: advisoryFloor, source: "config" } : undefined;
|
|
163
|
+
const lintMapPinFloor = (tier) => {
|
|
164
|
+
if (mapPinFloor && TIER_RANK[tier] < TIER_RANK[mapPinFloor.tier]) {
|
|
165
|
+
lints.push(`${task.id} (${task.shape}): map pin routes ${tier}, below ${mapPinFloor.source} floor ${mapPinFloor.tier} — map pins are supreme`);
|
|
166
|
+
}
|
|
167
|
+
};
|
|
170
168
|
const source = task.routingHints?.source;
|
|
171
169
|
const src = source ? `, ${source}` : ""; // never interpolate a possibly-undefined source
|
|
172
170
|
// task pin: planner-authored, try-first — degrades on miss or below-floor (D-05, research A3), never throws
|
|
@@ -188,12 +186,13 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
188
186
|
if (entry?.pin) {
|
|
189
187
|
disallowedPin(entry.pin.via, entry.pin.model, "map pin (config routing.map)");
|
|
190
188
|
const c = resolvePin(entry.pin, channels);
|
|
191
|
-
|
|
189
|
+
lintMapPinFloor(c.tier);
|
|
192
190
|
maybeSlaLint(lints, task, profile, slaMinutes, c);
|
|
193
191
|
return { assignment: toAssignment(c), ladder: ladderFor(task, entry), lints, provenance: `${degraded}pin ${entry.pin.via}:${entry.pin.model} (config routing.map)` };
|
|
194
192
|
}
|
|
195
193
|
const baseTier = floor ?? "cheap";
|
|
196
|
-
|
|
194
|
+
// D-04: a task floor is hard only on floor/auto paths; the map-pin branch above stays supreme.
|
|
195
|
+
const minTier = taskFloor && TIER_RANK[taskFloor] > TIER_RANK[baseTier] ? taskFloor : baseTier;
|
|
197
196
|
if (prefActive)
|
|
198
197
|
for (const p of prefer ?? [])
|
|
199
198
|
preflightPrefer(p);
|
|
@@ -293,13 +292,10 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
293
292
|
"tier cheap (default)";
|
|
294
293
|
// name the key that actually broke the tie: prefer outranks the marginal-cost/tier keys, so if the
|
|
295
294
|
// winner matched a prefer entry, prefer decided it — not "cheapest sufficient tier" (ROUTE-03, WR-01)
|
|
296
|
-
const preferVia = preferFromAuto(task.shape, preferCtx)
|
|
297
|
-
? `via prefer (auto-modernized ${preferCtx.autoPrefer.derivedAt.slice(0, 10)})`
|
|
298
|
-
: "via prefer";
|
|
299
295
|
// a spread-decided winner is never inside a prefer band (the spread skips those runs), so the
|
|
300
296
|
// three arms below are mutually exclusive by construction
|
|
301
297
|
const chosenBy = learnedChosen || (spreadDecided ? "via frontier spread"
|
|
302
|
-
: prefer && preferIndex(eligible[0], prefer) < prefer.length ?
|
|
298
|
+
: prefer && preferIndex(eligible[0], prefer) < prefer.length ? "via prefer" : "cheapest sufficient tier");
|
|
303
299
|
maybeSlaLint(lints, task, profile, slaMinutes, eligible[0]);
|
|
304
300
|
return { assignment: toAssignment(eligible[0]), ladder: ladderFor(task, entry), lints, provenance: `${degraded}${bound}, marginal-cost auto (${chosenBy})`, ...(deviation ? { deviation } : {}) };
|
|
305
301
|
}
|
package/dist/run/consult.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { WorkerAdapter } from "../adapters/types.js";
|
|
|
2
2
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import type { ExecutorDriver, Slot } from "../drivers/types.js";
|
|
4
4
|
import type { GateResult } from "../gates/types.js";
|
|
5
|
+
import { type VerdictUnparseableCause } from "../gates/verdict-cause.js";
|
|
5
6
|
export interface ConsultVerdict {
|
|
6
7
|
action: "retry" | "reroute" | "decompose" | "human";
|
|
7
8
|
notes: string;
|
|
@@ -23,6 +24,11 @@ export interface Dossier {
|
|
|
23
24
|
diff: string;
|
|
24
25
|
gates: GateResult[];
|
|
25
26
|
}
|
|
27
|
+
export interface ConsultParseResult {
|
|
28
|
+
verdict: ConsultVerdict | null;
|
|
29
|
+
cause?: VerdictUnparseableCause;
|
|
30
|
+
}
|
|
31
|
+
export declare function parseConsultVerdict(out: string, nonce: string): ConsultParseResult;
|
|
26
32
|
export declare function buildDossierPrompt(d: Dossier, nonce: string): string;
|
|
27
33
|
export declare function consult(d: Dossier, cfg: TickmarkrConfig, adapters: WorkerAdapter[], driver: ExecutorDriver, cwd: string, runDir: string, opts?: {
|
|
28
34
|
keep?: boolean;
|