tickmarkr 1.83.0 → 1.85.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.d.ts +1 -0
- package/dist/adapters/claude-code.js +57 -1
- package/dist/adapters/fake.js +9 -0
- package/dist/adapters/types.d.ts +3 -0
- package/dist/adapters/types.js +21 -0
- package/dist/cli/commands/status.js +160 -30
- package/dist/compile/collateral.d.ts +86 -2
- package/dist/compile/collateral.js +294 -3
- package/dist/config/config.d.ts +62 -0
- package/dist/config/config.js +157 -2
- package/dist/drivers/herdr.d.ts +20 -3
- package/dist/drivers/herdr.js +288 -105
- package/dist/gates/baseline.d.ts +1 -0
- package/dist/gates/baseline.js +91 -13
- package/dist/gates/review.d.ts +7 -0
- package/dist/gates/review.js +99 -6
- package/dist/gates/run-gates.d.ts +9 -0
- package/dist/gates/run-gates.js +285 -41
- package/dist/run/daemon.d.ts +48 -2
- package/dist/run/daemon.js +1417 -315
- package/dist/run/journal.d.ts +56 -3
- package/dist/run/journal.js +275 -1
- package/dist/run/stall.d.ts +35 -1
- package/dist/run/stall.js +118 -8
- package/dist/tui/cockpit/components.d.ts +30 -1
- package/dist/tui/cockpit/components.js +19 -3
- package/dist/tui/cockpit/derive.d.ts +29 -2
- package/dist/tui/cockpit/derive.js +219 -23
- package/dist/tui/cockpit/layout.d.ts +75 -6
- package/dist/tui/cockpit/layout.js +97 -19
- package/dist/tui/cockpit/live.d.ts +14 -1
- package/dist/tui/cockpit/live.js +223 -29
- package/dist/tui/cockpit/pointer.d.ts +261 -0
- package/dist/tui/cockpit/pointer.js +610 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +36 -5
- package/dist/tui/cockpit/run-cockpit.js +270 -51
- package/package.json +1 -1
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { posix } from "node:path";
|
|
3
|
+
import { channelKey, shq } from "../adapters/types.js";
|
|
2
4
|
import { TIER_RANK } from "../config/config.js";
|
|
3
5
|
import { getAdapter } from "../adapters/registry.js";
|
|
4
6
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
@@ -9,13 +11,118 @@ import { captureLlmOutput } from "./llm.js";
|
|
|
9
11
|
import { marginalCostRank } from "../route/router.js";
|
|
10
12
|
import { reviewGate } from "./review.js";
|
|
11
13
|
import { scopeGate } from "./scope.js";
|
|
14
|
+
import { shGit } from "../run/git.js";
|
|
12
15
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
16
|
+
const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
|
|
17
|
+
// relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
|
|
18
|
+
const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
|
|
19
|
+
const SELECTION_FILE_CAP = 3000;
|
|
20
|
+
/**
|
|
21
|
+
* T4: the tests covering this round's diff, or undefined when the diff cannot be attributed with
|
|
22
|
+
* certainty — a rename or delete (the old path's coverage is gone), or a changed file no test
|
|
23
|
+
* reaches. Coverage: a test file covers itself; a test covers every file reachable from it through
|
|
24
|
+
* relative imports, directly or transitively.
|
|
25
|
+
*
|
|
26
|
+
* ponytail: ceiling — relative specifiers only (no tsconfig paths, no bare aliases, no computed
|
|
27
|
+
* specifiers), and an import cycle contributes only what it had resolved when re-entered. So this
|
|
28
|
+
* CAN miss. The miss is bounded by construction, not by care: the merge-candidate round re-runs the
|
|
29
|
+
* full suite on the same commit, so a miss costs one round and can never merge. Teach it a resolver
|
|
30
|
+
* (tsconfig paths, package exports) if selection ever misses often enough to be worth a round.
|
|
31
|
+
*/
|
|
32
|
+
async function coveringTests(worktree, baseRef) {
|
|
33
|
+
const diff = await shGit(`git diff --name-status ${shq(baseRef)} HEAD`, worktree);
|
|
34
|
+
if (diff.code !== 0)
|
|
35
|
+
return undefined;
|
|
36
|
+
const changed = [];
|
|
37
|
+
for (const line of diff.stdout.split("\n")) {
|
|
38
|
+
if (!line.trim())
|
|
39
|
+
continue;
|
|
40
|
+
const parts = line.split("\t");
|
|
41
|
+
const status = parts[0] ?? "";
|
|
42
|
+
// R (rename) and D (delete): whatever used to cover the old path is unattributable now — full suite.
|
|
43
|
+
if (!status || status[0] === "R" || status[0] === "D" || parts.length < 2)
|
|
44
|
+
return undefined;
|
|
45
|
+
changed.push(parts[parts.length - 1]);
|
|
46
|
+
}
|
|
47
|
+
if (!changed.length)
|
|
48
|
+
return undefined;
|
|
49
|
+
const listed = await shGit("git ls-files", worktree);
|
|
50
|
+
if (listed.code !== 0)
|
|
51
|
+
return undefined;
|
|
52
|
+
const tracked = listed.stdout.split("\n").filter(Boolean);
|
|
53
|
+
if (tracked.length > SELECTION_FILE_CAP)
|
|
54
|
+
return undefined; // ponytail: a huge repo pays the full suite rather than a long scan
|
|
55
|
+
const trackedSet = new Set(tracked);
|
|
56
|
+
const tests = tracked.filter((p) => TEST_FILE_RE.test(p));
|
|
57
|
+
if (!tests.length)
|
|
58
|
+
return undefined;
|
|
59
|
+
const resolveSpec = (from, spec) => {
|
|
60
|
+
const base = posix.join(posix.dirname(from), spec);
|
|
61
|
+
// ESM-TS writes ".js" for a ".ts" source; a directory specifier means its index.
|
|
62
|
+
const candidates = [base, base.replace(/\.js$/, ".ts"), base.replace(/\.jsx$/, ".tsx"),
|
|
63
|
+
`${base}.ts`, `${base}.tsx`, `${base}.js`, `${base}/index.ts`, `${base}/index.js`];
|
|
64
|
+
return candidates.find((c) => trackedSet.has(c));
|
|
65
|
+
};
|
|
66
|
+
const reachCache = new Map();
|
|
67
|
+
const reachOf = (file) => {
|
|
68
|
+
const cached = reachCache.get(file);
|
|
69
|
+
if (cached)
|
|
70
|
+
return cached;
|
|
71
|
+
const out = new Set();
|
|
72
|
+
reachCache.set(file, out); // cycle guard: a re-entered file contributes what it has so far
|
|
73
|
+
let src;
|
|
74
|
+
try {
|
|
75
|
+
src = readFileSync(posix.join(worktree, file), "utf8");
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
return out;
|
|
79
|
+
}
|
|
80
|
+
for (const m of src.matchAll(IMPORT_RE)) {
|
|
81
|
+
const dep = m[1] ? resolveSpec(file, m[1]) : undefined;
|
|
82
|
+
if (!dep || out.has(dep))
|
|
83
|
+
continue;
|
|
84
|
+
out.add(dep);
|
|
85
|
+
for (const t of reachOf(dep))
|
|
86
|
+
out.add(t);
|
|
87
|
+
}
|
|
88
|
+
return out;
|
|
89
|
+
};
|
|
90
|
+
const selected = new Set();
|
|
91
|
+
for (const file of changed) {
|
|
92
|
+
if (TEST_FILE_RE.test(file)) {
|
|
93
|
+
selected.add(file);
|
|
94
|
+
continue;
|
|
95
|
+
}
|
|
96
|
+
const covering = tests.filter((t) => reachOf(t).has(file));
|
|
97
|
+
if (!covering.length)
|
|
98
|
+
return undefined; // nothing covers this file — only the full suite can speak for it
|
|
99
|
+
for (const t of covering)
|
|
100
|
+
selected.add(t);
|
|
101
|
+
}
|
|
102
|
+
return [...selected].sort();
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
|
|
106
|
+
* npm/yarn/pnpm/npx script wrappers need one `--` to forward positional filters to the underlying runner;
|
|
107
|
+
* a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
|
|
108
|
+
*/
|
|
109
|
+
export function testCommandForFiles(testCmd, files) {
|
|
110
|
+
const wrapped = /^\s*(?:npm|yarn|pnpm|npx)\b/.test(testCmd);
|
|
111
|
+
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
112
|
+
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
113
|
+
}
|
|
13
114
|
export async function runGates(task, ctx) {
|
|
14
115
|
const results = [];
|
|
15
116
|
let commits = [];
|
|
16
117
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
17
118
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
18
119
|
const failed = () => results.some((r) => !r.pass);
|
|
120
|
+
const v185 = ctx.pipeline === "v185";
|
|
121
|
+
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
122
|
+
// round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
|
|
123
|
+
// exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
|
|
124
|
+
// (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
|
|
125
|
+
let heldTest;
|
|
19
126
|
const sequence = GATE_NAMES.filter((g) => enabled(g));
|
|
20
127
|
const total = sequence.length;
|
|
21
128
|
const indexOf = (gate) => sequence.indexOf(gate) + 1;
|
|
@@ -23,42 +130,109 @@ export async function runGates(task, ctx) {
|
|
|
23
130
|
results.push(result);
|
|
24
131
|
await ctx.onGate?.({ phase: "end", gate: result.gate, result });
|
|
25
132
|
};
|
|
26
|
-
const emitStart = async (gate) => {
|
|
27
|
-
await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total });
|
|
133
|
+
const emitStart = async (gate, parentAt) => {
|
|
134
|
+
await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total, ...(parentAt === undefined ? {} : { parentAt }) });
|
|
135
|
+
};
|
|
136
|
+
// The returned array stays in GATE_NAMES order however few gates a short-circuiting round reached.
|
|
137
|
+
// A round that ends before its merge-candidate stage flushes the held screen on the way out, so a
|
|
138
|
+
// green subset run is still journaled exactly once — as a selected run, which is what it was.
|
|
139
|
+
const done = async () => {
|
|
140
|
+
if (heldTest) {
|
|
141
|
+
const held = heldTest;
|
|
142
|
+
heldTest = undefined;
|
|
143
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: held });
|
|
144
|
+
}
|
|
145
|
+
return {
|
|
146
|
+
results: [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate)),
|
|
147
|
+
commits,
|
|
148
|
+
};
|
|
28
149
|
};
|
|
29
|
-
// 1. build/test/lint vs shared baseline — deterministic and cheap, first
|
|
30
150
|
const toolGates = ["build", "test", "lint"].filter(enabled);
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
151
|
+
// build/test/lint vs the shared baseline
|
|
152
|
+
const runBattery = async (commands, selected) => {
|
|
153
|
+
if (!toolGates.length)
|
|
154
|
+
return;
|
|
155
|
+
if (!v185) {
|
|
156
|
+
// ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
|
|
157
|
+
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
158
|
+
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
159
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
|
|
160
|
+
for (const r of toolResults) {
|
|
161
|
+
await emitStart(r.gate);
|
|
162
|
+
await record(r);
|
|
163
|
+
}
|
|
164
|
+
return;
|
|
39
165
|
}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
166
|
+
// T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
|
|
167
|
+
// the full vitest suite before anyone reads its verdict.
|
|
168
|
+
for (const g of toolGates) {
|
|
169
|
+
await emitStart(g);
|
|
170
|
+
const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
|
|
171
|
+
if (g === "test" && selected) {
|
|
172
|
+
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
173
|
+
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
174
|
+
if (!screened.pass)
|
|
175
|
+
await record(screened);
|
|
176
|
+
else {
|
|
177
|
+
heldTest = screened;
|
|
178
|
+
results.push(screened);
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
else {
|
|
182
|
+
await record(r);
|
|
183
|
+
}
|
|
184
|
+
if (failed())
|
|
185
|
+
return;
|
|
186
|
+
}
|
|
187
|
+
};
|
|
188
|
+
// The two sub-second git checks, as pure verdicts: no journal, no results push. Both read committed
|
|
189
|
+
// state only (commits ahead of base, `git diff --name-only base..HEAD`), so neither can be moved by
|
|
190
|
+
// anything the battery does to the worktree — which is what lets the screen below trust them early.
|
|
191
|
+
const evidenceResult = async () => {
|
|
46
192
|
const e = await evidenceGate(ctx.worktree, ctx.baseRef);
|
|
47
193
|
commits = e.commits;
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
194
|
+
return { gate: e.gate, pass: e.pass, details: e.details };
|
|
195
|
+
};
|
|
196
|
+
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, ctx.cfg.scope?.allowDeviations ?? []);
|
|
197
|
+
const runGate = async (gate, compute) => {
|
|
198
|
+
await emitStart(gate);
|
|
199
|
+
await record(await compute());
|
|
200
|
+
};
|
|
201
|
+
/**
|
|
202
|
+
* T4 (OBS-265): the deterministic git checks run BEFORE the battery, as a screen — they answer
|
|
203
|
+
* "is this diff worth starting a ~3.7m tool battery for?" before the first command runs. A red
|
|
204
|
+
* screen IS the round's verdict: what it produced is journaled (in the order it ran) and the round
|
|
205
|
+
* ends there, so a drive-by out-of-scope edit costs <1s instead of the whole battery.
|
|
206
|
+
*
|
|
207
|
+
* A green screen changes nothing downstream. The recorded sequence stays GATE_NAMES order, so the
|
|
208
|
+
* journal, `tickmarkr report`, the surfaces, and resume's GATE_NAMES walk over already-satisfied
|
|
209
|
+
* gates all keep reading exactly one order.
|
|
210
|
+
*
|
|
211
|
+
* ponytail: the price of that is re-reading two git checks (~40ms) in their canonical positions
|
|
212
|
+
* rather than teaching every consumer of the gate stream a second order. Both reads see the same
|
|
213
|
+
* commits — the battery never moves HEAD — so the screen cannot disagree with the gate it screens
|
|
214
|
+
* for. Charge it only when there IS a battery command to protect.
|
|
215
|
+
*/
|
|
216
|
+
const screenBlocks = async () => {
|
|
217
|
+
if (!toolGates.some((g) => ctx.commands[g]))
|
|
218
|
+
return false;
|
|
219
|
+
const screened = [];
|
|
220
|
+
for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
|
|
221
|
+
if (!enabled(gate))
|
|
222
|
+
continue;
|
|
223
|
+
screened.push(await compute());
|
|
224
|
+
if (screened[screened.length - 1].pass)
|
|
225
|
+
continue;
|
|
226
|
+
for (const r of screened) {
|
|
227
|
+
await emitStart(r.gate);
|
|
228
|
+
await record(r);
|
|
229
|
+
}
|
|
230
|
+
return true;
|
|
231
|
+
}
|
|
232
|
+
return false;
|
|
233
|
+
};
|
|
234
|
+
// acceptance judge — LLM spend, so everything deterministic has already passed when this runs
|
|
235
|
+
const runAcceptance = async () => {
|
|
62
236
|
const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
|
|
63
237
|
const jvia = ctx.via
|
|
64
238
|
? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", judgeAdapter.id), label: ctx.via.labelFor("judge") }
|
|
@@ -121,13 +295,10 @@ export async function runGates(task, ctx) {
|
|
|
121
295
|
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
122
296
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
123
297
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
// 5. cross-vendor review
|
|
129
|
-
if (enabled("review")) {
|
|
130
|
-
await emitStart("review");
|
|
298
|
+
return { result: a, invocations };
|
|
299
|
+
};
|
|
300
|
+
// cross-vendor review
|
|
301
|
+
const runReview = async () => {
|
|
131
302
|
let rv = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir);
|
|
132
303
|
// OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
|
|
133
304
|
// never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
|
|
@@ -145,7 +316,80 @@ export async function runGates(task, ctx) {
|
|
|
145
316
|
rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
|
|
146
317
|
}
|
|
147
318
|
}
|
|
148
|
-
|
|
319
|
+
return rv;
|
|
320
|
+
};
|
|
321
|
+
if (v185 && await screenBlocks())
|
|
322
|
+
return done();
|
|
323
|
+
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
324
|
+
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
325
|
+
const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
|
|
326
|
+
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
327
|
+
: undefined;
|
|
328
|
+
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected);
|
|
329
|
+
if (failed())
|
|
330
|
+
return done();
|
|
331
|
+
if (enabled("evidence")) {
|
|
332
|
+
await runGate("evidence", evidenceResult);
|
|
333
|
+
if (failed())
|
|
334
|
+
return done();
|
|
335
|
+
}
|
|
336
|
+
if (enabled("scope")) {
|
|
337
|
+
await runGate("scope", scopeResult);
|
|
338
|
+
if (failed())
|
|
339
|
+
return done();
|
|
340
|
+
}
|
|
341
|
+
if (v185 && (enabled("acceptance") || enabled("review"))) {
|
|
342
|
+
// Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
|
|
343
|
+
// unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
|
|
344
|
+
// verdict: each gets the same commit and the same brief it always got, and neither promise is
|
|
345
|
+
// reachable from inside the other. Only the waiting is gone.
|
|
346
|
+
const parentAt = Date.now();
|
|
347
|
+
// Both starts are emitted before either gate is launched, so the stream's order is the round's
|
|
348
|
+
// order and not a race between two dispatches. BOTH ARE IN FLIGHT BEFORE EITHER IS AWAITED:
|
|
349
|
+
// whichever adapter is slower no longer decides when the other one runs.
|
|
350
|
+
if (enabled("acceptance"))
|
|
351
|
+
await emitStart("acceptance", parentAt);
|
|
352
|
+
if (enabled("review"))
|
|
353
|
+
await emitStart("review", parentAt);
|
|
354
|
+
const judging = enabled("acceptance") ? runAcceptance() : undefined;
|
|
355
|
+
const reviewing = enabled("review") ? runReview() : undefined;
|
|
356
|
+
// Attach BOTH publication handlers before awaiting either. Dispatch concurrency alone is not
|
|
357
|
+
// enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
|
|
358
|
+
// process death can lose that already-earned verdict. The returned result is still sorted into
|
|
359
|
+
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
360
|
+
const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
|
|
361
|
+
const reviewed = reviewing?.then((outcome) => record(outcome));
|
|
362
|
+
await Promise.all([judged, reviewed]);
|
|
363
|
+
if (failed())
|
|
364
|
+
return done();
|
|
365
|
+
}
|
|
366
|
+
else if (!v185) {
|
|
367
|
+
// Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
|
|
368
|
+
if (enabled("acceptance")) {
|
|
369
|
+
await emitStart("acceptance");
|
|
370
|
+
const judged = await runAcceptance();
|
|
371
|
+
await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
|
|
372
|
+
if (failed())
|
|
373
|
+
return done();
|
|
374
|
+
}
|
|
375
|
+
if (enabled("review")) {
|
|
376
|
+
await emitStart("review");
|
|
377
|
+
await record(await runReview());
|
|
378
|
+
}
|
|
379
|
+
return done();
|
|
380
|
+
}
|
|
381
|
+
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
382
|
+
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
383
|
+
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
|
|
384
|
+
// held screen rather than joining it: one `test` entry in the record, one `test` end event in the
|
|
385
|
+
// stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
|
|
386
|
+
if (selected) {
|
|
387
|
+
await emitStart("test");
|
|
388
|
+
const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
|
|
389
|
+
const merged = { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
|
|
390
|
+
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
391
|
+
heldTest = undefined;
|
|
392
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
|
149
393
|
}
|
|
150
|
-
return
|
|
394
|
+
return done();
|
|
151
395
|
}
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import { type ExecutorDriver } from "../drivers/types.js";
|
|
4
|
-
import { type JournalEvent } from "./journal.js";
|
|
4
|
+
import { Journal, type JournalEvent } from "./journal.js";
|
|
5
5
|
export interface RunOptions {
|
|
6
6
|
runId?: string;
|
|
7
7
|
resume?: boolean;
|
|
@@ -42,13 +42,59 @@ export interface RunSummary {
|
|
|
42
42
|
lastMergedTask?: string;
|
|
43
43
|
}
|
|
44
44
|
export declare function formatSummary(s: RunSummary): string;
|
|
45
|
+
/**
|
|
46
|
+
* T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
|
|
47
|
+
* review are now launched together, so a round can journal a failed review that the serial walk would
|
|
48
|
+
* never have asked for — it returned at the judge. Those verdicts stay on the record (they are real,
|
|
49
|
+
* and the retry brief carries them), but they must not spend the OPERATOR-facing review round budget:
|
|
50
|
+
* otherwise concurrency alone parks a task rounds early for objections the old pipeline never bought.
|
|
51
|
+
* A round is the gate-result span opened by each `gates` phase-start, per task.
|
|
52
|
+
*/
|
|
53
|
+
export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
|
|
45
54
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
46
55
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
47
56
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
48
57
|
export declare function resetEarlyLaunchLivenessMsForTests(): void;
|
|
49
|
-
export
|
|
58
|
+
export { NUDGEABLE_ADAPTERS } from "./stall.js";
|
|
50
59
|
export declare const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
|
|
51
60
|
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
52
61
|
export declare function setNudgeTimingForTests(silentMs: number, graceMs: number): void;
|
|
53
62
|
export declare function resetNudgeTimingForTests(): void;
|
|
63
|
+
/** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
|
|
64
|
+
export declare function setQuotaBannerSilentMsForTests(ms: number): void;
|
|
65
|
+
export declare function resetQuotaBannerSilentMsForTests(): void;
|
|
66
|
+
/** Test seam — shrink the repeat-page cadence without minute-long sleeps. */
|
|
67
|
+
export declare function setPageRepeatMsForTests(ms: number): void;
|
|
68
|
+
export declare function resetPageRepeatMsForTests(): void;
|
|
69
|
+
/** Test seam — shrink the fast-kill window without minute-long sleeps. */
|
|
70
|
+
export declare function setDeadChannelFastKillMsForTests(ms: number): void;
|
|
71
|
+
export declare function resetDeadChannelFastKillMsForTests(): void;
|
|
72
|
+
/** Test seam — shrink the harvest silence gate without minute-long sleeps. */
|
|
73
|
+
export declare function setHarvestSilentMsForTests(ms: number): void;
|
|
74
|
+
export declare function resetHarvestSilentMsForTests(): void;
|
|
75
|
+
export declare function harvestCpuFlatWindowMs(resolutionMs: number): number;
|
|
76
|
+
/** Test seam — pin the flat window so a probe case need not sit through a real one. */
|
|
77
|
+
export declare function setHarvestCpuFlatMsForTests(ms: number): void;
|
|
78
|
+
export declare function resetHarvestCpuFlatMsForTests(): void;
|
|
79
|
+
export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
|
|
80
|
+
/** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
|
|
81
|
+
export declare function commandsHash(commands: Record<string, string>): string;
|
|
82
|
+
/**
|
|
83
|
+
* OBS-34's strict tip verify, but it stops re-paying for an unmoved tip (~334m corpus-wide; 69.5m in
|
|
84
|
+
* one park-heavy run whose 18 resume cycles merged nothing new). The verify journals the SHA it
|
|
85
|
+
* verified and the hash of the command set, so a later run-end can recognize the same verified state:
|
|
86
|
+
* head equals the LAST green verified SHA, commands unchanged, tree clean → journal
|
|
87
|
+
* `tip-verify-cached` and skip. ANY doubt — moved head, changed commands, dirty tree, a gate missing
|
|
88
|
+
* from that cycle's green set, a failure recorded in it, a cycle older than the last one — runs the
|
|
89
|
+
* full verify. The tip-verify-before-green law is untouched: a cached green is a verified green OF
|
|
90
|
+
* THAT EXACT COMMIT, established by the most recent real run of the same commands.
|
|
91
|
+
* Returns whether the tip is failing.
|
|
92
|
+
*/
|
|
93
|
+
export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
|
|
94
|
+
lastMergedTask?: string;
|
|
95
|
+
}): Promise<boolean>;
|
|
96
|
+
export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
|
|
97
|
+
ms: number;
|
|
98
|
+
resolutionMs: number;
|
|
99
|
+
} | undefined>;
|
|
54
100
|
export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;
|