tickmarkr 2.2.1 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/dist/adapters/catalog-remote.d.ts +1 -4
- package/dist/adapters/catalog-remote.js +52 -42
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/types.d.ts +21 -1
- package/dist/adapters/types.js +43 -2
- package/dist/cli/commands/approve.js +5 -4
- package/dist/cli/commands/beat.js +7 -4
- package/dist/cli/commands/compile.js +32 -6
- package/dist/cli/commands/doctor.d.ts +9 -4
- package/dist/cli/commands/doctor.js +87 -13
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +53 -14
- package/dist/cli/commands/init.js +36 -21
- package/dist/cli/commands/plan.js +45 -7
- package/dist/cli/commands/report.js +37 -1
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +45 -1
- package/dist/cli/commands/verify.d.ts +6 -0
- package/dist/cli/commands/verify.js +145 -25
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +2 -9
- package/dist/compile/collateral.js +17 -18
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +41 -7
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +58 -9
- package/dist/compile/ownership.js +34 -9
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +52 -6
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +11 -1
- package/dist/drivers/index.d.ts +6 -0
- package/dist/drivers/index.js +19 -4
- package/dist/drivers/orca.d.ts +35 -1
- package/dist/drivers/orca.js +260 -20
- package/dist/drivers/subprocess.d.ts +3 -3
- package/dist/drivers/subprocess.js +16 -9
- package/dist/drivers/types.d.ts +12 -0
- package/dist/gates/baseline.d.ts +2 -0
- package/dist/gates/baseline.js +47 -11
- package/dist/gates/llm.d.ts +6 -0
- package/dist/gates/llm.js +25 -9
- package/dist/gates/review.d.ts +7 -3
- package/dist/gates/review.js +61 -22
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +50 -26
- package/dist/gates/verdict-cause.d.ts +6 -2
- package/dist/gates/verdict-cause.js +8 -4
- package/dist/route/preference.d.ts +4 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +39 -8
- package/dist/run/daemon.d.ts +16 -0
- package/dist/run/daemon.js +345 -74
- package/dist/run/git.d.ts +3 -0
- package/dist/run/git.js +40 -5
- package/dist/run/journal.d.ts +15 -2
- package/dist/run/journal.js +73 -12
- package/dist/run/supervision.d.ts +6 -0
- package/dist/run/supervision.js +29 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/dist/tui/ink/init-app.js +4 -4
- package/package.json +59 -1
- package/skills/tickmarkr-overseer/SKILL.md +77 -18
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
package/dist/gates/baseline.js
CHANGED
|
@@ -5,10 +5,19 @@ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "..
|
|
|
5
5
|
// codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
|
|
6
6
|
// a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
|
|
7
7
|
// baselines stored by pre-hardening code.
|
|
8
|
-
|
|
8
|
+
// OBS-891 (run 3372): vitest toggles the cursor (`\x1b[?25l` / `\x1b[?25h`) around its progress
|
|
9
|
+
// output, and the private-mode parameter byte `?` never matched [\d;#], so an echo-block HEADER glued
|
|
10
|
+
// to a cursor-show sequence stayed invisible to withoutVitestEchoBlocks and its whole block leaked as
|
|
11
|
+
// runner evidence — seven prose-only "infra" parks in one night. Full CSI grammar: parameter bytes
|
|
12
|
+
// 0x30–0x3F (plus `#` for digit-normalized stored baselines), intermediates 0x20–0x2F, final 0x40–0x7E.
|
|
13
|
+
const ANSI_RE = /\x1b\[[0-?#]*[ -/]*[@-~]/g;
|
|
9
14
|
// ponytail: only leading ✓/✔ after optional "label:" prefixes (turbo/vitest), or tickmarkr's own run
|
|
10
15
|
// summary, counts as a pass line — other runners' pass markers (PASS, ok) stay fingerprintable
|
|
11
16
|
const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:tickmarkr\s+[\w.-]+:\s+)?(?:\d+|#)\s+done,\s+(?:\d+|#)\s+failed(?:,\s+(?:\d+|#)\s+awaiting human)?\b)/;
|
|
17
|
+
// OBS-888: tickmarkr's own operator lines (`tickmarkr: baseline capture for "test" … spawn EAGAIN`)
|
|
18
|
+
// are printed by this product, never by a runner about the work. When this repository's tests exercise
|
|
19
|
+
// the capture path they print them too, carrying errno tokens INFRA_RE would read as host evidence.
|
|
20
|
+
const OPERATOR_LINE_RE = /^\s*tickmarkr: /;
|
|
12
21
|
// HYG-08 (D-01, incident run-20260711-154920): a failing test went unnamed for 3 attempts because details
|
|
13
22
|
// headlined benign fingerprint-diff noise. These anchors harvest the runner's OWN failure naming from fresh
|
|
14
23
|
// output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
|
|
@@ -149,7 +158,10 @@ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
|
|
|
149
158
|
* unreadable-runner case the existing fail-closed path already owns.
|
|
150
159
|
*/
|
|
151
160
|
export function classifyFailureOutput(output) {
|
|
152
|
-
|
|
161
|
+
// OBS-891: the gate reads the WHOLE output here when the fresh-fingerprint diff is empty, so an errno
|
|
162
|
+
// token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
|
|
163
|
+
// to fingerprint(): test-owned output is never runner evidence about the work.
|
|
164
|
+
const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
153
165
|
if (lines.some(namesRegression))
|
|
154
166
|
return "regression";
|
|
155
167
|
return lines.some(isInfraLine) ? "infra" : undefined;
|
|
@@ -161,7 +173,11 @@ export function classifyFailureOutput(output) {
|
|
|
161
173
|
* no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
|
|
162
174
|
* separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
|
|
163
175
|
*/
|
|
164
|
-
|
|
176
|
+
// OBS-888 row 1: vitest 3.2.7 heads an echo block `std{out,err} | <file> > <test>` when the log is
|
|
177
|
+
// attributed to a test, `stderr | unknown test` when it is not, `stderr | <file>` for file-level output
|
|
178
|
+
// and `stderr | <task id>` (digits and underscores) when the reporter no longer knows the task
|
|
179
|
+
// (dist/chunks/index.*.js, `headerText`). The stripper knew only the first form.
|
|
180
|
+
const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+(?:unknown test\s*$|\d+_[\d_]+\s*$|\S+\.(?:test|spec)\.[cm]?[jt]sx?(?:\s*$|\s+>\s+\S))/;
|
|
165
181
|
/** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
|
|
166
182
|
const withoutVitestEchoBlocks = (output) => {
|
|
167
183
|
const outside = [];
|
|
@@ -186,7 +202,7 @@ const captureInvalidatingLines = (output) => {
|
|
|
186
202
|
const invalidating = [];
|
|
187
203
|
for (const line of withoutVitestEchoBlocks(output)) {
|
|
188
204
|
const clean = line.replace(ANSI_RE, "");
|
|
189
|
-
if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
|
|
205
|
+
if (!PASS_LINE_RE.test(clean) && !OPERATOR_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
|
|
190
206
|
invalidating.push(line);
|
|
191
207
|
}
|
|
192
208
|
return invalidating;
|
|
@@ -230,16 +246,18 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
|
|
|
230
246
|
export function fingerprint(output) {
|
|
231
247
|
const lines = withoutVitestEchoBlocks(output)
|
|
232
248
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
233
|
-
.filter((l) => !PASS_LINE_RE.test(l));
|
|
249
|
+
.filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
234
250
|
// GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
|
|
235
251
|
// prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
|
|
236
252
|
// failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
|
|
237
253
|
// prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
|
|
238
254
|
const shaped = [];
|
|
239
255
|
for (const l of lines) {
|
|
256
|
+
const stripped = stripTurboPrefix(l);
|
|
257
|
+
if (stripped !== undefined && OPERATOR_LINE_RE.test(stripped))
|
|
258
|
+
continue; // OBS-888: operator prose under a turbo prefix
|
|
240
259
|
if (isFingerprintShaped(l))
|
|
241
260
|
shaped.push(l);
|
|
242
|
-
const stripped = stripTurboPrefix(l);
|
|
243
261
|
if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
|
|
244
262
|
shaped.push(stripped);
|
|
245
263
|
}
|
|
@@ -427,8 +445,9 @@ export function ceilingKillResult(gate, r, ceilingMs) {
|
|
|
427
445
|
* shell asks it to finish faster than the thing it is measuring.
|
|
428
446
|
*/
|
|
429
447
|
export const CAPTURE_CEILING_MS = 1_800_000;
|
|
430
|
-
const invalidCaptureEntry = (durationMs, invalidatingLines = []) => ({
|
|
448
|
+
const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) => ({
|
|
431
449
|
infra: true,
|
|
450
|
+
invalidCause,
|
|
432
451
|
fingerprints: [],
|
|
433
452
|
durationMs,
|
|
434
453
|
fileDurationSumMs: null,
|
|
@@ -463,7 +482,7 @@ export async function captureBaseline(cwd, commands) {
|
|
|
463
482
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
464
483
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
465
484
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
466
|
-
base.commands[name] = invalidCaptureEntry(durationMs);
|
|
485
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
|
|
467
486
|
continue;
|
|
468
487
|
}
|
|
469
488
|
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
@@ -479,7 +498,7 @@ export async function captureBaseline(cwd, commands) {
|
|
|
479
498
|
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
480
499
|
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
|
|
481
500
|
+ `First invalidating line: ${invalidatingLines[0]}`);
|
|
482
|
-
base.commands[name] = invalidCaptureEntry(durationMs, invalidatingLines);
|
|
501
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
483
502
|
continue;
|
|
484
503
|
}
|
|
485
504
|
base.commands[name] = {
|
|
@@ -575,7 +594,8 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
575
594
|
// result rather than re-derived after the fact. The skip row above ran no command and therefore
|
|
576
595
|
// states no capacity — a row that never divided the machine must not claim that it did.
|
|
577
596
|
const record = (g) => {
|
|
578
|
-
|
|
597
|
+
const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
|
|
598
|
+
results.push(r.capacity ? { ...withReap, capacity: r.capacity } : withReap);
|
|
579
599
|
};
|
|
580
600
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
581
601
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -591,6 +611,19 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
591
611
|
record(killed);
|
|
592
612
|
continue;
|
|
593
613
|
}
|
|
614
|
+
const runner = shellToken(cmd) ?? name;
|
|
615
|
+
if (entry?.missingCommand === true || r.code === 127) {
|
|
616
|
+
const cause = entry?.missingCommand === true && r.code !== 127
|
|
617
|
+
? "the baseline capture recorded this runner as missing"
|
|
618
|
+
: `the head command exited 127${/\bENOENT\b/i.test(`${r.stdout}\n${r.stderr}`) ? " (spawn ENOENT)" : ""}`;
|
|
619
|
+
record({
|
|
620
|
+
gate: name,
|
|
621
|
+
pass: false,
|
|
622
|
+
details: `unreadable — ${cause}; runner ${JSON.stringify(runner)} did not produce a trustworthy verdict`,
|
|
623
|
+
meta: { unreadable: true, runner },
|
|
624
|
+
});
|
|
625
|
+
continue;
|
|
626
|
+
}
|
|
594
627
|
if (r.code === 0) {
|
|
595
628
|
record({ gate: name, pass: true, details: "exit 0" });
|
|
596
629
|
continue;
|
|
@@ -629,8 +662,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
629
662
|
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
630
663
|
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
631
664
|
if (!failing.length && !baselineRed) {
|
|
665
|
+
const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
|
|
666
|
+
? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
|
|
667
|
+
: "was killed at its ceiling";
|
|
632
668
|
const closed = entry?.infra === true
|
|
633
|
-
? `the baseline capture for this command
|
|
669
|
+
? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
634
670
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
635
671
|
const evidence = unrecognizedEvidence(raw);
|
|
636
672
|
record({
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -57,9 +57,15 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
|
57
57
|
value: T;
|
|
58
58
|
outputs: string[];
|
|
59
59
|
}>;
|
|
60
|
+
export interface LlmRunResult {
|
|
61
|
+
output: string;
|
|
62
|
+
exitCode?: number;
|
|
63
|
+
timedOut: boolean;
|
|
64
|
+
}
|
|
60
65
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
61
66
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
62
67
|
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
|
68
|
+
export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
|
|
63
69
|
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
|
64
70
|
export declare function extractJson<T>(raw: string): T | null;
|
|
65
71
|
/** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
|
package/dist/gates/llm.js
CHANGED
|
@@ -129,21 +129,24 @@ export async function captureLlmOutput(run) {
|
|
|
129
129
|
const value = await llmOutputCapture.run(outputs, run);
|
|
130
130
|
return { value, outputs };
|
|
131
131
|
}
|
|
132
|
-
|
|
132
|
+
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
133
133
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
134
134
|
try {
|
|
135
135
|
const pf = join(dir, "prompt.md");
|
|
136
136
|
writeFileSync(pf, prompt);
|
|
137
137
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
138
|
-
return r.stdout + "\n" + r.stderr;
|
|
138
|
+
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
|
|
139
139
|
}
|
|
140
140
|
finally {
|
|
141
141
|
rmSync(dir, { recursive: true, force: true });
|
|
142
142
|
}
|
|
143
143
|
}
|
|
144
|
+
export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
145
|
+
return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
|
|
146
|
+
}
|
|
144
147
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
145
148
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
146
|
-
|
|
149
|
+
async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
147
150
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
148
151
|
let slot;
|
|
149
152
|
let accountant;
|
|
@@ -176,6 +179,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
176
179
|
// false-complete — same guard the worker path uses (daemon.ts:330-331).
|
|
177
180
|
const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
|
|
178
181
|
let out;
|
|
182
|
+
let timedOut = false;
|
|
179
183
|
const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
|
|
180
184
|
if (!gatePrompt) {
|
|
181
185
|
await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
|
|
@@ -241,8 +245,14 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
241
245
|
break;
|
|
242
246
|
}
|
|
243
247
|
}
|
|
248
|
+
timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
|
|
244
249
|
}
|
|
245
|
-
|
|
250
|
+
const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
|
|
251
|
+
return {
|
|
252
|
+
output: dewrapPaneVerdict(out, nonce),
|
|
253
|
+
...(Number.isFinite(exitCode) ? { exitCode } : {}),
|
|
254
|
+
timedOut,
|
|
255
|
+
};
|
|
246
256
|
}
|
|
247
257
|
finally {
|
|
248
258
|
try {
|
|
@@ -256,6 +266,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
256
266
|
}
|
|
257
267
|
}
|
|
258
268
|
}
|
|
269
|
+
export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
270
|
+
return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
271
|
+
}
|
|
259
272
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
260
273
|
// continuation indent, splitting words mid-token — so literal newlines land inside JSON string
|
|
261
274
|
// literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
|
|
@@ -315,12 +328,15 @@ export function dewrapPaneVerdict(out, nonce) {
|
|
|
315
328
|
}
|
|
316
329
|
return out;
|
|
317
330
|
}
|
|
331
|
+
export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
332
|
+
const result = await (via
|
|
333
|
+
? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
|
|
334
|
+
: runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
|
|
335
|
+
llmOutputCapture.getStore()?.push(result.output);
|
|
336
|
+
return result;
|
|
337
|
+
}
|
|
318
338
|
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
319
|
-
|
|
320
|
-
? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
|
|
321
|
-
: runHeadless(adapter, model, prompt, cwd, timeoutMs));
|
|
322
|
-
llmOutputCapture.getStore()?.push(out);
|
|
323
|
-
return out;
|
|
339
|
+
return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
324
340
|
}
|
|
325
341
|
export function extractJson(raw) {
|
|
326
342
|
const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -41,16 +41,20 @@ export type TaskDiffMeasurement = {
|
|
|
41
41
|
readonly fullMeasurement: ArtifactDiffMeasurement;
|
|
42
42
|
readonly capMeasurement: ArtifactDiffMeasurement;
|
|
43
43
|
};
|
|
44
|
-
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
|
|
44
|
+
export declare function fetchTaskDiff(worktree: string, baseRef: string, files?: readonly string[]): Promise<TaskDiffMeasurement>;
|
|
45
45
|
export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
|
|
46
46
|
/** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
|
|
47
47
|
export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
|
|
48
48
|
export declare function isDiffCapPark(result: GateResult): boolean;
|
|
49
49
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
50
50
|
export declare function modelId(model: string): string;
|
|
51
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
52
|
+
export declare function modelProvider(model: string, fallback?: string): string;
|
|
51
53
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
52
54
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
53
|
-
floor?: Tier
|
|
55
|
+
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
|
56
|
+
history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
57
|
+
onSeat?: (seat: number) => void): BillingChannel | null;
|
|
54
58
|
export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
55
59
|
/**
|
|
56
60
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
@@ -58,4 +62,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
|
58
62
|
* judgement rather than a guarantee made by this renderer.
|
|
59
63
|
*/
|
|
60
64
|
export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
|
|
61
|
-
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
|
|
65
|
+
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -2,12 +2,13 @@ import { writeFileSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
5
|
+
import { filesGlob } from "../graph/files-glob.js";
|
|
5
6
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
6
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
8
|
import { shOk } from "../run/git.js";
|
|
8
9
|
import { redactSecrets } from "../run/redact.js";
|
|
9
10
|
import { marginalCostRank } from "../route/router.js";
|
|
10
|
-
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce,
|
|
11
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
11
12
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
12
13
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
13
14
|
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
@@ -99,12 +100,14 @@ export async function mirrorsVersionOnly(worktree, baseRef, path) {
|
|
|
99
100
|
const changed = diff.split("\n").filter((l) => /^[+-]/.test(l) && !/^(?:\+\+\+|---)/.test(l));
|
|
100
101
|
return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
|
|
101
102
|
}
|
|
102
|
-
export async function fetchTaskDiff(worktree, baseRef) {
|
|
103
|
+
export async function fetchTaskDiff(worktree, baseRef, files = []) {
|
|
103
104
|
// --full-index: abbreviated index lines vary with object-store density, so two measurements of
|
|
104
105
|
// the same diff could disagree by a few bytes between invocations (CI-only red, release 1.89.0).
|
|
105
|
-
const
|
|
106
|
-
|
|
107
|
-
|
|
106
|
+
const matched = files.length ? (await changedPaths(worktree, baseRef)).filter(filesGlob([...files])) : [];
|
|
107
|
+
const pathspec = matched.length ? ` -- ${matched.map(shq).join(" ")}` : "";
|
|
108
|
+
const [rawFull, rawForCap] = files.length && matched.length === 0 ? ["", ""] : await Promise.all([
|
|
109
|
+
shOk(`git diff --full-index '${baseRef}..HEAD'${pathspec}`, worktree),
|
|
110
|
+
shOk(`git diff --full-index -U0 '${baseRef}..HEAD'${pathspec}`, worktree),
|
|
108
111
|
]);
|
|
109
112
|
const fullMeasurement = measureArtifactDiff(rawFull);
|
|
110
113
|
const capMeasurement = measureArtifactDiff(rawForCap);
|
|
@@ -162,6 +165,24 @@ export function diffCapParkReason(results) {
|
|
|
162
165
|
export function modelId(model) {
|
|
163
166
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
164
167
|
}
|
|
168
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
169
|
+
export function modelProvider(model, fallback = "unknown") {
|
|
170
|
+
const id = model.toLowerCase();
|
|
171
|
+
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
172
|
+
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
173
|
+
return "openai";
|
|
174
|
+
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
175
|
+
return "anthropic";
|
|
176
|
+
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
177
|
+
return "google";
|
|
178
|
+
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
179
|
+
return "xai";
|
|
180
|
+
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
181
|
+
return "zhipu";
|
|
182
|
+
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
183
|
+
return "moonshot";
|
|
184
|
+
return fallback;
|
|
185
|
+
}
|
|
165
186
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
166
187
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
167
188
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -171,7 +192,9 @@ function reviewPreferIndex(c, prefer) {
|
|
|
171
192
|
}
|
|
172
193
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
173
194
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
174
|
-
floor
|
|
195
|
+
floor, // task-declared only; config floors govern workers and must not silently move review seats
|
|
196
|
+
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
197
|
+
onSeat) {
|
|
175
198
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
176
199
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
177
200
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -179,15 +202,23 @@ floor) {
|
|
|
179
202
|
const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
|
|
180
203
|
if (!authorChannel)
|
|
181
204
|
return null;
|
|
182
|
-
|
|
205
|
+
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
206
|
+
const ranked = channels
|
|
183
207
|
// two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
|
|
184
|
-
// rule, never replacing it — a future edit can't silently drop either).
|
|
185
|
-
//
|
|
208
|
+
// rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
|
|
209
|
+
// true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
|
|
210
|
+
// filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
|
|
186
211
|
.filter((c) => c.vendor !== authorChannel.vendor
|
|
212
|
+
&& (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
|
|
187
213
|
&& modelId(c.model) !== modelId(author.model)
|
|
188
214
|
&& !exclude.includes(channelKey(c))
|
|
189
215
|
&& (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
|
|
190
|
-
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))
|
|
216
|
+
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
217
|
+
const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
218
|
+
|| ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
|
|
219
|
+
if (reviewer)
|
|
220
|
+
onSeat?.(ranked.indexOf(reviewer) + 1);
|
|
221
|
+
return reviewer;
|
|
191
222
|
}
|
|
192
223
|
/**
|
|
193
224
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
@@ -205,7 +236,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
205
236
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
206
237
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
207
238
|
// direct tests) skips persistence and changes nothing else.
|
|
208
|
-
artifactDir) {
|
|
239
|
+
artifactDir, reviewHistory) {
|
|
209
240
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
210
241
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
211
242
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -277,7 +308,8 @@ artifactDir) {
|
|
|
277
308
|
// A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
|
|
278
309
|
// historical seat for every task that never asked for review-tier coupling.
|
|
279
310
|
const reviewerFloor = task.routingHints?.floor;
|
|
280
|
-
|
|
311
|
+
let rotationSeat;
|
|
312
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
|
|
281
313
|
if (!reviewer) {
|
|
282
314
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
283
315
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
@@ -285,10 +317,12 @@ artifactDir) {
|
|
|
285
317
|
? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
|
|
286
318
|
: "no cross-vendor reviewer available (diversity rule)";
|
|
287
319
|
return cfg.review.required
|
|
288
|
-
? { gate: "review", pass: false, details:
|
|
320
|
+
? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
289
321
|
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
|
|
290
322
|
}
|
|
291
|
-
|
|
323
|
+
reviewHistory?.push(channelKey(reviewer));
|
|
324
|
+
const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
|
|
325
|
+
const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
|
|
292
326
|
// Keep the reader payload identical to the text charged to the strict cap:
|
|
293
327
|
// whole-file source deletions are represented by their citable operation fact.
|
|
294
328
|
const diff = reviewableLogicDiff(measuredDiff.full);
|
|
@@ -327,7 +361,7 @@ Approve iff no material finding remains; an empty findings list is a clean appro
|
|
|
327
361
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
328
362
|
`;
|
|
329
363
|
let concludedOnInactivity = false;
|
|
330
|
-
const
|
|
364
|
+
const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
|
|
331
365
|
driver: via.driver,
|
|
332
366
|
keep: via.keep,
|
|
333
367
|
onSlot: via.onSlot,
|
|
@@ -338,16 +372,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
338
372
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
339
373
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
340
374
|
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
341
|
-
// (run-20260709-104447 P87-09).
|
|
342
|
-
|
|
343
|
-
|
|
375
|
+
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
376
|
+
cfg.review.timeoutMs);
|
|
377
|
+
const raw = llm.output;
|
|
378
|
+
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
344
379
|
const v = extractVerdictJson(raw, nonce);
|
|
345
380
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
346
381
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
347
382
|
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
348
383
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
349
384
|
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
350
|
-
const cause = classifyVerdictCause(raw, nonce, "approve");
|
|
385
|
+
const cause = classifyVerdictCause(raw, nonce, "approve", llm);
|
|
351
386
|
const bytes = Buffer.byteLength(raw, "utf8");
|
|
352
387
|
let saved;
|
|
353
388
|
if (artifactDir) {
|
|
@@ -367,13 +402,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
367
402
|
return {
|
|
368
403
|
gate: "review",
|
|
369
404
|
pass: false,
|
|
370
|
-
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
405
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
371
406
|
meta: {
|
|
372
407
|
...policyMeta,
|
|
408
|
+
...rotationMeta,
|
|
373
409
|
reviewer: channelKey(reviewer),
|
|
410
|
+
vendor: reviewer.vendor,
|
|
411
|
+
provider,
|
|
374
412
|
unparseable: true,
|
|
375
413
|
cause,
|
|
376
414
|
...(cause === "empty-output" ? { bytes } : {}),
|
|
415
|
+
...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
|
|
377
416
|
...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
|
|
378
417
|
},
|
|
379
418
|
};
|
|
@@ -381,11 +420,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
381
420
|
const decided = findings !== null
|
|
382
421
|
? classifyReviewFindings(findings)
|
|
383
422
|
: classifyReviewIssues(v.approve, v.issues);
|
|
384
|
-
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
423
|
+
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
385
424
|
return {
|
|
386
425
|
gate: "review",
|
|
387
426
|
pass: decided.pass,
|
|
388
427
|
details: appendAnchoredReview(prose, v),
|
|
389
|
-
meta: { ...policyMeta, reviewer: channelKey(reviewer) },
|
|
428
|
+
meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
|
|
390
429
|
};
|
|
391
430
|
}
|
|
@@ -13,13 +13,15 @@ export declare function resetLoadProviderForTests(): void;
|
|
|
13
13
|
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
14
14
|
* gates, then the full suite) reports the two suites' cost rather than the span containing them —
|
|
15
15
|
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
16
|
-
* queue as well as the work.
|
|
17
|
-
*
|
|
16
|
+
* queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
|
|
17
|
+
* start preserves the scheduling input while max and mean retain sustained interior saturation.
|
|
18
18
|
*/
|
|
19
19
|
export interface GateTelemetry {
|
|
20
20
|
durationMs: number;
|
|
21
21
|
load1Start: number;
|
|
22
22
|
load1End: number;
|
|
23
|
+
load1Max: number;
|
|
24
|
+
load1Mean: number;
|
|
23
25
|
}
|
|
24
26
|
export type GateEvent = {
|
|
25
27
|
phase: "start";
|
|
@@ -51,6 +53,7 @@ export interface GateContext {
|
|
|
51
53
|
cfg: TickmarkrConfig;
|
|
52
54
|
via?: GateVia;
|
|
53
55
|
excludeReviewers?: string[];
|
|
56
|
+
reviewHistory?: string[];
|
|
54
57
|
artifactDir?: string;
|
|
55
58
|
pipeline?: "v185" | "legacy";
|
|
56
59
|
selectTests?: boolean;
|