tickmarkr 2.5.1 → 2.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.d.ts +1 -0
- package/dist/adapters/qwen.js +8 -0
- package/dist/adapters/types.d.ts +1 -0
- package/dist/cli/commands/approve.js +1 -1
- package/dist/cli/commands/doctor.d.ts +8 -0
- package/dist/cli/commands/doctor.js +50 -1
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/run.d.ts +5 -0
- package/dist/cli/commands/run.js +16 -1
- package/dist/cli/commands/status.js +8 -14
- package/dist/compile/native.js +3 -0
- package/dist/compile/ownership.d.ts +9 -0
- package/dist/compile/ownership.js +65 -2
- package/dist/config/config.d.ts +6 -0
- package/dist/config/config.js +6 -0
- package/dist/drivers/subprocess.d.ts +2 -3
- package/dist/drivers/subprocess.js +2 -3
- package/dist/gates/baseline.d.ts +13 -0
- package/dist/gates/baseline.js +99 -30
- package/dist/gates/llm.d.ts +1 -1
- package/dist/gates/llm.js +35 -13
- package/dist/gates/review.d.ts +11 -0
- package/dist/gates/review.js +67 -18
- package/dist/gates/run-gates.js +2 -2
- package/dist/run/daemon.d.ts +8 -3
- package/dist/run/daemon.js +3198 -2934
- package/dist/run/git.d.ts +4 -2
- package/dist/run/git.js +11 -23
- package/dist/run/journal.js +16 -7
- package/dist/run/merge.d.ts +1 -1
- package/dist/run/merge.js +126 -18
- package/dist/run/operator-state.d.ts +1 -1
- package/dist/run/operator-state.js +5 -9
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +61 -4
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +5 -2
- package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
package/dist/gates/baseline.js
CHANGED
|
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
|
|
|
30
30
|
// to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
|
|
31
31
|
// Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
|
|
32
32
|
// "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
|
|
33
|
-
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
33
|
+
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
34
34
|
const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
|
|
35
35
|
const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
|
|
36
36
|
const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
|
|
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
|
|
|
103
103
|
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
104
104
|
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
105
105
|
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
|
|
106
|
-
// The stripped form
|
|
107
|
-
//
|
|
108
|
-
// reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
|
|
106
|
+
// The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
|
|
107
|
+
// Classification applies the same prefix stripping before its infra/regression vetoes.
|
|
109
108
|
const namesFailureEitherForm = (l) => {
|
|
110
109
|
if (namesFailure(l))
|
|
111
110
|
return true;
|
|
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
|
|
|
163
162
|
// token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
|
|
164
163
|
// to fingerprint(): test-owned output is never runner evidence about the work.
|
|
165
164
|
const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
166
|
-
|
|
165
|
+
const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
|
|
166
|
+
const infra = evidence.some(isInfraLine);
|
|
167
|
+
// A diagnostic section heading names no failing test. It cannot outvote the RPC death
|
|
168
|
+
// beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
|
|
169
|
+
if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
|
|
167
170
|
return "regression";
|
|
168
|
-
return
|
|
171
|
+
return infra ? "infra" : undefined;
|
|
169
172
|
}
|
|
170
173
|
/**
|
|
171
174
|
* Capture validity asks a different question from gate classification. At a gate, one genuine
|
|
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
|
|
|
338
341
|
else if (scripts[name])
|
|
339
342
|
out[name] = `${runPrefix} ${name}`;
|
|
340
343
|
}
|
|
344
|
+
if (cfg.gates.tipTest)
|
|
345
|
+
out.tipTest = cfg.gates.tipTest;
|
|
341
346
|
return out;
|
|
342
347
|
}
|
|
343
348
|
/**
|
|
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
452
457
|
fingerprints: [],
|
|
453
458
|
durationMs,
|
|
454
459
|
fileDurationSumMs: null,
|
|
460
|
+
fileCount: null,
|
|
455
461
|
impliedParallelism: null,
|
|
456
462
|
longestFile: null,
|
|
457
463
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
460
466
|
export async function captureBaseline(cwd, commands) {
|
|
461
467
|
const base = { commands: {} };
|
|
462
468
|
for (const [name, cmd] of Object.entries(commands)) {
|
|
469
|
+
if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
|
|
470
|
+
continue;
|
|
471
|
+
}
|
|
463
472
|
const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
|
|
464
473
|
// ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
|
|
465
474
|
// ponytail: a capture that was itself killed records the ceiling as its "measurement", which
|
|
466
475
|
// scales the next ceiling up — the right direction for a suite that never finished once.
|
|
467
476
|
const durationMs = r.durationMs ?? 0;
|
|
477
|
+
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
478
|
+
const raw = combinedOutput.split(cwd).join("");
|
|
468
479
|
// OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
|
|
469
480
|
// kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
|
|
470
481
|
// exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
|
|
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
|
|
|
483
494
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
484
495
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
485
496
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
486
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
|
|
497
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
|
|
487
498
|
continue;
|
|
488
499
|
}
|
|
489
|
-
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
490
|
-
const raw = combinedOutput.split(cwd).join("");
|
|
491
500
|
// Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
|
|
492
501
|
// discriminator correctly called the mixed output a regression. But the same output also said
|
|
493
502
|
// `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
|
|
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
|
|
|
499
508
|
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
500
509
|
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
|
|
501
510
|
+ `First invalidating line: ${invalidatingLines[0]}`);
|
|
502
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
511
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
|
|
503
512
|
continue;
|
|
504
513
|
}
|
|
505
|
-
// OBS-
|
|
506
|
-
//
|
|
514
|
+
// OBS-966: a worker RPC timeout is infra even beside an all-green summary.
|
|
515
|
+
// Capture and both gate readers share this discriminator; genuine test failures still dominate.
|
|
507
516
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
508
517
|
if (runnerVerdict === "infra") {
|
|
509
|
-
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence
|
|
510
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "infra");
|
|
518
|
+
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
|
|
519
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
|
|
511
520
|
continue;
|
|
512
521
|
}
|
|
513
522
|
base.commands[name] = {
|
|
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
|
|
|
519
528
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
520
529
|
durationMs,
|
|
521
530
|
...fileTiming(raw, durationMs),
|
|
531
|
+
fileCount: runnerFileCount(raw),
|
|
522
532
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
523
533
|
// T7: the world this measurement was taken in, so a later reader can ask whether its own world
|
|
524
534
|
// is the same one. Recorded from THIS command's own shell result, never re-derived here.
|
|
525
535
|
...(r.capacity ? { capacity: r.capacity } : {}),
|
|
526
536
|
};
|
|
527
537
|
}
|
|
528
|
-
const names = Object.keys(commands);
|
|
538
|
+
const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
|
|
529
539
|
const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
|
|
530
540
|
if (names.length > 0 && missing.length === names.length) {
|
|
531
541
|
base.warnings = [{
|
|
@@ -609,10 +619,24 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
609
619
|
const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
|
|
610
620
|
const withRerun = rerunOf ? {
|
|
611
621
|
...withReapError,
|
|
612
|
-
details:
|
|
622
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
|
|
613
623
|
meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
|
|
614
624
|
} : withReapError;
|
|
615
|
-
|
|
625
|
+
const final = opts.infraRerun ? {
|
|
626
|
+
...withRerun,
|
|
627
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
|
|
628
|
+
meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
|
|
629
|
+
} : withRerun;
|
|
630
|
+
const withSelected = name === "test" && opts.selected
|
|
631
|
+
? {
|
|
632
|
+
...final,
|
|
633
|
+
meta: {
|
|
634
|
+
...final.meta,
|
|
635
|
+
...(Array.isArray(opts.selected) ? { selectedTests: [...opts.selected] } : {}),
|
|
636
|
+
},
|
|
637
|
+
}
|
|
638
|
+
: final;
|
|
639
|
+
results.push(r.capacity ? { ...withSelected, capacity: r.capacity } : withSelected);
|
|
616
640
|
};
|
|
617
641
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
618
642
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -641,11 +665,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
641
665
|
});
|
|
642
666
|
continue;
|
|
643
667
|
}
|
|
668
|
+
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
669
|
+
const deficit = fileCountDeficit(entry, raw, { name, selected: opts.selected });
|
|
670
|
+
if (deficit) {
|
|
671
|
+
record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
|
|
672
|
+
continue;
|
|
673
|
+
}
|
|
644
674
|
if (r.code === 0) {
|
|
645
675
|
record({ gate: name, pass: true, details: "exit 0" });
|
|
646
676
|
continue;
|
|
647
677
|
}
|
|
648
|
-
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
649
678
|
// OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
|
|
650
679
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
651
680
|
if (runnerVerdict === "green-teardown") {
|
|
@@ -665,16 +694,20 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
665
694
|
// that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
|
|
666
695
|
// `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
|
|
667
696
|
// retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
|
|
668
|
-
const
|
|
669
|
-
|
|
670
|
-
|
|
697
|
+
const classification = classifyFreshRunnerOutput(entry, raw, r.code);
|
|
698
|
+
if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
|
|
699
|
+
const waitedMs = await waitForCalmWindow();
|
|
700
|
+
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
701
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
|
|
702
|
+
continue;
|
|
703
|
+
}
|
|
671
704
|
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
672
705
|
// own baseline measurement. The first read buys one calm rerun here, never a worker repair.
|
|
673
|
-
if (name === "test" && classification !== "infra" && failing.length && !rerunOf
|
|
706
|
+
if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
|
|
674
707
|
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
675
708
|
const waitedMs = await waitForCalmWindow();
|
|
676
709
|
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
677
|
-
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
|
|
710
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
|
|
678
711
|
continue;
|
|
679
712
|
}
|
|
680
713
|
if (classification === "infra") {
|
|
@@ -684,7 +717,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
684
717
|
record({
|
|
685
718
|
gate: name,
|
|
686
719
|
pass: false,
|
|
687
|
-
details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
720
|
+
details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
688
721
|
meta: { classification, infra: true },
|
|
689
722
|
});
|
|
690
723
|
continue;
|
|
@@ -695,9 +728,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
695
728
|
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
696
729
|
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
697
730
|
if (!failing.length && !baselineRed) {
|
|
698
|
-
const recordedCause = entry?.invalidCause === "
|
|
699
|
-
?
|
|
700
|
-
:
|
|
731
|
+
const recordedCause = entry?.invalidCause === "infra"
|
|
732
|
+
? "was invalidated by its recorded runner-infrastructure cause"
|
|
733
|
+
: entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
|
|
734
|
+
? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
|
|
735
|
+
: "was killed at its ceiling";
|
|
701
736
|
const closed = entry?.infra === true
|
|
702
737
|
? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
703
738
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
@@ -742,7 +777,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
742
777
|
return results;
|
|
743
778
|
}
|
|
744
779
|
const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
|
|
745
|
-
const TEARDOWN_RE = /\[
|
|
780
|
+
const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
|
|
746
781
|
const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
|
|
747
782
|
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
748
783
|
export function classifyRunnerOutput(raw, code) {
|
|
@@ -750,6 +785,8 @@ export function classifyRunnerOutput(raw, code) {
|
|
|
750
785
|
return undefined;
|
|
751
786
|
const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
|
|
752
787
|
const text = lines.join("\n");
|
|
788
|
+
if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
|
|
789
|
+
return classifyFailureOutput(text);
|
|
753
790
|
const summary = SUMMARY_LINE_RE.exec(text);
|
|
754
791
|
if (summary) {
|
|
755
792
|
const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
|
|
@@ -780,14 +817,46 @@ export function hostStarved(fresh, durationMs, referenceMs) {
|
|
|
780
817
|
const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
|
|
781
818
|
return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
|
|
782
819
|
}
|
|
783
|
-
const DEFAULT_CALM = {
|
|
820
|
+
const DEFAULT_CALM = {
|
|
821
|
+
pollMs: process.env.VITEST ? 10 : 5_000,
|
|
822
|
+
maxWaitMs: process.env.VITEST ? 50 : 600_000,
|
|
823
|
+
loadProvider: () => loadavg()[0] ?? 0,
|
|
824
|
+
calmLoad: () => availableParallelism() / 2,
|
|
825
|
+
};
|
|
784
826
|
let calm = DEFAULT_CALM;
|
|
785
827
|
export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
|
|
786
828
|
export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
|
|
787
|
-
async function waitForCalmWindow() {
|
|
829
|
+
export async function waitForCalmWindow() {
|
|
788
830
|
const started = Date.now();
|
|
789
831
|
while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
|
|
790
832
|
await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
|
|
791
833
|
}
|
|
792
834
|
return Date.now() - started;
|
|
793
835
|
}
|
|
836
|
+
/** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
|
|
837
|
+
export function runnerFileCount(raw) {
|
|
838
|
+
const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
|
|
839
|
+
const clean = line.replace(ANSI_RE, "");
|
|
840
|
+
const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
|
|
841
|
+
return match ? [Number(match[1])] : [];
|
|
842
|
+
});
|
|
843
|
+
return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
|
|
844
|
+
}
|
|
845
|
+
export function fileCountDeficit(entry, raw, opts) {
|
|
846
|
+
// OBS-985: only a real, named selected-test run of the TEST gate is exempt — a truthy flag or an
|
|
847
|
+
// empty list named no selection and let a full-suite call opt itself out of the deficit guard.
|
|
848
|
+
if (opts?.name === "test" && opts.selected !== undefined && opts.selected.length > 0)
|
|
849
|
+
return undefined;
|
|
850
|
+
const actual = runnerFileCount(raw);
|
|
851
|
+
return entry?.fileCount != null && actual !== null && actual < entry.fileCount
|
|
852
|
+
? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
|
|
853
|
+
: undefined;
|
|
854
|
+
}
|
|
855
|
+
/** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
|
|
856
|
+
export function classifyFreshRunnerOutput(entry, raw, code) {
|
|
857
|
+
if (classifyRunnerOutput(raw, code) === "green-teardown")
|
|
858
|
+
return undefined;
|
|
859
|
+
const { failing } = freshFailures(entry, raw);
|
|
860
|
+
const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
|
|
861
|
+
return verdict === "infra" || verdict === "regression" ? verdict : undefined;
|
|
862
|
+
}
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -65,7 +65,7 @@ export interface LlmRunResult {
|
|
|
65
65
|
seatAuthoredBytes?: number;
|
|
66
66
|
}
|
|
67
67
|
export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
|
|
68
|
-
export declare function reviewSeatOutput(raw: string, nonce: string): string;
|
|
68
|
+
export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
|
|
69
69
|
export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
|
|
70
70
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
71
71
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
package/dist/gates/llm.js
CHANGED
|
@@ -204,12 +204,14 @@ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
|
|
|
204
204
|
// Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
|
|
205
205
|
const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
|
|
206
206
|
// The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
|
|
207
|
-
function seatStart(output) {
|
|
207
|
+
function seatStart(output, adapterBannerRows) {
|
|
208
208
|
const rows = output.split("\n");
|
|
209
209
|
if (rows.length > 1 && rows[rows.length - 1] === "")
|
|
210
210
|
rows.pop(); // the read's own line terminator
|
|
211
211
|
let stage = ECHO;
|
|
212
212
|
let bannerAt; // next banner row expected once the banner has begun
|
|
213
|
+
let adapterBannerAt;
|
|
214
|
+
let identitySeen = false;
|
|
213
215
|
let offset = 0;
|
|
214
216
|
for (let i = 0; i < rows.length; i++) {
|
|
215
217
|
const row = rows[i].replace(/[ \t]+$/, "");
|
|
@@ -217,7 +219,7 @@ function seatStart(output) {
|
|
|
217
219
|
const last = i === rows.length - 1;
|
|
218
220
|
// Complete harness rows: equality against the shape the preamble allows at this stage.
|
|
219
221
|
let accepted = false;
|
|
220
|
-
if (stage < SEAT && t.length === 0)
|
|
222
|
+
if (!identitySeen && stage < SEAT && t.length === 0)
|
|
221
223
|
accepted = true; // blank rows between preamble rows
|
|
222
224
|
else if (stage <= ECHO && completeEchoRow(row))
|
|
223
225
|
accepted = true;
|
|
@@ -225,17 +227,27 @@ function seatStart(output) {
|
|
|
225
227
|
stage = BANNER;
|
|
226
228
|
accepted = true;
|
|
227
229
|
}
|
|
228
|
-
else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
|
|
230
|
+
else if (!identitySeen && stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
|
|
229
231
|
stage = BANNER;
|
|
230
232
|
bannerAt = BANNER_ROWS.indexOf(row) + 1;
|
|
231
233
|
accepted = true;
|
|
232
234
|
}
|
|
233
|
-
else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
|
|
235
|
+
else if (!identitySeen && stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
|
|
234
236
|
bannerAt++;
|
|
235
237
|
accepted = true;
|
|
236
238
|
}
|
|
237
|
-
else if (stage <= IDENTITY &&
|
|
238
|
-
stage =
|
|
239
|
+
else if (stage <= IDENTITY && adapterBannerAt === undefined && adapterBannerRows.includes(row)) {
|
|
240
|
+
stage = BANNER;
|
|
241
|
+
adapterBannerAt = adapterBannerRows.indexOf(row) + 1;
|
|
242
|
+
accepted = true;
|
|
243
|
+
}
|
|
244
|
+
else if (stage <= IDENTITY && adapterBannerAt !== undefined && row === adapterBannerRows[adapterBannerAt]) {
|
|
245
|
+
adapterBannerAt++;
|
|
246
|
+
accepted = true;
|
|
247
|
+
}
|
|
248
|
+
else if (!identitySeen && stage <= IDENTITY && IDENTITY_LINE.test(t)) {
|
|
249
|
+
stage = IDENTITY;
|
|
250
|
+
identitySeen = true;
|
|
239
251
|
accepted = true;
|
|
240
252
|
}
|
|
241
253
|
if (accepted) {
|
|
@@ -253,9 +265,16 @@ function seatStart(output) {
|
|
|
253
265
|
? BANNER_ROWS.some((b) => b.startsWith(row))
|
|
254
266
|
: BANNER_ROWS[bannerAt]?.startsWith(row) === true))
|
|
255
267
|
return -1;
|
|
268
|
+
if (stage <= IDENTITY && (adapterBannerAt === undefined
|
|
269
|
+
? adapterBannerRows.some((b) => b.startsWith(row))
|
|
270
|
+
: adapterBannerRows[adapterBannerAt]?.startsWith(row) === true))
|
|
271
|
+
return -1;
|
|
256
272
|
// The identity row is painted right after the banner, so its prefix is a partial paint only there;
|
|
257
273
|
// with no banner in the capture, "review" or "tick" alone is the seat's own first row.
|
|
258
|
-
if (stage <= IDENTITY
|
|
274
|
+
if (!identitySeen && stage <= IDENTITY
|
|
275
|
+
&& (bannerAt === BANNER_ROWS.length
|
|
276
|
+
|| (adapterBannerRows.length > 0 && adapterBannerAt === adapterBannerRows.length))
|
|
277
|
+
&& IDENTITY_OPENERS.some((o) => o.startsWith(t)))
|
|
259
278
|
return -1;
|
|
260
279
|
return offset;
|
|
261
280
|
}
|
|
@@ -266,9 +285,9 @@ function seatStart(output) {
|
|
|
266
285
|
// Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
|
|
267
286
|
// which is what makes the caller's running Math.max safe: a partial banner counted once would be
|
|
268
287
|
// retained for the whole call and buy a silent seat its full ceiling.
|
|
269
|
-
export function reviewSeatOutput(raw, nonce) {
|
|
288
|
+
export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
|
|
270
289
|
const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
|
|
271
|
-
const start = seatStart(output);
|
|
290
|
+
const start = seatStart(output, adapterBannerRows);
|
|
272
291
|
if (start < 0)
|
|
273
292
|
return "";
|
|
274
293
|
const seat = output.slice(start);
|
|
@@ -287,7 +306,10 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
|
|
|
287
306
|
const pf = join(dir, "prompt.md");
|
|
288
307
|
writeFileSync(pf, prompt);
|
|
289
308
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
290
|
-
|
|
309
|
+
const output = r.stdout + "\n" + r.stderr;
|
|
310
|
+
const nonce = extractPromptNonce(prompt) ?? "";
|
|
311
|
+
return { output, exitCode: r.code, timedOut: r.timedOut === true,
|
|
312
|
+
seatAuthoredBytes: Buffer.byteLength(reviewSeatOutput(output, nonce, adapter.harnessBannerRows).trim()) };
|
|
291
313
|
}
|
|
292
314
|
finally {
|
|
293
315
|
rmSync(dir, { recursive: true, force: true });
|
|
@@ -349,7 +371,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
349
371
|
const startedAt = Date.now();
|
|
350
372
|
out = await via.driver.read(slot, 400);
|
|
351
373
|
const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
|
|
352
|
-
seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
|
|
374
|
+
seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce, adapter.harnessBannerRows));
|
|
353
375
|
let firstLivenessObserved = false;
|
|
354
376
|
let priorSnapshot = normalizeStallSnapshot(out);
|
|
355
377
|
const anchoredAt = Date.now();
|
|
@@ -365,7 +387,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
365
387
|
const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
|
|
366
388
|
const raw = await via.driver.read(slot, 400);
|
|
367
389
|
out = raw;
|
|
368
|
-
seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
|
|
390
|
+
seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce, adapter.harnessBannerRows)));
|
|
369
391
|
// waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
|
|
370
392
|
// wait timed out at the same boundary the marker landed; either way a trailer completes
|
|
371
393
|
// normally and is never mistaken for inactivity.
|
|
@@ -427,7 +449,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
427
449
|
}
|
|
428
450
|
}
|
|
429
451
|
timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
|
|
430
|
-
if (timedOut)
|
|
452
|
+
if (timedOut && reviewing)
|
|
431
453
|
forceClose = true;
|
|
432
454
|
}
|
|
433
455
|
const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -53,6 +53,17 @@ export declare function isDiffCapPark(result: GateResult): boolean;
|
|
|
53
53
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
54
54
|
export declare function modelId(model: string): string;
|
|
55
55
|
export { modelProvider };
|
|
56
|
+
/**
|
|
57
|
+
* One function decides whether a reviewer's `resolved` or `reraised` id names a carried fingerprint,
|
|
58
|
+
* comparing both sides with every whitespace run removed (`s.replace(/\s+/g, "")`).
|
|
59
|
+
*/
|
|
60
|
+
export declare function matchClosureId(candidate: unknown, fingerprint: string): boolean;
|
|
61
|
+
export declare function matchClosureId(candidate: unknown, fingerprints: Iterable<string>): string | undefined;
|
|
62
|
+
/**
|
|
63
|
+
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
64
|
+
* all route through matchClosureId.
|
|
65
|
+
*/
|
|
66
|
+
export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
|
|
56
67
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
57
68
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
58
69
|
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
package/dist/gates/review.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { writeFileSync } from "node:fs";
|
|
1
|
+
import { existsSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
@@ -168,6 +168,31 @@ export function modelId(model) {
|
|
|
168
168
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
169
169
|
}
|
|
170
170
|
export { modelProvider };
|
|
171
|
+
export function matchClosureId(candidate, target) {
|
|
172
|
+
if (typeof candidate !== "string")
|
|
173
|
+
return typeof target === "string" ? false : undefined;
|
|
174
|
+
const normCandidate = candidate.replace(/\s+/g, "");
|
|
175
|
+
if (typeof target === "string") {
|
|
176
|
+
return normCandidate === target.replace(/\s+/g, "");
|
|
177
|
+
}
|
|
178
|
+
for (const fp of target) {
|
|
179
|
+
if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
|
|
180
|
+
return fp;
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
186
|
+
* all route through matchClosureId.
|
|
187
|
+
*/
|
|
188
|
+
export function isReviewClosureInvalid(v, priorIds) {
|
|
189
|
+
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
190
|
+
const closureLists = [v?.resolved, v?.reraised];
|
|
191
|
+
const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
|
|
192
|
+
return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
|
|
193
|
+
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
194
|
+
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
195
|
+
}
|
|
171
196
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
172
197
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
173
198
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -352,7 +377,24 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
|
|
|
352
377
|
Approve iff no material finding remains and every prior material is resolved.
|
|
353
378
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
354
379
|
`;
|
|
355
|
-
|
|
380
|
+
// Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
|
|
381
|
+
// must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
|
|
382
|
+
// make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
|
|
383
|
+
// disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
|
|
384
|
+
// so it can never collide with the attempt it replaces.
|
|
385
|
+
const baseArtifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
|
|
386
|
+
let artifactId = baseArtifactId;
|
|
387
|
+
if (artifactDir) {
|
|
388
|
+
if (existsSync(join(artifactDir, `review-brief-${baseArtifactId}.md`)) ||
|
|
389
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}.txt`))) {
|
|
390
|
+
let counter = 2;
|
|
391
|
+
while (existsSync(join(artifactDir, `review-brief-${baseArtifactId}-${counter}.md`)) ||
|
|
392
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}-${counter}.txt`))) {
|
|
393
|
+
counter++;
|
|
394
|
+
}
|
|
395
|
+
artifactId = `${baseArtifactId}-${counter}`;
|
|
396
|
+
}
|
|
397
|
+
}
|
|
356
398
|
const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
|
|
357
399
|
// Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
|
|
358
400
|
let savedBrief;
|
|
@@ -378,14 +420,21 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
378
420
|
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
379
421
|
cfg.review.timeoutMs);
|
|
380
422
|
const raw = llm.output;
|
|
423
|
+
let saved;
|
|
424
|
+
if (artifactDir) {
|
|
425
|
+
try {
|
|
426
|
+
saved = join(artifactDir, `review-raw-${artifactId}.txt`);
|
|
427
|
+
writeFileSync(saved, redactSecrets(raw));
|
|
428
|
+
}
|
|
429
|
+
catch {
|
|
430
|
+
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
431
|
+
}
|
|
432
|
+
}
|
|
381
433
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
382
434
|
const v = extractVerdictJson(raw, nonce);
|
|
383
435
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
384
436
|
const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
|
|
385
|
-
const
|
|
386
|
-
const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
|
|
387
|
-
|| new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
|
|
388
|
-
|| [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
|
|
437
|
+
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
389
438
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
390
439
|
if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
391
440
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
@@ -394,16 +443,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
394
443
|
const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
|
|
395
444
|
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
396
445
|
: classifyVerdictCause(raw, nonce, "approve", llm);
|
|
397
|
-
let saved;
|
|
398
|
-
if (artifactDir) {
|
|
399
|
-
try {
|
|
400
|
-
saved = join(artifactDir, `review-raw-${artifactId}.txt`);
|
|
401
|
-
writeFileSync(saved, redactSecrets(raw));
|
|
402
|
-
}
|
|
403
|
-
catch {
|
|
404
|
-
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
405
|
-
}
|
|
406
|
-
}
|
|
407
446
|
const failure = cause === "malformed-verdict"
|
|
408
447
|
? "review output unparseable"
|
|
409
448
|
: "review dispatch failed — no structurally valid nonce-bound response";
|
|
@@ -429,7 +468,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
429
468
|
const decided = findings !== null
|
|
430
469
|
? classifyReviewFindings(findings)
|
|
431
470
|
: classifyReviewIssues(v.approve, v.issues);
|
|
432
|
-
const reraised = priorMaterials.filter((finding) => v.reraised?.
|
|
471
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
|
|
433
472
|
if (reraised.length) {
|
|
434
473
|
if (decided.pass)
|
|
435
474
|
decided.headline = "requested changes";
|
|
@@ -450,11 +489,21 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
450
489
|
details,
|
|
451
490
|
meta: {
|
|
452
491
|
...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
|
|
453
|
-
...(priorMaterials.length ? {
|
|
492
|
+
...(priorMaterials.length ? {
|
|
493
|
+
resolved: v.resolved,
|
|
494
|
+
reraised: v.reraised,
|
|
495
|
+
normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
496
|
+
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
497
|
+
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
498
|
+
normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
499
|
+
normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
500
|
+
} : {}),
|
|
454
501
|
...(reraised.length ? { findings: [
|
|
455
502
|
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
456
503
|
...reraised,
|
|
457
504
|
] } : {}),
|
|
505
|
+
...(saved ? { rawPath: saved } : {}),
|
|
506
|
+
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
458
507
|
},
|
|
459
508
|
};
|
|
460
509
|
}
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -366,7 +366,7 @@ export async function runGates(task, ctx) {
|
|
|
366
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
367
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
368
368
|
const finish = startMeasurement();
|
|
369
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
369
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
|
|
370
370
|
const batch = finish();
|
|
371
371
|
for (const g of gates)
|
|
372
372
|
addMeasurement(g, batch);
|
|
@@ -386,7 +386,7 @@ export async function runGates(task, ctx) {
|
|
|
386
386
|
// any later tool before anyone reads its verdict.
|
|
387
387
|
for (const g of gates) {
|
|
388
388
|
await emitStart(g);
|
|
389
|
-
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
|
|
389
|
+
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
|
|
390
390
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
391
|
if (g === "test" && selected)
|
|
392
392
|
selectedDurationMs = spans.get("test").durationMs;
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -14,6 +14,8 @@ export interface RunOptions {
|
|
|
14
14
|
graphChanged?: boolean;
|
|
15
15
|
retryFailed?: boolean;
|
|
16
16
|
concurrency?: number;
|
|
17
|
+
/** Bounded wait at a drain caused solely by parked tasks. */
|
|
18
|
+
approvalWindowMs?: number;
|
|
17
19
|
driver?: ExecutorDriver;
|
|
18
20
|
driverOverride?: DriverChoice;
|
|
19
21
|
adapters?: WorkerAdapter[];
|
|
@@ -55,9 +57,8 @@ export interface RunSummary {
|
|
|
55
57
|
* T14, amended by v2.2 T3: approvals the run accepted and never acted on. `approved` above is still
|
|
56
58
|
* built ONCE at startup — replay determinism depends on it — but a live approval is no longer inert:
|
|
57
59
|
* the boundary sweep in the task loop releases what lands while the daemon runs, so an approval
|
|
58
|
-
* written mid-run is
|
|
59
|
-
* fold still
|
|
60
|
-
* run-end sample below — meets no further boundary, so nothing can release it before this run ends.
|
|
60
|
+
* written mid-run is enacted at a boundary, during the approval window, or by cancelling tip verify.
|
|
61
|
+
* This fold still exposes decisions that could not enact, including a failure before dispatch.
|
|
61
62
|
* Without this the run-end record stated only buckets and tipVerify, both accurate, over a milestone
|
|
62
63
|
* that was silently incomplete: run …230 ended tipVerify "passed" with two upheld approvals and zero
|
|
63
64
|
* subsequent dispatches. Scored per task on its NEWEST approval: a later approval is the live
|
|
@@ -108,6 +109,9 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
|
108
109
|
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
109
110
|
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
110
111
|
export declare const APPROVAL_POLL_MS = 250;
|
|
112
|
+
export declare const APPROVAL_WINDOW_MS = 120000;
|
|
113
|
+
export declare const setApprovalWindowForTests: (ms: number) => void;
|
|
114
|
+
export declare const resetApprovalWindowForTests: () => void;
|
|
111
115
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
112
116
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
113
117
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
@@ -151,6 +155,7 @@ export declare function commandsHash(commands: Record<string, string>): string;
|
|
|
151
155
|
export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
|
|
152
156
|
lastMergedTask?: string;
|
|
153
157
|
baseline?: Baseline;
|
|
158
|
+
signal?: AbortSignal;
|
|
154
159
|
}): Promise<boolean>;
|
|
155
160
|
type SuitePidProbe = (pid: number) => number | undefined;
|
|
156
161
|
/** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
|