tickmarkr 2.5.0 → 2.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/fake.d.ts +1 -0
- package/dist/adapters/fake.js +2 -1
- package/dist/adapters/opencode.d.ts +1 -0
- package/dist/adapters/opencode.js +4 -1
- package/dist/adapters/types.d.ts +14 -0
- package/dist/adapters/types.js +25 -0
- package/dist/cli/commands/approve.js +1 -1
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/init.d.ts +2 -0
- package/dist/cli/commands/init.js +46 -6
- package/dist/cli/commands/plan.js +58 -6
- package/dist/cli/commands/report.js +3 -1
- package/dist/cli/commands/run.d.ts +5 -0
- package/dist/cli/commands/run.js +17 -1
- package/dist/compile/native.js +3 -0
- package/dist/config/config.d.ts +6 -0
- package/dist/config/config.js +6 -0
- package/dist/drivers/subprocess.d.ts +2 -3
- package/dist/drivers/subprocess.js +2 -3
- package/dist/gates/baseline.d.ts +9 -0
- package/dist/gates/baseline.js +85 -29
- package/dist/gates/llm.d.ts +5 -0
- package/dist/gates/llm.js +188 -3
- package/dist/gates/review.d.ts +6 -3
- package/dist/gates/review.js +95 -42
- package/dist/gates/run-gates.d.ts +3 -0
- package/dist/gates/run-gates.js +14 -4
- package/dist/run/daemon.d.ts +8 -4
- package/dist/run/daemon.js +3150 -2839
- package/dist/run/git.d.ts +4 -2
- package/dist/run/git.js +11 -23
- package/dist/run/merge.d.ts +1 -1
- package/dist/run/merge.js +126 -18
- package/dist/tui/ink/init-app.d.ts +4 -0
- package/dist/tui/ink/init-app.js +18 -7
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +16 -2
package/dist/gates/baseline.js
CHANGED
|
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
|
|
|
30
30
|
// to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
|
|
31
31
|
// Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
|
|
32
32
|
// "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
|
|
33
|
-
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
33
|
+
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
34
34
|
const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
|
|
35
35
|
const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
|
|
36
36
|
const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
|
|
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
|
|
|
103
103
|
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
104
104
|
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
105
105
|
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
|
|
106
|
-
// The stripped form
|
|
107
|
-
//
|
|
108
|
-
// reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
|
|
106
|
+
// The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
|
|
107
|
+
// Classification applies the same prefix stripping before its infra/regression vetoes.
|
|
109
108
|
const namesFailureEitherForm = (l) => {
|
|
110
109
|
if (namesFailure(l))
|
|
111
110
|
return true;
|
|
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
|
|
|
163
162
|
// token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
|
|
164
163
|
// to fingerprint(): test-owned output is never runner evidence about the work.
|
|
165
164
|
const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
166
|
-
|
|
165
|
+
const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
|
|
166
|
+
const infra = evidence.some(isInfraLine);
|
|
167
|
+
// A diagnostic section heading names no failing test. It cannot outvote the RPC death
|
|
168
|
+
// beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
|
|
169
|
+
if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
|
|
167
170
|
return "regression";
|
|
168
|
-
return
|
|
171
|
+
return infra ? "infra" : undefined;
|
|
169
172
|
}
|
|
170
173
|
/**
|
|
171
174
|
* Capture validity asks a different question from gate classification. At a gate, one genuine
|
|
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
|
|
|
338
341
|
else if (scripts[name])
|
|
339
342
|
out[name] = `${runPrefix} ${name}`;
|
|
340
343
|
}
|
|
344
|
+
if (cfg.gates.tipTest)
|
|
345
|
+
out.tipTest = cfg.gates.tipTest;
|
|
341
346
|
return out;
|
|
342
347
|
}
|
|
343
348
|
/**
|
|
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
452
457
|
fingerprints: [],
|
|
453
458
|
durationMs,
|
|
454
459
|
fileDurationSumMs: null,
|
|
460
|
+
fileCount: null,
|
|
455
461
|
impliedParallelism: null,
|
|
456
462
|
longestFile: null,
|
|
457
463
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
460
466
|
export async function captureBaseline(cwd, commands) {
|
|
461
467
|
const base = { commands: {} };
|
|
462
468
|
for (const [name, cmd] of Object.entries(commands)) {
|
|
469
|
+
if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
|
|
470
|
+
continue;
|
|
471
|
+
}
|
|
463
472
|
const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
|
|
464
473
|
// ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
|
|
465
474
|
// ponytail: a capture that was itself killed records the ceiling as its "measurement", which
|
|
466
475
|
// scales the next ceiling up — the right direction for a suite that never finished once.
|
|
467
476
|
const durationMs = r.durationMs ?? 0;
|
|
477
|
+
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
478
|
+
const raw = combinedOutput.split(cwd).join("");
|
|
468
479
|
// OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
|
|
469
480
|
// kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
|
|
470
481
|
// exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
|
|
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
|
|
|
483
494
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
484
495
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
485
496
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
486
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
|
|
497
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
|
|
487
498
|
continue;
|
|
488
499
|
}
|
|
489
|
-
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
490
|
-
const raw = combinedOutput.split(cwd).join("");
|
|
491
500
|
// Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
|
|
492
501
|
// discriminator correctly called the mixed output a regression. But the same output also said
|
|
493
502
|
// `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
|
|
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
|
|
|
499
508
|
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
500
509
|
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
|
|
501
510
|
+ `First invalidating line: ${invalidatingLines[0]}`);
|
|
502
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
511
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
|
|
503
512
|
continue;
|
|
504
513
|
}
|
|
505
|
-
// OBS-
|
|
506
|
-
//
|
|
514
|
+
// OBS-966: a worker RPC timeout is infra even beside an all-green summary.
|
|
515
|
+
// Capture and both gate readers share this discriminator; genuine test failures still dominate.
|
|
507
516
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
508
517
|
if (runnerVerdict === "infra") {
|
|
509
|
-
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence
|
|
510
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "infra");
|
|
518
|
+
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
|
|
519
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
|
|
511
520
|
continue;
|
|
512
521
|
}
|
|
513
522
|
base.commands[name] = {
|
|
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
|
|
|
519
528
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
520
529
|
durationMs,
|
|
521
530
|
...fileTiming(raw, durationMs),
|
|
531
|
+
fileCount: runnerFileCount(raw),
|
|
522
532
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
523
533
|
// T7: the world this measurement was taken in, so a later reader can ask whether its own world
|
|
524
534
|
// is the same one. Recorded from THIS command's own shell result, never re-derived here.
|
|
525
535
|
...(r.capacity ? { capacity: r.capacity } : {}),
|
|
526
536
|
};
|
|
527
537
|
}
|
|
528
|
-
const names = Object.keys(commands);
|
|
538
|
+
const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
|
|
529
539
|
const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
|
|
530
540
|
if (names.length > 0 && missing.length === names.length) {
|
|
531
541
|
base.warnings = [{
|
|
@@ -609,10 +619,15 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
609
619
|
const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
|
|
610
620
|
const withRerun = rerunOf ? {
|
|
611
621
|
...withReapError,
|
|
612
|
-
details:
|
|
622
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
|
|
613
623
|
meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
|
|
614
624
|
} : withReapError;
|
|
615
|
-
|
|
625
|
+
const final = opts.infraRerun ? {
|
|
626
|
+
...withRerun,
|
|
627
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
|
|
628
|
+
meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
|
|
629
|
+
} : withRerun;
|
|
630
|
+
results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
|
|
616
631
|
};
|
|
617
632
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
618
633
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -641,11 +656,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
641
656
|
});
|
|
642
657
|
continue;
|
|
643
658
|
}
|
|
659
|
+
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
660
|
+
const deficit = fileCountDeficit(entry, raw);
|
|
661
|
+
if (deficit) {
|
|
662
|
+
record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
|
|
663
|
+
continue;
|
|
664
|
+
}
|
|
644
665
|
if (r.code === 0) {
|
|
645
666
|
record({ gate: name, pass: true, details: "exit 0" });
|
|
646
667
|
continue;
|
|
647
668
|
}
|
|
648
|
-
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
649
669
|
// OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
|
|
650
670
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
651
671
|
if (runnerVerdict === "green-teardown") {
|
|
@@ -665,12 +685,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
665
685
|
// that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
|
|
666
686
|
// `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
|
|
667
687
|
// retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
|
|
668
|
-
const
|
|
669
|
-
|
|
670
|
-
|
|
688
|
+
const classification = classifyFreshRunnerOutput(entry, raw, r.code);
|
|
689
|
+
if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
|
|
690
|
+
const waitedMs = await waitForCalmWindow();
|
|
691
|
+
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
692
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
671
695
|
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
672
696
|
// own baseline measurement. The first read buys one calm rerun here, never a worker repair.
|
|
673
|
-
if (name === "test" && classification !== "infra" && failing.length && !rerunOf
|
|
697
|
+
if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
|
|
674
698
|
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
675
699
|
const waitedMs = await waitForCalmWindow();
|
|
676
700
|
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
@@ -684,7 +708,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
684
708
|
record({
|
|
685
709
|
gate: name,
|
|
686
710
|
pass: false,
|
|
687
|
-
details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
711
|
+
details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
688
712
|
meta: { classification, infra: true },
|
|
689
713
|
});
|
|
690
714
|
continue;
|
|
@@ -695,9 +719,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
695
719
|
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
696
720
|
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
697
721
|
if (!failing.length && !baselineRed) {
|
|
698
|
-
const recordedCause = entry?.invalidCause === "
|
|
699
|
-
?
|
|
700
|
-
:
|
|
722
|
+
const recordedCause = entry?.invalidCause === "infra"
|
|
723
|
+
? "was invalidated by its recorded runner-infrastructure cause"
|
|
724
|
+
: entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
|
|
725
|
+
? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
|
|
726
|
+
: "was killed at its ceiling";
|
|
701
727
|
const closed = entry?.infra === true
|
|
702
728
|
? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
703
729
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
@@ -742,7 +768,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
742
768
|
return results;
|
|
743
769
|
}
|
|
744
770
|
const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
|
|
745
|
-
const TEARDOWN_RE = /\[
|
|
771
|
+
const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
|
|
746
772
|
const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
|
|
747
773
|
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
748
774
|
export function classifyRunnerOutput(raw, code) {
|
|
@@ -750,6 +776,8 @@ export function classifyRunnerOutput(raw, code) {
|
|
|
750
776
|
return undefined;
|
|
751
777
|
const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
|
|
752
778
|
const text = lines.join("\n");
|
|
779
|
+
if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
|
|
780
|
+
return classifyFailureOutput(text);
|
|
753
781
|
const summary = SUMMARY_LINE_RE.exec(text);
|
|
754
782
|
if (summary) {
|
|
755
783
|
const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
|
|
@@ -780,14 +808,42 @@ export function hostStarved(fresh, durationMs, referenceMs) {
|
|
|
780
808
|
const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
|
|
781
809
|
return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
|
|
782
810
|
}
|
|
783
|
-
const DEFAULT_CALM = {
|
|
811
|
+
const DEFAULT_CALM = {
|
|
812
|
+
pollMs: process.env.VITEST ? 10 : 5_000,
|
|
813
|
+
maxWaitMs: process.env.VITEST ? 50 : 600_000,
|
|
814
|
+
loadProvider: () => loadavg()[0] ?? 0,
|
|
815
|
+
calmLoad: () => availableParallelism() / 2,
|
|
816
|
+
};
|
|
784
817
|
let calm = DEFAULT_CALM;
|
|
785
818
|
export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
|
|
786
819
|
export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
|
|
787
|
-
async function waitForCalmWindow() {
|
|
820
|
+
export async function waitForCalmWindow() {
|
|
788
821
|
const started = Date.now();
|
|
789
822
|
while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
|
|
790
823
|
await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
|
|
791
824
|
}
|
|
792
825
|
return Date.now() - started;
|
|
793
826
|
}
|
|
827
|
+
/** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
|
|
828
|
+
export function runnerFileCount(raw) {
|
|
829
|
+
const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
|
|
830
|
+
const clean = line.replace(ANSI_RE, "");
|
|
831
|
+
const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
|
|
832
|
+
return match ? [Number(match[1])] : [];
|
|
833
|
+
});
|
|
834
|
+
return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
|
|
835
|
+
}
|
|
836
|
+
export function fileCountDeficit(entry, raw) {
|
|
837
|
+
const actual = runnerFileCount(raw);
|
|
838
|
+
return entry?.fileCount != null && actual !== null && actual < entry.fileCount
|
|
839
|
+
? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
|
|
840
|
+
: undefined;
|
|
841
|
+
}
|
|
842
|
+
/** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
|
|
843
|
+
export function classifyFreshRunnerOutput(entry, raw, code) {
|
|
844
|
+
if (classifyRunnerOutput(raw, code) === "green-teardown")
|
|
845
|
+
return undefined;
|
|
846
|
+
const { failing } = freshFailures(entry, raw);
|
|
847
|
+
const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
|
|
848
|
+
return verdict === "infra" || verdict === "regression" ? verdict : undefined;
|
|
849
|
+
}
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -61,7 +61,12 @@ export interface LlmRunResult {
|
|
|
61
61
|
output: string;
|
|
62
62
|
exitCode?: number;
|
|
63
63
|
timedOut: boolean;
|
|
64
|
+
launchNeverStarted?: boolean;
|
|
65
|
+
seatAuthoredBytes?: number;
|
|
64
66
|
}
|
|
67
|
+
export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
|
|
68
|
+
export declare function reviewSeatOutput(raw: string, nonce: string): string;
|
|
69
|
+
export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
|
|
65
70
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
66
71
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
67
72
|
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
package/dist/gates/llm.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { stripVTControlCharacters } from "node:util";
|
|
1
2
|
import { AsyncLocalStorage } from "node:async_hooks";
|
|
2
3
|
import { randomBytes } from "node:crypto";
|
|
3
4
|
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
|
@@ -5,7 +6,7 @@ import { tmpdir } from "node:os";
|
|
|
5
6
|
import { join } from "node:path";
|
|
6
7
|
import { matchesTrustDialog } from "../adapters/types.js";
|
|
7
8
|
import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
|
|
8
|
-
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
9
|
+
import { bannerShell, paneDispatchCommand, PLAIN_BANNER } from "../brand.js";
|
|
9
10
|
import { sh } from "../run/git.js";
|
|
10
11
|
import { harvestCpuFlatWindowMs, normalizeStallSnapshot, WorkerTreeCpuAccountant, } from "../run/stall.js";
|
|
11
12
|
export const GATE_PANE_SEP = " · ";
|
|
@@ -129,13 +130,164 @@ export async function captureLlmOutput(run) {
|
|
|
129
130
|
const value = await llmOutputCapture.run(outputs, run);
|
|
130
131
|
return { value, outputs };
|
|
131
132
|
}
|
|
133
|
+
// The harness preamble has a fixed row shape, in order: the dispatch echo rows (the identity export
|
|
134
|
+
// and the START printf, each possibly re-echoed behind the shell prompt), the START acknowledgement
|
|
135
|
+
// row, the banner rows, blank rows, and the identity row. RULING-229-15 add.1 makes the boundary
|
|
136
|
+
// STRUCTURAL: a pane read is line-terminated, so only the LAST row of a capture can be a partial
|
|
137
|
+
// paint. Every row before it is complete, and a complete row is harness only when it EQUALS a full
|
|
138
|
+
// harness row in the position the preamble allows. A complete row that merely starts with "T",
|
|
139
|
+
// "export", "review" or a banner glyph is seat text, and seat prose that MENTIONS a marker mid-row
|
|
140
|
+
// counts in full — "contains TICKMARKR_START_ somewhere" is never by itself a reason to measure zero.
|
|
141
|
+
const BANNER_ROWS = PLAIN_BANNER.replace(/\n$/, "").split("\n");
|
|
142
|
+
const ECHO_OPENER = "export HERDR_WORKSPACE_ID=";
|
|
143
|
+
const ECHO_START_OPENER = "printf '%s%s\\n' 'TICKMARKR_START_' '";
|
|
144
|
+
const ECHO_OPENERS = [ECHO_OPENER, ECHO_START_OPENER];
|
|
145
|
+
// RULING-229-15 add.5: the prompt glyph set is this CLOSED list — the corpus sweeps it, the grammar
|
|
146
|
+
// consumes the LONGEST glyph that matches (so ">>" is one glyph and ">" is still one).
|
|
147
|
+
export const PROMPT_GLYPHS = ["➜", "❯", "$", "%", ">>", ">"];
|
|
148
|
+
const GIT_SEGMENT_OPEN = "git:(";
|
|
149
|
+
function echoRowMatch(row) {
|
|
150
|
+
const opener = (at) => {
|
|
151
|
+
const body = row.slice(at);
|
|
152
|
+
if (ECHO_OPENERS.some((o) => body.startsWith(o)))
|
|
153
|
+
return "complete";
|
|
154
|
+
return ECHO_OPENERS.some((o) => o.startsWith(body)) ? "partial" : "none";
|
|
155
|
+
};
|
|
156
|
+
const ws = (at) => /^\s+/.exec(row.slice(at))?.[0].length;
|
|
157
|
+
const bare = opener(0);
|
|
158
|
+
if (bare !== "none")
|
|
159
|
+
return bare;
|
|
160
|
+
const glyph = PROMPT_GLYPHS.find((g) => row.startsWith(g));
|
|
161
|
+
if (glyph === undefined)
|
|
162
|
+
return "none";
|
|
163
|
+
let i = glyph.length;
|
|
164
|
+
const ws1 = ws(i);
|
|
165
|
+
if (ws1 === undefined)
|
|
166
|
+
return row.length === i ? "partial" : "none";
|
|
167
|
+
i += ws1;
|
|
168
|
+
const dir = /^\S+/.exec(row.slice(i));
|
|
169
|
+
if (!dir)
|
|
170
|
+
return "partial";
|
|
171
|
+
i += dir[0].length;
|
|
172
|
+
const ws2 = ws(i);
|
|
173
|
+
if (ws2 === undefined)
|
|
174
|
+
return "partial";
|
|
175
|
+
i += ws2;
|
|
176
|
+
const rest = row.slice(i);
|
|
177
|
+
if (rest.startsWith(GIT_SEGMENT_OPEN)) {
|
|
178
|
+
const close = row.indexOf(")", i + GIT_SEGMENT_OPEN.length);
|
|
179
|
+
if (close < 0)
|
|
180
|
+
return "partial";
|
|
181
|
+
i = close + 1;
|
|
182
|
+
const ws3 = ws(i);
|
|
183
|
+
if (ws3 === undefined)
|
|
184
|
+
return row.length === i ? "partial" : "none";
|
|
185
|
+
return opener(i + ws3);
|
|
186
|
+
}
|
|
187
|
+
if (GIT_SEGMENT_OPEN.startsWith(rest))
|
|
188
|
+
return "partial"; // "", "g", "gi", "git", "git:"
|
|
189
|
+
return opener(i);
|
|
190
|
+
}
|
|
191
|
+
// A complete dispatch echo row: the grammar reaches the opener. Never "contains the opener" — a row
|
|
192
|
+
// that mentions it mid-prose is the seat's.
|
|
193
|
+
function completeEchoRow(row) {
|
|
194
|
+
return echoRowMatch(row) === "complete";
|
|
195
|
+
}
|
|
196
|
+
// A LAST row still being painted: any prefix of a row in the grammar.
|
|
197
|
+
function partialEchoRow(row) {
|
|
198
|
+
return echoRowMatch(row) !== "none";
|
|
199
|
+
}
|
|
200
|
+
const START_MARKER = "TICKMARKR_START_";
|
|
201
|
+
const START_ROW = /^TICKMARKR_START_[\w-]+$/;
|
|
202
|
+
const IDENTITY_LINE = /^(?:review\s*·|tickmarkr(?::|$))/;
|
|
203
|
+
const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
|
|
204
|
+
// Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
|
|
205
|
+
const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
|
|
206
|
+
// The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
|
|
207
|
+
function seatStart(output) {
|
|
208
|
+
const rows = output.split("\n");
|
|
209
|
+
if (rows.length > 1 && rows[rows.length - 1] === "")
|
|
210
|
+
rows.pop(); // the read's own line terminator
|
|
211
|
+
let stage = ECHO;
|
|
212
|
+
let bannerAt; // next banner row expected once the banner has begun
|
|
213
|
+
let offset = 0;
|
|
214
|
+
for (let i = 0; i < rows.length; i++) {
|
|
215
|
+
const row = rows[i].replace(/[ \t]+$/, "");
|
|
216
|
+
const t = row.trim();
|
|
217
|
+
const last = i === rows.length - 1;
|
|
218
|
+
// Complete harness rows: equality against the shape the preamble allows at this stage.
|
|
219
|
+
let accepted = false;
|
|
220
|
+
if (stage < SEAT && t.length === 0)
|
|
221
|
+
accepted = true; // blank rows between preamble rows
|
|
222
|
+
else if (stage <= ECHO && completeEchoRow(row))
|
|
223
|
+
accepted = true;
|
|
224
|
+
else if (stage <= START && START_ROW.test(row)) {
|
|
225
|
+
stage = BANNER;
|
|
226
|
+
accepted = true;
|
|
227
|
+
}
|
|
228
|
+
else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
|
|
229
|
+
stage = BANNER;
|
|
230
|
+
bannerAt = BANNER_ROWS.indexOf(row) + 1;
|
|
231
|
+
accepted = true;
|
|
232
|
+
}
|
|
233
|
+
else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
|
|
234
|
+
bannerAt++;
|
|
235
|
+
accepted = true;
|
|
236
|
+
}
|
|
237
|
+
else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
|
|
238
|
+
stage = SEAT;
|
|
239
|
+
accepted = true;
|
|
240
|
+
}
|
|
241
|
+
if (accepted) {
|
|
242
|
+
offset += rows[i].length + 1;
|
|
243
|
+
continue;
|
|
244
|
+
}
|
|
245
|
+
if (!last)
|
|
246
|
+
return offset;
|
|
247
|
+
// The last row may be a partial paint: a prefix of the next harness row the preamble allows.
|
|
248
|
+
if (stage <= ECHO && partialEchoRow(row))
|
|
249
|
+
return -1;
|
|
250
|
+
if (stage <= START && (START_MARKER.startsWith(row) || /^TICKMARKR_START_[\w-]*$/.test(row)))
|
|
251
|
+
return -1;
|
|
252
|
+
if (stage <= BANNER && (bannerAt === undefined
|
|
253
|
+
? BANNER_ROWS.some((b) => b.startsWith(row))
|
|
254
|
+
: BANNER_ROWS[bannerAt]?.startsWith(row) === true))
|
|
255
|
+
return -1;
|
|
256
|
+
// The identity row is painted right after the banner, so its prefix is a partial paint only there;
|
|
257
|
+
// with no banner in the capture, "review" or "tick" alone is the seat's own first row.
|
|
258
|
+
if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
|
|
259
|
+
return -1;
|
|
260
|
+
return offset;
|
|
261
|
+
}
|
|
262
|
+
return -1; // every row was harness — the seat has not taken its turn
|
|
263
|
+
}
|
|
264
|
+
// A pane's dispatch echo, start acknowledgement, banner and identity are harness bytes.
|
|
265
|
+
// Remove only that leading preamble, never matching text later in the seat's response.
|
|
266
|
+
// Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
|
|
267
|
+
// which is what makes the caller's running Math.max safe: a partial banner counted once would be
|
|
268
|
+
// retained for the whole call and buy a silent seat its full ceiling.
|
|
269
|
+
export function reviewSeatOutput(raw, nonce) {
|
|
270
|
+
const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
|
|
271
|
+
const start = seatStart(output);
|
|
272
|
+
if (start < 0)
|
|
273
|
+
return "";
|
|
274
|
+
const seat = output.slice(start);
|
|
275
|
+
// Anything after the nonce-bound trailer belongs to the terminal (typically the next shell
|
|
276
|
+
// prompt), not the reviewer. Deleting only the marker would count that postlude as seat output and
|
|
277
|
+
// let a zero-byte seat escape first-silence demotion.
|
|
278
|
+
const trailer = new RegExp(`(?:^|\\n)TICKMARKR_EXIT_${nonce}:\\d+[^\\n]*(?:\\n|$)`).exec(seat);
|
|
279
|
+
// A trailer row still being typed is not stripped: after the preamble a last row "T" is far more
|
|
280
|
+
// often the seat's first byte than the harness's exit marker, and the next read completes either.
|
|
281
|
+
return trailer ? seat.slice(0, trailer.index) : seat;
|
|
282
|
+
}
|
|
283
|
+
export const REVIEW_FIRST_LIVENESS_MS = 30_000;
|
|
132
284
|
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
133
285
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
134
286
|
try {
|
|
135
287
|
const pf = join(dir, "prompt.md");
|
|
136
288
|
writeFileSync(pf, prompt);
|
|
137
289
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
138
|
-
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
|
|
290
|
+
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
|
|
139
291
|
}
|
|
140
292
|
finally {
|
|
141
293
|
rmSync(dir, { recursive: true, force: true });
|
|
@@ -150,6 +302,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
150
302
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
151
303
|
let slot;
|
|
152
304
|
let accountant;
|
|
305
|
+
let forceClose = false;
|
|
153
306
|
try {
|
|
154
307
|
const pf = join(dir, "prompt.md");
|
|
155
308
|
writeFileSync(pf, prompt);
|
|
@@ -180,6 +333,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
180
333
|
const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
|
|
181
334
|
let out;
|
|
182
335
|
let timedOut = false;
|
|
336
|
+
let launchNeverStarted = false;
|
|
337
|
+
let seatAuthoredBytes = 0;
|
|
183
338
|
const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
|
|
184
339
|
if (!gatePrompt) {
|
|
185
340
|
await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
|
|
@@ -193,6 +348,9 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
193
348
|
await accountant.start();
|
|
194
349
|
const startedAt = Date.now();
|
|
195
350
|
out = await via.driver.read(slot, 400);
|
|
351
|
+
const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
|
|
352
|
+
seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
|
|
353
|
+
let firstLivenessObserved = false;
|
|
196
354
|
let priorSnapshot = normalizeStallSnapshot(out);
|
|
197
355
|
const anchoredAt = Date.now();
|
|
198
356
|
let quietSince = anchoredAt;
|
|
@@ -207,12 +365,35 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
207
365
|
const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
|
|
208
366
|
const raw = await via.driver.read(slot, 400);
|
|
209
367
|
out = raw;
|
|
368
|
+
seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
|
|
210
369
|
// waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
|
|
211
370
|
// wait timed out at the same boundary the marker landed; either way a trailer completes
|
|
212
371
|
// normally and is never mistaken for inactivity.
|
|
213
372
|
if (matched || new RegExp(exitPattern).test(raw))
|
|
214
373
|
break;
|
|
215
374
|
const now = Date.now();
|
|
375
|
+
// The ceiling wins if it coincides with the first beat (or the read crosses it).
|
|
376
|
+
// That seat was killed by its configured timeout, not an early launch reroute.
|
|
377
|
+
if (now - startedAt >= timeoutMs)
|
|
378
|
+
break;
|
|
379
|
+
if (reviewing && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
|
|
380
|
+
firstLivenessObserved = true;
|
|
381
|
+
// RS-2: the beat reads seat-authored bytes ALONE. CPU evidence never holds a preamble-only
|
|
382
|
+
// capture open to the ceiling; a seat that has not written one byte of its own is re-routed.
|
|
383
|
+
// OBS-944 is why a buffering runner earns no exemption here: the claude-code seat's 901 s
|
|
384
|
+
// capture was byte-identical across two legs and ended at the pane-identity line — that
|
|
385
|
+
// seat never started, it was not quietly working. RULING-229-06 puts the beat on the PANE
|
|
386
|
+
// path only; a headless `-p` runner (runHeadlessDetailed, no pane, no beat) buffers every
|
|
387
|
+
// byte until completion and keeps its full ceiling.
|
|
388
|
+
if (seatAuthoredBytes === 0) {
|
|
389
|
+
launchNeverStarted = true;
|
|
390
|
+
forceClose = true;
|
|
391
|
+
break;
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
// Producing reviews own their full ceiling; inactivity is not a review verdict.
|
|
395
|
+
if (reviewing)
|
|
396
|
+
continue;
|
|
216
397
|
const snapshot = normalizeStallSnapshot(raw);
|
|
217
398
|
if (snapshot !== priorSnapshot) {
|
|
218
399
|
priorSnapshot = snapshot;
|
|
@@ -246,18 +427,22 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
246
427
|
}
|
|
247
428
|
}
|
|
248
429
|
timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
|
|
430
|
+
if (timedOut && reviewing)
|
|
431
|
+
forceClose = true;
|
|
249
432
|
}
|
|
250
433
|
const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
|
|
251
434
|
return {
|
|
252
435
|
output: dewrapPaneVerdict(out, nonce),
|
|
253
436
|
...(Number.isFinite(exitCode) ? { exitCode } : {}),
|
|
254
437
|
timedOut,
|
|
438
|
+
launchNeverStarted,
|
|
439
|
+
seatAuthoredBytes,
|
|
255
440
|
};
|
|
256
441
|
}
|
|
257
442
|
finally {
|
|
258
443
|
try {
|
|
259
444
|
await accountant?.stop();
|
|
260
|
-
if (slot && !via.keep)
|
|
445
|
+
if (slot && (forceClose || !via.keep))
|
|
261
446
|
await via.driver.close(slot);
|
|
262
447
|
}
|
|
263
448
|
finally {
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type TickmarkrConfig, type Tier } from "../config/config.js";
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
|
+
import { type StructuredFinding } from "../run/journal.js";
|
|
4
5
|
import { modelProvider } from "../route/preference.js";
|
|
5
6
|
import { type GateVia } from "./llm.js";
|
|
6
7
|
import type { GateResult } from "./types.js";
|
|
@@ -16,6 +17,8 @@ export interface ReviewFinding {
|
|
|
16
17
|
}
|
|
17
18
|
export interface ReviewVerdict {
|
|
18
19
|
approve?: boolean;
|
|
20
|
+
resolved?: string[];
|
|
21
|
+
reraised?: string[];
|
|
19
22
|
issues?: string[];
|
|
20
23
|
findings?: ReviewFinding[];
|
|
21
24
|
comments?: Array<{
|
|
@@ -54,12 +57,12 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
|
|
|
54
57
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
55
58
|
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
|
56
59
|
history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
57
|
-
onSeat?: (seat: number) => void): BillingChannel | null;
|
|
58
|
-
export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
60
|
+
onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
|
|
61
|
+
export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
|
|
59
62
|
/**
|
|
60
63
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
61
64
|
* remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
|
|
62
65
|
* judgement rather than a guarantee made by this renderer.
|
|
63
66
|
*/
|
|
64
67
|
export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
|
|
65
|
-
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
|
|
68
|
+
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[]): Promise<GateResult>;
|