tickmarkr 2.5.1 → 2.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/adapters/qwen.d.ts +1 -0
  2. package/dist/adapters/qwen.js +8 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/cli/commands/approve.js +1 -1
  5. package/dist/cli/commands/doctor.d.ts +8 -0
  6. package/dist/cli/commands/doctor.js +50 -1
  7. package/dist/cli/commands/init.js +1 -1
  8. package/dist/cli/commands/run.d.ts +5 -0
  9. package/dist/cli/commands/run.js +16 -1
  10. package/dist/cli/commands/status.js +8 -14
  11. package/dist/compile/native.js +3 -0
  12. package/dist/compile/ownership.d.ts +9 -0
  13. package/dist/compile/ownership.js +65 -2
  14. package/dist/config/config.d.ts +6 -0
  15. package/dist/config/config.js +6 -0
  16. package/dist/drivers/subprocess.d.ts +2 -3
  17. package/dist/drivers/subprocess.js +2 -3
  18. package/dist/gates/baseline.d.ts +13 -0
  19. package/dist/gates/baseline.js +99 -30
  20. package/dist/gates/llm.d.ts +1 -1
  21. package/dist/gates/llm.js +35 -13
  22. package/dist/gates/review.d.ts +11 -0
  23. package/dist/gates/review.js +67 -18
  24. package/dist/gates/run-gates.js +2 -2
  25. package/dist/run/daemon.d.ts +8 -3
  26. package/dist/run/daemon.js +3198 -2934
  27. package/dist/run/git.d.ts +4 -2
  28. package/dist/run/git.js +11 -23
  29. package/dist/run/journal.js +16 -7
  30. package/dist/run/merge.d.ts +1 -1
  31. package/dist/run/merge.js +126 -18
  32. package/dist/run/operator-state.d.ts +1 -1
  33. package/dist/run/operator-state.js +5 -9
  34. package/package.json +1 -1
  35. package/skills/tickmarkr-auto/SKILL.md +1 -1
  36. package/skills/tickmarkr-loop/SKILL.md +1 -1
  37. package/skills/tickmarkr-overseer/SKILL.md +61 -4
  38. package/skills/tickmarkr-overseer/scripts/watch-context.sh +5 -2
  39. package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
30
30
  // to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
31
31
  // Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
32
32
  // "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
33
- const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
33
+ const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
34
34
  const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
35
35
  const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
36
36
  const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
103
103
  // Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
104
104
  // and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
105
105
  const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
106
- // The stripped form is a second READ of the same line, for the recognition/headline paths that ask
107
- // "does anything here name a failure" — verdict classification (isInfraLine/namesRegression) keeps
108
- // reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
106
+ // The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
107
+ // Classification applies the same prefix stripping before its infra/regression vetoes.
109
108
  const namesFailureEitherForm = (l) => {
110
109
  if (namesFailure(l))
111
110
  return true;
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
163
162
  // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
164
163
  // to fingerprint(): test-owned output is never runner evidence about the work.
165
164
  const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
166
- if (lines.some(namesRegression))
165
+ const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
166
+ const infra = evidence.some(isInfraLine);
167
+ // A diagnostic section heading names no failing test. It cannot outvote the RPC death
168
+ // beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
169
+ if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
167
170
  return "regression";
168
- return lines.some(isInfraLine) ? "infra" : undefined;
171
+ return infra ? "infra" : undefined;
169
172
  }
170
173
  /**
171
174
  * Capture validity asks a different question from gate classification. At a gate, one genuine
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
338
341
  else if (scripts[name])
339
342
  out[name] = `${runPrefix} ${name}`;
340
343
  }
344
+ if (cfg.gates.tipTest)
345
+ out.tipTest = cfg.gates.tipTest;
341
346
  return out;
342
347
  }
343
348
  /**
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
452
457
  fingerprints: [],
453
458
  durationMs,
454
459
  fileDurationSumMs: null,
460
+ fileCount: null,
455
461
  impliedParallelism: null,
456
462
  longestFile: null,
457
463
  ceilingMs: effectiveCeilingMs({ durationMs }),
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
460
466
  export async function captureBaseline(cwd, commands) {
461
467
  const base = { commands: {} };
462
468
  for (const [name, cmd] of Object.entries(commands)) {
469
+ if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
470
+ continue;
471
+ }
463
472
  const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
464
473
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
465
474
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
466
475
  // scales the next ceiling up — the right direction for a suite that never finished once.
467
476
  const durationMs = r.durationMs ?? 0;
477
+ const combinedOutput = r.stdout + "\n" + r.stderr;
478
+ const raw = combinedOutput.split(cwd).join("");
468
479
  // OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
469
480
  // kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
470
481
  // exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
483
494
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
484
495
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
485
496
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
486
- base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
497
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
487
498
  continue;
488
499
  }
489
- const combinedOutput = r.stdout + "\n" + r.stderr;
490
- const raw = combinedOutput.split(cwd).join("");
491
500
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
492
501
  // discriminator correctly called the mixed output a regression. But the same output also said
493
502
  // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
499
508
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
500
509
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
501
510
  + `First invalidating line: ${invalidatingLines[0]}`);
502
- base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
511
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
503
512
  continue;
504
513
  }
505
- // OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
506
- // teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
514
+ // OBS-966: a worker RPC timeout is infra even beside an all-green summary.
515
+ // Capture and both gate readers share this discriminator; genuine test failures still dominate.
507
516
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
508
517
  if (runnerVerdict === "infra") {
509
- console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
510
- base.commands[name] = invalidCaptureEntry(durationMs, "infra");
518
+ console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
519
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
511
520
  continue;
512
521
  }
513
522
  base.commands[name] = {
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
519
528
  missingCommand: missingConfiguredCommand(cmd, r),
520
529
  durationMs,
521
530
  ...fileTiming(raw, durationMs),
531
+ fileCount: runnerFileCount(raw),
522
532
  ceilingMs: effectiveCeilingMs({ durationMs }),
523
533
  // T7: the world this measurement was taken in, so a later reader can ask whether its own world
524
534
  // is the same one. Recorded from THIS command's own shell result, never re-derived here.
525
535
  ...(r.capacity ? { capacity: r.capacity } : {}),
526
536
  };
527
537
  }
528
- const names = Object.keys(commands);
538
+ const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
529
539
  const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
530
540
  if (names.length > 0 && missing.length === names.length) {
531
541
  base.warnings = [{
@@ -609,10 +619,24 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
609
619
  const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
610
620
  const withRerun = rerunOf ? {
611
621
  ...withReapError,
612
- details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
622
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
613
623
  meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
614
624
  } : withReapError;
615
- results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
625
+ const final = opts.infraRerun ? {
626
+ ...withRerun,
627
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
628
+ meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
629
+ } : withRerun;
630
+ const withSelected = name === "test" && opts.selected
631
+ ? {
632
+ ...final,
633
+ meta: {
634
+ ...final.meta,
635
+ ...(Array.isArray(opts.selected) ? { selectedTests: [...opts.selected] } : {}),
636
+ },
637
+ }
638
+ : final;
639
+ results.push(r.capacity ? { ...withSelected, capacity: r.capacity } : withSelected);
616
640
  };
617
641
  // …and whether the entry that would forgive this command was measured in the same world. A
618
642
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -641,11 +665,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
641
665
  });
642
666
  continue;
643
667
  }
668
+ const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
669
+ const deficit = fileCountDeficit(entry, raw, { name, selected: opts.selected });
670
+ if (deficit) {
671
+ record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
672
+ continue;
673
+ }
644
674
  if (r.code === 0) {
645
675
  record({ gate: name, pass: true, details: "exit 0" });
646
676
  continue;
647
677
  }
648
- const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
649
678
  // OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
650
679
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
651
680
  if (runnerVerdict === "green-teardown") {
@@ -665,16 +694,20 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
665
694
  // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
666
695
  // `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
667
696
  // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
668
- const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
669
- const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
670
- const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
697
+ const classification = classifyFreshRunnerOutput(entry, raw, r.code);
698
+ if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
699
+ const waitedMs = await waitForCalmWindow();
700
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
701
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
702
+ continue;
703
+ }
671
704
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
672
705
  // own baseline measurement. The first read buys one calm rerun here, never a worker repair.
673
- if (name === "test" && classification !== "infra" && failing.length && !rerunOf
706
+ if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
674
707
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
675
708
  const waitedMs = await waitForCalmWindow();
676
709
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
677
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
710
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
678
711
  continue;
679
712
  }
680
713
  if (classification === "infra") {
@@ -684,7 +717,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
684
717
  record({
685
718
  gate: name,
686
719
  pass: false,
687
- details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
720
+ details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
688
721
  meta: { classification, infra: true },
689
722
  });
690
723
  continue;
@@ -695,9 +728,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
695
728
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
696
729
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
697
730
  if (!failing.length && !baselineRed) {
698
- const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
699
- ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
700
- : "was killed at its ceiling";
731
+ const recordedCause = entry?.invalidCause === "infra"
732
+ ? "was invalidated by its recorded runner-infrastructure cause"
733
+ : entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
734
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
735
+ : "was killed at its ceiling";
701
736
  const closed = entry?.infra === true
702
737
  ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
703
738
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
@@ -742,7 +777,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
742
777
  return results;
743
778
  }
744
779
  const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
745
- const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
780
+ const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
746
781
  const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
747
782
  /** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
748
783
  export function classifyRunnerOutput(raw, code) {
@@ -750,6 +785,8 @@ export function classifyRunnerOutput(raw, code) {
750
785
  return undefined;
751
786
  const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
752
787
  const text = lines.join("\n");
788
+ if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
789
+ return classifyFailureOutput(text);
753
790
  const summary = SUMMARY_LINE_RE.exec(text);
754
791
  if (summary) {
755
792
  const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
@@ -780,14 +817,46 @@ export function hostStarved(fresh, durationMs, referenceMs) {
780
817
  const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
781
818
  return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
782
819
  }
783
- const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
820
+ const DEFAULT_CALM = {
821
+ pollMs: process.env.VITEST ? 10 : 5_000,
822
+ maxWaitMs: process.env.VITEST ? 50 : 600_000,
823
+ loadProvider: () => loadavg()[0] ?? 0,
824
+ calmLoad: () => availableParallelism() / 2,
825
+ };
784
826
  let calm = DEFAULT_CALM;
785
827
  export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
786
828
  export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
787
- async function waitForCalmWindow() {
829
+ export async function waitForCalmWindow() {
788
830
  const started = Date.now();
789
831
  while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
790
832
  await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
791
833
  }
792
834
  return Date.now() - started;
793
835
  }
836
+ /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
837
+ export function runnerFileCount(raw) {
838
+ const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
839
+ const clean = line.replace(ANSI_RE, "");
840
+ const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
841
+ return match ? [Number(match[1])] : [];
842
+ });
843
+ return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
844
+ }
845
+ export function fileCountDeficit(entry, raw, opts) {
846
+ // OBS-985: only a real, named selected-test run of the TEST gate is exempt — a truthy flag or an
847
+ // empty list named no selection and let a full-suite call opt itself out of the deficit guard.
848
+ if (opts?.name === "test" && opts.selected !== undefined && opts.selected.length > 0)
849
+ return undefined;
850
+ const actual = runnerFileCount(raw);
851
+ return entry?.fileCount != null && actual !== null && actual < entry.fileCount
852
+ ? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
853
+ : undefined;
854
+ }
855
+ /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
856
+ export function classifyFreshRunnerOutput(entry, raw, code) {
857
+ if (classifyRunnerOutput(raw, code) === "green-teardown")
858
+ return undefined;
859
+ const { failing } = freshFailures(entry, raw);
860
+ const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
861
+ return verdict === "infra" || verdict === "regression" ? verdict : undefined;
862
+ }
@@ -65,7 +65,7 @@ export interface LlmRunResult {
65
65
  seatAuthoredBytes?: number;
66
66
  }
67
67
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
- export declare function reviewSeatOutput(raw: string, nonce: string): string;
68
+ export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
69
69
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
70
70
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
71
71
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
package/dist/gates/llm.js CHANGED
@@ -204,12 +204,14 @@ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
204
204
  // Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
205
205
  const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
206
206
  // The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
207
- function seatStart(output) {
207
+ function seatStart(output, adapterBannerRows) {
208
208
  const rows = output.split("\n");
209
209
  if (rows.length > 1 && rows[rows.length - 1] === "")
210
210
  rows.pop(); // the read's own line terminator
211
211
  let stage = ECHO;
212
212
  let bannerAt; // next banner row expected once the banner has begun
213
+ let adapterBannerAt;
214
+ let identitySeen = false;
213
215
  let offset = 0;
214
216
  for (let i = 0; i < rows.length; i++) {
215
217
  const row = rows[i].replace(/[ \t]+$/, "");
@@ -217,7 +219,7 @@ function seatStart(output) {
217
219
  const last = i === rows.length - 1;
218
220
  // Complete harness rows: equality against the shape the preamble allows at this stage.
219
221
  let accepted = false;
220
- if (stage < SEAT && t.length === 0)
222
+ if (!identitySeen && stage < SEAT && t.length === 0)
221
223
  accepted = true; // blank rows between preamble rows
222
224
  else if (stage <= ECHO && completeEchoRow(row))
223
225
  accepted = true;
@@ -225,17 +227,27 @@ function seatStart(output) {
225
227
  stage = BANNER;
226
228
  accepted = true;
227
229
  }
228
- else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
230
+ else if (!identitySeen && stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
229
231
  stage = BANNER;
230
232
  bannerAt = BANNER_ROWS.indexOf(row) + 1;
231
233
  accepted = true;
232
234
  }
233
- else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
235
+ else if (!identitySeen && stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
234
236
  bannerAt++;
235
237
  accepted = true;
236
238
  }
237
- else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
238
- stage = SEAT;
239
+ else if (stage <= IDENTITY && adapterBannerAt === undefined && adapterBannerRows.includes(row)) {
240
+ stage = BANNER;
241
+ adapterBannerAt = adapterBannerRows.indexOf(row) + 1;
242
+ accepted = true;
243
+ }
244
+ else if (stage <= IDENTITY && adapterBannerAt !== undefined && row === adapterBannerRows[adapterBannerAt]) {
245
+ adapterBannerAt++;
246
+ accepted = true;
247
+ }
248
+ else if (!identitySeen && stage <= IDENTITY && IDENTITY_LINE.test(t)) {
249
+ stage = IDENTITY;
250
+ identitySeen = true;
239
251
  accepted = true;
240
252
  }
241
253
  if (accepted) {
@@ -253,9 +265,16 @@ function seatStart(output) {
253
265
  ? BANNER_ROWS.some((b) => b.startsWith(row))
254
266
  : BANNER_ROWS[bannerAt]?.startsWith(row) === true))
255
267
  return -1;
268
+ if (stage <= IDENTITY && (adapterBannerAt === undefined
269
+ ? adapterBannerRows.some((b) => b.startsWith(row))
270
+ : adapterBannerRows[adapterBannerAt]?.startsWith(row) === true))
271
+ return -1;
256
272
  // The identity row is painted right after the banner, so its prefix is a partial paint only there;
257
273
  // with no banner in the capture, "review" or "tick" alone is the seat's own first row.
258
- if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
274
+ if (!identitySeen && stage <= IDENTITY
275
+ && (bannerAt === BANNER_ROWS.length
276
+ || (adapterBannerRows.length > 0 && adapterBannerAt === adapterBannerRows.length))
277
+ && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
259
278
  return -1;
260
279
  return offset;
261
280
  }
@@ -266,9 +285,9 @@ function seatStart(output) {
266
285
  // Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
267
286
  // which is what makes the caller's running Math.max safe: a partial banner counted once would be
268
287
  // retained for the whole call and buy a silent seat its full ceiling.
269
- export function reviewSeatOutput(raw, nonce) {
288
+ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
270
289
  const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
271
- const start = seatStart(output);
290
+ const start = seatStart(output, adapterBannerRows);
272
291
  if (start < 0)
273
292
  return "";
274
293
  const seat = output.slice(start);
@@ -287,7 +306,10 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
287
306
  const pf = join(dir, "prompt.md");
288
307
  writeFileSync(pf, prompt);
289
308
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
290
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength(r.stdout + r.stderr) };
309
+ const output = r.stdout + "\n" + r.stderr;
310
+ const nonce = extractPromptNonce(prompt) ?? "";
311
+ return { output, exitCode: r.code, timedOut: r.timedOut === true,
312
+ seatAuthoredBytes: Buffer.byteLength(reviewSeatOutput(output, nonce, adapter.harnessBannerRows).trim()) };
291
313
  }
292
314
  finally {
293
315
  rmSync(dir, { recursive: true, force: true });
@@ -349,7 +371,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
349
371
  const startedAt = Date.now();
350
372
  out = await via.driver.read(slot, 400);
351
373
  const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
352
- seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
374
+ seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce, adapter.harnessBannerRows));
353
375
  let firstLivenessObserved = false;
354
376
  let priorSnapshot = normalizeStallSnapshot(out);
355
377
  const anchoredAt = Date.now();
@@ -365,7 +387,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
365
387
  const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
366
388
  const raw = await via.driver.read(slot, 400);
367
389
  out = raw;
368
- seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
390
+ seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce, adapter.harnessBannerRows)));
369
391
  // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
370
392
  // wait timed out at the same boundary the marker landed; either way a trailer completes
371
393
  // normally and is never mistaken for inactivity.
@@ -427,7 +449,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
427
449
  }
428
450
  }
429
451
  timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
430
- if (timedOut)
452
+ if (timedOut && reviewing)
431
453
  forceClose = true;
432
454
  }
433
455
  const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
@@ -53,6 +53,17 @@ export declare function isDiffCapPark(result: GateResult): boolean;
53
53
  export declare function diffCapParkReason(results: GateResult[]): string | null;
54
54
  export declare function modelId(model: string): string;
55
55
  export { modelProvider };
56
+ /**
57
+ * One function decides whether a reviewer's `resolved` or `reraised` id names a carried fingerprint,
58
+ * comparing both sides with every whitespace run removed (`s.replace(/\s+/g, "")`).
59
+ */
60
+ export declare function matchClosureId(candidate: unknown, fingerprint: string): boolean;
61
+ export declare function matchClosureId(candidate: unknown, fingerprints: Iterable<string>): string | undefined;
62
+ /**
63
+ * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
64
+ * all route through matchClosureId.
65
+ */
66
+ export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
56
67
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
57
68
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
58
69
  floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
@@ -1,4 +1,4 @@
1
- import { writeFileSync } from "node:fs";
1
+ import { existsSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
@@ -168,6 +168,31 @@ export function modelId(model) {
168
168
  return model.slice(model.lastIndexOf("/") + 1);
169
169
  }
170
170
  export { modelProvider };
171
+ export function matchClosureId(candidate, target) {
172
+ if (typeof candidate !== "string")
173
+ return typeof target === "string" ? false : undefined;
174
+ const normCandidate = candidate.replace(/\s+/g, "");
175
+ if (typeof target === "string") {
176
+ return normCandidate === target.replace(/\s+/g, "");
177
+ }
178
+ for (const fp of target) {
179
+ if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
180
+ return fp;
181
+ }
182
+ return undefined;
183
+ }
184
+ /**
185
+ * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
186
+ * all route through matchClosureId.
187
+ */
188
+ export function isReviewClosureInvalid(v, priorIds) {
189
+ const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
190
+ const closureLists = [v?.resolved, v?.reraised];
191
+ const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
192
+ return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
193
+ || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
194
+ || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
195
+ }
171
196
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
172
197
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
173
198
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -352,7 +377,24 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
352
377
  Approve iff no material finding remains and every prior material is resolved.
353
378
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
354
379
  `;
355
- const artifactId = `${task.id}-${nonce}`;
380
+ // Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
381
+ // must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
382
+ // make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
383
+ // disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
384
+ // so it can never collide with the attempt it replaces.
385
+ const baseArtifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
386
+ let artifactId = baseArtifactId;
387
+ if (artifactDir) {
388
+ if (existsSync(join(artifactDir, `review-brief-${baseArtifactId}.md`)) ||
389
+ existsSync(join(artifactDir, `review-raw-${baseArtifactId}.txt`))) {
390
+ let counter = 2;
391
+ while (existsSync(join(artifactDir, `review-brief-${baseArtifactId}-${counter}.md`)) ||
392
+ existsSync(join(artifactDir, `review-raw-${baseArtifactId}-${counter}.txt`))) {
393
+ counter++;
394
+ }
395
+ artifactId = `${baseArtifactId}-${counter}`;
396
+ }
397
+ }
356
398
  const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
357
399
  // Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
358
400
  let savedBrief;
@@ -378,14 +420,21 @@ The top-level comments array is optional. Use it only for actionable line-anchor
378
420
  // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
379
421
  cfg.review.timeoutMs);
380
422
  const raw = llm.output;
423
+ let saved;
424
+ if (artifactDir) {
425
+ try {
426
+ saved = join(artifactDir, `review-raw-${artifactId}.txt`);
427
+ writeFileSync(saved, redactSecrets(raw));
428
+ }
429
+ catch {
430
+ saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
431
+ }
432
+ }
381
433
  const provider = modelProvider(reviewer.model, reviewer.vendor);
382
434
  const v = extractVerdictJson(raw, nonce);
383
435
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
384
436
  const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
385
- const closureLists = [v?.resolved, v?.reraised];
386
- const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
387
- || new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
388
- || [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
437
+ const closureInvalid = isReviewClosureInvalid(v, priorIds);
389
438
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
390
439
  if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
391
440
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
@@ -394,16 +443,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
394
443
  const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
395
444
  : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
396
445
  : classifyVerdictCause(raw, nonce, "approve", llm);
397
- let saved;
398
- if (artifactDir) {
399
- try {
400
- saved = join(artifactDir, `review-raw-${artifactId}.txt`);
401
- writeFileSync(saved, redactSecrets(raw));
402
- }
403
- catch {
404
- saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
405
- }
406
- }
407
446
  const failure = cause === "malformed-verdict"
408
447
  ? "review output unparseable"
409
448
  : "review dispatch failed — no structurally valid nonce-bound response";
@@ -429,7 +468,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
429
468
  const decided = findings !== null
430
469
  ? classifyReviewFindings(findings)
431
470
  : classifyReviewIssues(v.approve, v.issues);
432
- const reraised = priorMaterials.filter((finding) => v.reraised?.includes(finding.fingerprint));
471
+ const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
433
472
  if (reraised.length) {
434
473
  if (decided.pass)
435
474
  decided.headline = "requested changes";
@@ -450,11 +489,21 @@ The top-level comments array is optional. Use it only for actionable line-anchor
450
489
  details,
451
490
  meta: {
452
491
  ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
453
- ...(priorMaterials.length ? { resolved: v.resolved, reraised: v.reraised } : {}),
492
+ ...(priorMaterials.length ? {
493
+ resolved: v.resolved,
494
+ reraised: v.reraised,
495
+ normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
496
+ resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
497
+ reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
498
+ normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
499
+ normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
500
+ } : {}),
454
501
  ...(reraised.length ? { findings: [
455
502
  ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
456
503
  ...reraised,
457
504
  ] } : {}),
505
+ ...(saved ? { rawPath: saved } : {}),
506
+ ...(savedBrief ? { briefPath: savedBrief } : {}),
458
507
  },
459
508
  };
460
509
  }
@@ -366,7 +366,7 @@ export async function runGates(task, ctx) {
366
366
  // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
367
367
  // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
368
368
  const finish = startMeasurement();
369
- const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
369
+ const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
370
370
  const batch = finish();
371
371
  for (const g of gates)
372
372
  addMeasurement(g, batch);
@@ -386,7 +386,7 @@ export async function runGates(task, ctx) {
386
386
  // any later tool before anyone reads its verdict.
387
387
  for (const g of gates) {
388
388
  await emitStart(g);
389
- const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
389
+ const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
390
390
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
391
391
  if (g === "test" && selected)
392
392
  selectedDurationMs = spans.get("test").durationMs;
@@ -14,6 +14,8 @@ export interface RunOptions {
14
14
  graphChanged?: boolean;
15
15
  retryFailed?: boolean;
16
16
  concurrency?: number;
17
+ /** Bounded wait at a drain caused solely by parked tasks. */
18
+ approvalWindowMs?: number;
17
19
  driver?: ExecutorDriver;
18
20
  driverOverride?: DriverChoice;
19
21
  adapters?: WorkerAdapter[];
@@ -55,9 +57,8 @@ export interface RunSummary {
55
57
  * T14, amended by v2.2 T3: approvals the run accepted and never acted on. `approved` above is still
56
58
  * built ONCE at startup — replay determinism depends on it — but a live approval is no longer inert:
57
59
  * the boundary sweep in the task loop releases what lands while the daemon runs, so an approval
58
- * written mid-run is normally enacted by this run. ONE window survives, and it is the reason this
59
- * fold still exists: an approval accepted after the task loop exits — during tip verify, before the
60
- * run-end sample below — meets no further boundary, so nothing can release it before this run ends.
60
+ * written mid-run is enacted at a boundary, during the approval window, or by cancelling tip verify.
61
+ * This fold still exposes decisions that could not enact, including a failure before dispatch.
61
62
  * Without this the run-end record stated only buckets and tipVerify, both accurate, over a milestone
62
63
  * that was silently incomplete: run …230 ended tipVerify "passed" with two upheld approvals and zero
63
64
  * subsequent dispatches. Scored per task on its NEWEST approval: a later approval is the live
@@ -108,6 +109,9 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
108
109
  export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
109
110
  export declare const resetSuiteWaitCeilingForTests: () => void;
110
111
  export declare const APPROVAL_POLL_MS = 250;
112
+ export declare const APPROVAL_WINDOW_MS = 120000;
113
+ export declare const setApprovalWindowForTests: (ms: number) => void;
114
+ export declare const resetApprovalWindowForTests: () => void;
111
115
  export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
112
116
  /** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
113
117
  export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
@@ -151,6 +155,7 @@ export declare function commandsHash(commands: Record<string, string>): string;
151
155
  export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
152
156
  lastMergedTask?: string;
153
157
  baseline?: Baseline;
158
+ signal?: AbortSignal;
154
159
  }): Promise<boolean>;
155
160
  type SuitePidProbe = (pid: number) => number | undefined;
156
161
  /** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership