tickmarkr 2.5.0 → 2.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
30
30
  // to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
31
31
  // Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
32
32
  // "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
33
- const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
33
+ const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
34
34
  const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
35
35
  const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
36
36
  const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
103
103
  // Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
104
104
  // and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
105
105
  const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
106
- // The stripped form is a second READ of the same line, for the recognition/headline paths that ask
107
- // "does anything here name a failure" — verdict classification (isInfraLine/namesRegression) keeps
108
- // reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
106
+ // The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
107
+ // Classification applies the same prefix stripping before its infra/regression vetoes.
109
108
  const namesFailureEitherForm = (l) => {
110
109
  if (namesFailure(l))
111
110
  return true;
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
163
162
  // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
164
163
  // to fingerprint(): test-owned output is never runner evidence about the work.
165
164
  const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
166
- if (lines.some(namesRegression))
165
+ const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
166
+ const infra = evidence.some(isInfraLine);
167
+ // A diagnostic section heading names no failing test. It cannot outvote the RPC death
168
+ // beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
169
+ if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
167
170
  return "regression";
168
- return lines.some(isInfraLine) ? "infra" : undefined;
171
+ return infra ? "infra" : undefined;
169
172
  }
170
173
  /**
171
174
  * Capture validity asks a different question from gate classification. At a gate, one genuine
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
338
341
  else if (scripts[name])
339
342
  out[name] = `${runPrefix} ${name}`;
340
343
  }
344
+ if (cfg.gates.tipTest)
345
+ out.tipTest = cfg.gates.tipTest;
341
346
  return out;
342
347
  }
343
348
  /**
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
452
457
  fingerprints: [],
453
458
  durationMs,
454
459
  fileDurationSumMs: null,
460
+ fileCount: null,
455
461
  impliedParallelism: null,
456
462
  longestFile: null,
457
463
  ceilingMs: effectiveCeilingMs({ durationMs }),
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
460
466
  export async function captureBaseline(cwd, commands) {
461
467
  const base = { commands: {} };
462
468
  for (const [name, cmd] of Object.entries(commands)) {
469
+ if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
470
+ continue;
471
+ }
463
472
  const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
464
473
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
465
474
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
466
475
  // scales the next ceiling up — the right direction for a suite that never finished once.
467
476
  const durationMs = r.durationMs ?? 0;
477
+ const combinedOutput = r.stdout + "\n" + r.stderr;
478
+ const raw = combinedOutput.split(cwd).join("");
468
479
  // OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
469
480
  // kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
470
481
  // exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
483
494
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
484
495
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
485
496
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
486
- base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
497
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
487
498
  continue;
488
499
  }
489
- const combinedOutput = r.stdout + "\n" + r.stderr;
490
- const raw = combinedOutput.split(cwd).join("");
491
500
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
492
501
  // discriminator correctly called the mixed output a regression. But the same output also said
493
502
  // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
499
508
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
500
509
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
501
510
  + `First invalidating line: ${invalidatingLines[0]}`);
502
- base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
511
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
503
512
  continue;
504
513
  }
505
- // OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
506
- // teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
514
+ // OBS-966: a worker RPC timeout is infra even beside an all-green summary.
515
+ // Capture and both gate readers share this discriminator; genuine test failures still dominate.
507
516
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
508
517
  if (runnerVerdict === "infra") {
509
- console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
510
- base.commands[name] = invalidCaptureEntry(durationMs, "infra");
518
+ console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
519
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
511
520
  continue;
512
521
  }
513
522
  base.commands[name] = {
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
519
528
  missingCommand: missingConfiguredCommand(cmd, r),
520
529
  durationMs,
521
530
  ...fileTiming(raw, durationMs),
531
+ fileCount: runnerFileCount(raw),
522
532
  ceilingMs: effectiveCeilingMs({ durationMs }),
523
533
  // T7: the world this measurement was taken in, so a later reader can ask whether its own world
524
534
  // is the same one. Recorded from THIS command's own shell result, never re-derived here.
525
535
  ...(r.capacity ? { capacity: r.capacity } : {}),
526
536
  };
527
537
  }
528
- const names = Object.keys(commands);
538
+ const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
529
539
  const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
530
540
  if (names.length > 0 && missing.length === names.length) {
531
541
  base.warnings = [{
@@ -609,10 +619,15 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
609
619
  const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
610
620
  const withRerun = rerunOf ? {
611
621
  ...withReapError,
612
- details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
622
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
613
623
  meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
614
624
  } : withReapError;
615
- results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
625
+ const final = opts.infraRerun ? {
626
+ ...withRerun,
627
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
628
+ meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
629
+ } : withRerun;
630
+ results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
616
631
  };
617
632
  // …and whether the entry that would forgive this command was measured in the same world. A
618
633
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -641,11 +656,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
641
656
  });
642
657
  continue;
643
658
  }
659
+ const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
660
+ const deficit = fileCountDeficit(entry, raw);
661
+ if (deficit) {
662
+ record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
663
+ continue;
664
+ }
644
665
  if (r.code === 0) {
645
666
  record({ gate: name, pass: true, details: "exit 0" });
646
667
  continue;
647
668
  }
648
- const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
649
669
  // OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
650
670
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
651
671
  if (runnerVerdict === "green-teardown") {
@@ -665,12 +685,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
665
685
  // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
666
686
  // `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
667
687
  // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
668
- const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
669
- const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
670
- const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
688
+ const classification = classifyFreshRunnerOutput(entry, raw, r.code);
689
+ if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
690
+ const waitedMs = await waitForCalmWindow();
691
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
692
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
693
+ continue;
694
+ }
671
695
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
672
696
  // own baseline measurement. The first read buys one calm rerun here, never a worker repair.
673
- if (name === "test" && classification !== "infra" && failing.length && !rerunOf
697
+ if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
674
698
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
675
699
  const waitedMs = await waitForCalmWindow();
676
700
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
@@ -684,7 +708,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
684
708
  record({
685
709
  gate: name,
686
710
  pass: false,
687
- details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
711
+ details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
688
712
  meta: { classification, infra: true },
689
713
  });
690
714
  continue;
@@ -695,9 +719,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
695
719
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
696
720
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
697
721
  if (!failing.length && !baselineRed) {
698
- const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
699
- ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
700
- : "was killed at its ceiling";
722
+ const recordedCause = entry?.invalidCause === "infra"
723
+ ? "was invalidated by its recorded runner-infrastructure cause"
724
+ : entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
725
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
726
+ : "was killed at its ceiling";
701
727
  const closed = entry?.infra === true
702
728
  ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
703
729
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
@@ -742,7 +768,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
742
768
  return results;
743
769
  }
744
770
  const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
745
- const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
771
+ const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
746
772
  const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
747
773
  /** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
748
774
  export function classifyRunnerOutput(raw, code) {
@@ -750,6 +776,8 @@ export function classifyRunnerOutput(raw, code) {
750
776
  return undefined;
751
777
  const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
752
778
  const text = lines.join("\n");
779
+ if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
780
+ return classifyFailureOutput(text);
753
781
  const summary = SUMMARY_LINE_RE.exec(text);
754
782
  if (summary) {
755
783
  const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
@@ -780,14 +808,42 @@ export function hostStarved(fresh, durationMs, referenceMs) {
780
808
  const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
781
809
  return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
782
810
  }
783
- const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
811
+ const DEFAULT_CALM = {
812
+ pollMs: process.env.VITEST ? 10 : 5_000,
813
+ maxWaitMs: process.env.VITEST ? 50 : 600_000,
814
+ loadProvider: () => loadavg()[0] ?? 0,
815
+ calmLoad: () => availableParallelism() / 2,
816
+ };
784
817
  let calm = DEFAULT_CALM;
785
818
  export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
786
819
  export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
787
- async function waitForCalmWindow() {
820
+ export async function waitForCalmWindow() {
788
821
  const started = Date.now();
789
822
  while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
790
823
  await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
791
824
  }
792
825
  return Date.now() - started;
793
826
  }
827
+ /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
828
+ export function runnerFileCount(raw) {
829
+ const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
830
+ const clean = line.replace(ANSI_RE, "");
831
+ const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
832
+ return match ? [Number(match[1])] : [];
833
+ });
834
+ return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
835
+ }
836
+ export function fileCountDeficit(entry, raw) {
837
+ const actual = runnerFileCount(raw);
838
+ return entry?.fileCount != null && actual !== null && actual < entry.fileCount
839
+ ? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
840
+ : undefined;
841
+ }
842
+ /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
843
+ export function classifyFreshRunnerOutput(entry, raw, code) {
844
+ if (classifyRunnerOutput(raw, code) === "green-teardown")
845
+ return undefined;
846
+ const { failing } = freshFailures(entry, raw);
847
+ const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
848
+ return verdict === "infra" || verdict === "regression" ? verdict : undefined;
849
+ }
@@ -61,7 +61,12 @@ export interface LlmRunResult {
61
61
  output: string;
62
62
  exitCode?: number;
63
63
  timedOut: boolean;
64
+ launchNeverStarted?: boolean;
65
+ seatAuthoredBytes?: number;
64
66
  }
67
+ export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
+ export declare function reviewSeatOutput(raw: string, nonce: string): string;
69
+ export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
65
70
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
66
71
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
67
72
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
package/dist/gates/llm.js CHANGED
@@ -1,3 +1,4 @@
1
+ import { stripVTControlCharacters } from "node:util";
1
2
  import { AsyncLocalStorage } from "node:async_hooks";
2
3
  import { randomBytes } from "node:crypto";
3
4
  import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
@@ -5,7 +6,7 @@ import { tmpdir } from "node:os";
5
6
  import { join } from "node:path";
6
7
  import { matchesTrustDialog } from "../adapters/types.js";
7
8
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
8
- import { bannerShell, paneDispatchCommand } from "../brand.js";
9
+ import { bannerShell, paneDispatchCommand, PLAIN_BANNER } from "../brand.js";
9
10
  import { sh } from "../run/git.js";
10
11
  import { harvestCpuFlatWindowMs, normalizeStallSnapshot, WorkerTreeCpuAccountant, } from "../run/stall.js";
11
12
  export const GATE_PANE_SEP = " · ";
@@ -129,13 +130,164 @@ export async function captureLlmOutput(run) {
129
130
  const value = await llmOutputCapture.run(outputs, run);
130
131
  return { value, outputs };
131
132
  }
133
+ // The harness preamble has a fixed row shape, in order: the dispatch echo rows (the identity export
134
+ // and the START printf, each possibly re-echoed behind the shell prompt), the START acknowledgement
135
+ // row, the banner rows, blank rows, and the identity row. RULING-229-15 add.1 makes the boundary
136
+ // STRUCTURAL: a pane read is line-terminated, so only the LAST row of a capture can be a partial
137
+ // paint. Every row before it is complete, and a complete row is harness only when it EQUALS a full
138
+ // harness row in the position the preamble allows. A complete row that merely starts with "T",
139
+ // "export", "review" or a banner glyph is seat text, and seat prose that MENTIONS a marker mid-row
140
+ // counts in full — "contains TICKMARKR_START_ somewhere" is never by itself a reason to measure zero.
141
+ const BANNER_ROWS = PLAIN_BANNER.replace(/\n$/, "").split("\n");
142
+ const ECHO_OPENER = "export HERDR_WORKSPACE_ID=";
143
+ const ECHO_START_OPENER = "printf '%s%s\\n' 'TICKMARKR_START_' '";
144
+ const ECHO_OPENERS = [ECHO_OPENER, ECHO_START_OPENER];
145
+ // RULING-229-15 add.5: the prompt glyph set is this CLOSED list — the corpus sweeps it, the grammar
146
+ // consumes the LONGEST glyph that matches (so ">>" is one glyph and ">" is still one).
147
+ export const PROMPT_GLYPHS = ["➜", "❯", "$", "%", ">>", ">"];
148
+ const GIT_SEGMENT_OPEN = "git:(";
149
+ function echoRowMatch(row) {
150
+ const opener = (at) => {
151
+ const body = row.slice(at);
152
+ if (ECHO_OPENERS.some((o) => body.startsWith(o)))
153
+ return "complete";
154
+ return ECHO_OPENERS.some((o) => o.startsWith(body)) ? "partial" : "none";
155
+ };
156
+ const ws = (at) => /^\s+/.exec(row.slice(at))?.[0].length;
157
+ const bare = opener(0);
158
+ if (bare !== "none")
159
+ return bare;
160
+ const glyph = PROMPT_GLYPHS.find((g) => row.startsWith(g));
161
+ if (glyph === undefined)
162
+ return "none";
163
+ let i = glyph.length;
164
+ const ws1 = ws(i);
165
+ if (ws1 === undefined)
166
+ return row.length === i ? "partial" : "none";
167
+ i += ws1;
168
+ const dir = /^\S+/.exec(row.slice(i));
169
+ if (!dir)
170
+ return "partial";
171
+ i += dir[0].length;
172
+ const ws2 = ws(i);
173
+ if (ws2 === undefined)
174
+ return "partial";
175
+ i += ws2;
176
+ const rest = row.slice(i);
177
+ if (rest.startsWith(GIT_SEGMENT_OPEN)) {
178
+ const close = row.indexOf(")", i + GIT_SEGMENT_OPEN.length);
179
+ if (close < 0)
180
+ return "partial";
181
+ i = close + 1;
182
+ const ws3 = ws(i);
183
+ if (ws3 === undefined)
184
+ return row.length === i ? "partial" : "none";
185
+ return opener(i + ws3);
186
+ }
187
+ if (GIT_SEGMENT_OPEN.startsWith(rest))
188
+ return "partial"; // "", "g", "gi", "git", "git:"
189
+ return opener(i);
190
+ }
191
+ // A complete dispatch echo row: the grammar reaches the opener. Never "contains the opener" — a row
192
+ // that mentions it mid-prose is the seat's.
193
+ function completeEchoRow(row) {
194
+ return echoRowMatch(row) === "complete";
195
+ }
196
+ // A LAST row still being painted: any prefix of a row in the grammar.
197
+ function partialEchoRow(row) {
198
+ return echoRowMatch(row) !== "none";
199
+ }
200
+ const START_MARKER = "TICKMARKR_START_";
201
+ const START_ROW = /^TICKMARKR_START_[\w-]+$/;
202
+ const IDENTITY_LINE = /^(?:review\s*·|tickmarkr(?::|$))/;
203
+ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
204
+ // Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
205
+ const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
206
+ // The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
207
+ function seatStart(output) {
208
+ const rows = output.split("\n");
209
+ if (rows.length > 1 && rows[rows.length - 1] === "")
210
+ rows.pop(); // the read's own line terminator
211
+ let stage = ECHO;
212
+ let bannerAt; // next banner row expected once the banner has begun
213
+ let offset = 0;
214
+ for (let i = 0; i < rows.length; i++) {
215
+ const row = rows[i].replace(/[ \t]+$/, "");
216
+ const t = row.trim();
217
+ const last = i === rows.length - 1;
218
+ // Complete harness rows: equality against the shape the preamble allows at this stage.
219
+ let accepted = false;
220
+ if (stage < SEAT && t.length === 0)
221
+ accepted = true; // blank rows between preamble rows
222
+ else if (stage <= ECHO && completeEchoRow(row))
223
+ accepted = true;
224
+ else if (stage <= START && START_ROW.test(row)) {
225
+ stage = BANNER;
226
+ accepted = true;
227
+ }
228
+ else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
229
+ stage = BANNER;
230
+ bannerAt = BANNER_ROWS.indexOf(row) + 1;
231
+ accepted = true;
232
+ }
233
+ else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
234
+ bannerAt++;
235
+ accepted = true;
236
+ }
237
+ else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
238
+ stage = SEAT;
239
+ accepted = true;
240
+ }
241
+ if (accepted) {
242
+ offset += rows[i].length + 1;
243
+ continue;
244
+ }
245
+ if (!last)
246
+ return offset;
247
+ // The last row may be a partial paint: a prefix of the next harness row the preamble allows.
248
+ if (stage <= ECHO && partialEchoRow(row))
249
+ return -1;
250
+ if (stage <= START && (START_MARKER.startsWith(row) || /^TICKMARKR_START_[\w-]*$/.test(row)))
251
+ return -1;
252
+ if (stage <= BANNER && (bannerAt === undefined
253
+ ? BANNER_ROWS.some((b) => b.startsWith(row))
254
+ : BANNER_ROWS[bannerAt]?.startsWith(row) === true))
255
+ return -1;
256
+ // The identity row is painted right after the banner, so its prefix is a partial paint only there;
257
+ // with no banner in the capture, "review" or "tick" alone is the seat's own first row.
258
+ if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
259
+ return -1;
260
+ return offset;
261
+ }
262
+ return -1; // every row was harness — the seat has not taken its turn
263
+ }
264
+ // A pane's dispatch echo, start acknowledgement, banner and identity are harness bytes.
265
+ // Remove only that leading preamble, never matching text later in the seat's response.
266
+ // Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
267
+ // which is what makes the caller's running Math.max safe: a partial banner counted once would be
268
+ // retained for the whole call and buy a silent seat its full ceiling.
269
+ export function reviewSeatOutput(raw, nonce) {
270
+ const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
271
+ const start = seatStart(output);
272
+ if (start < 0)
273
+ return "";
274
+ const seat = output.slice(start);
275
+ // Anything after the nonce-bound trailer belongs to the terminal (typically the next shell
276
+ // prompt), not the reviewer. Deleting only the marker would count that postlude as seat output and
277
+ // let a zero-byte seat escape first-silence demotion.
278
+ const trailer = new RegExp(`(?:^|\\n)TICKMARKR_EXIT_${nonce}:\\d+[^\\n]*(?:\\n|$)`).exec(seat);
279
+ // A trailer row still being typed is not stripped: after the preamble a last row "T" is far more
280
+ // often the seat's first byte than the harness's exit marker, and the next read completes either.
281
+ return trailer ? seat.slice(0, trailer.index) : seat;
282
+ }
283
+ export const REVIEW_FIRST_LIVENESS_MS = 30_000;
132
284
  async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
133
285
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
134
286
  try {
135
287
  const pf = join(dir, "prompt.md");
136
288
  writeFileSync(pf, prompt);
137
289
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
138
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
290
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
139
291
  }
140
292
  finally {
141
293
  rmSync(dir, { recursive: true, force: true });
@@ -150,6 +302,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
150
302
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
151
303
  let slot;
152
304
  let accountant;
305
+ let forceClose = false;
153
306
  try {
154
307
  const pf = join(dir, "prompt.md");
155
308
  writeFileSync(pf, prompt);
@@ -180,6 +333,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
180
333
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
181
334
  let out;
182
335
  let timedOut = false;
336
+ let launchNeverStarted = false;
337
+ let seatAuthoredBytes = 0;
183
338
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
184
339
  if (!gatePrompt) {
185
340
  await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
@@ -193,6 +348,9 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
193
348
  await accountant.start();
194
349
  const startedAt = Date.now();
195
350
  out = await via.driver.read(slot, 400);
351
+ const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
352
+ seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
353
+ let firstLivenessObserved = false;
196
354
  let priorSnapshot = normalizeStallSnapshot(out);
197
355
  const anchoredAt = Date.now();
198
356
  let quietSince = anchoredAt;
@@ -207,12 +365,35 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
207
365
  const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
208
366
  const raw = await via.driver.read(slot, 400);
209
367
  out = raw;
368
+ seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
210
369
  // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
211
370
  // wait timed out at the same boundary the marker landed; either way a trailer completes
212
371
  // normally and is never mistaken for inactivity.
213
372
  if (matched || new RegExp(exitPattern).test(raw))
214
373
  break;
215
374
  const now = Date.now();
375
+ // The ceiling wins if it coincides with the first beat (or the read crosses it).
376
+ // That seat was killed by its configured timeout, not an early launch reroute.
377
+ if (now - startedAt >= timeoutMs)
378
+ break;
379
+ if (reviewing && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
380
+ firstLivenessObserved = true;
381
+ // RS-2: the beat reads seat-authored bytes ALONE. CPU evidence never holds a preamble-only
382
+ // capture open to the ceiling; a seat that has not written one byte of its own is re-routed.
383
+ // OBS-944 is why a buffering runner earns no exemption here: the claude-code seat's 901 s
384
+ // capture was byte-identical across two legs and ended at the pane-identity line — that
385
+ // seat never started, it was not quietly working. RULING-229-06 puts the beat on the PANE
386
+ // path only; a headless `-p` runner (runHeadlessDetailed, no pane, no beat) buffers every
387
+ // byte until completion and keeps its full ceiling.
388
+ if (seatAuthoredBytes === 0) {
389
+ launchNeverStarted = true;
390
+ forceClose = true;
391
+ break;
392
+ }
393
+ }
394
+ // Producing reviews own their full ceiling; inactivity is not a review verdict.
395
+ if (reviewing)
396
+ continue;
216
397
  const snapshot = normalizeStallSnapshot(raw);
217
398
  if (snapshot !== priorSnapshot) {
218
399
  priorSnapshot = snapshot;
@@ -246,18 +427,22 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
246
427
  }
247
428
  }
248
429
  timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
430
+ if (timedOut && reviewing)
431
+ forceClose = true;
249
432
  }
250
433
  const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
251
434
  return {
252
435
  output: dewrapPaneVerdict(out, nonce),
253
436
  ...(Number.isFinite(exitCode) ? { exitCode } : {}),
254
437
  timedOut,
438
+ launchNeverStarted,
439
+ seatAuthoredBytes,
255
440
  };
256
441
  }
257
442
  finally {
258
443
  try {
259
444
  await accountant?.stop();
260
- if (slot && !via.keep)
445
+ if (slot && (forceClose || !via.keep))
261
446
  await via.driver.close(slot);
262
447
  }
263
448
  finally {
@@ -1,6 +1,7 @@
1
1
  import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type TickmarkrConfig, type Tier } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
+ import { type StructuredFinding } from "../run/journal.js";
4
5
  import { modelProvider } from "../route/preference.js";
5
6
  import { type GateVia } from "./llm.js";
6
7
  import type { GateResult } from "./types.js";
@@ -16,6 +17,8 @@ export interface ReviewFinding {
16
17
  }
17
18
  export interface ReviewVerdict {
18
19
  approve?: boolean;
20
+ resolved?: string[];
21
+ reraised?: string[];
19
22
  issues?: string[];
20
23
  findings?: ReviewFinding[];
21
24
  comments?: Array<{
@@ -54,12 +57,12 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
54
57
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
55
58
  floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
56
59
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
57
- onSeat?: (seat: number) => void): BillingChannel | null;
58
- export type ReviewUnparseableCause = VerdictUnparseableCause;
60
+ onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
61
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
59
62
  /**
60
63
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
61
64
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
62
65
  * judgement rather than a guarantee made by this renderer.
63
66
  */
64
67
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
65
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
68
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[]): Promise<GateResult>;