tickmarkr 2.5.1 → 2.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -114,7 +114,7 @@ export function approvalRunOwner(cwd, runId) {
114
114
  /** The one sentence that says who enacts this release and what it buys. */
115
115
  export function approvalEnactment(token, run) {
116
116
  if (run.live) {
117
- return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`;
117
+ return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}, including in the approval window or by cancelling an in-progress tip verify`;
118
118
  }
119
119
  if (run.blockingRunId) {
120
120
  return `release recorded; live run \`${run.blockingRunId}\` holds the repository lock, so resume \`${run.runId}\` after it ends to ${APPROVAL_ENACTS[token]}`;
@@ -162,6 +162,7 @@ export declare function doctorProbePreflight(cwd?: string, cfg?: {
162
162
  gates: {
163
163
  build?: string | undefined;
164
164
  test?: string | undefined;
165
+ tipTest?: string | undefined;
165
166
  lint?: string | undefined;
166
167
  diffCap?: number | undefined;
167
168
  byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
@@ -278,6 +279,7 @@ export declare function cachedDoctorDiagnostics(cwd?: string, adapters?: WorkerA
278
279
  gates: {
279
280
  build?: string | undefined;
280
281
  test?: string | undefined;
282
+ tipTest?: string | undefined;
281
283
  lint?: string | undefined;
282
284
  diffCap?: number | undefined;
283
285
  byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
@@ -114,7 +114,7 @@ A run is green only when ALL of these hold: the run-end event exists in the jour
114
114
 
115
115
  ### Verified handoffs
116
116
 
117
- When relaying missions between agents, never use bare send-text (\`herdr agent send\` / pane send-text) — it omits Enter. Use \`herdr pane run <pane> "<message>"\` or \`herdr notification show "<message>"\`. Confirm delivery by reading the target pane afterward; never report "relayed" without read-back.
117
+ When relaying missions between agents, never use bare send-text (\`herdr agent send\` / pane send-text) — it omits Enter. Use \`herdr pane run <pane> "<message>"\` or \`herdr notification show "<message>"\`. Confirm delivery by reading the TARGET composer afterward; if the sent text still sits unsubmitted after ~15 s, send-keys enter to that pane once and read again (OBS-975); never report "relayed" without read-back.
118
118
 
119
119
  ### Orient before you act — this block may be the ONLY guidance your host loaded
120
120
 
@@ -4,6 +4,11 @@ import { type JournalEvent } from "../../run/journal.js";
4
4
  * it is a phase counter, while a phase-start CARRYING a gate is the gate start the rail draws. */
5
5
  export declare const TTY_NOISE_EVENTS: readonly ["worker-contact", "worker-status"];
6
6
  type RailTone = "pass" | "fail" | "attention" | "active" | "neutral";
7
+ /** Approval-close lifecycle labels extend the established closed rail vocabulary. */
8
+ export declare const APPROVAL_RAIL_ROWS: Record<string, {
9
+ label: string;
10
+ tone: RailTone;
11
+ }>;
7
12
  /** Closed retained set: the short operator label and the row's default tone. Labels are the rail's
8
13
  * own vocabulary — a raw journal event name is what this surface exists to stop printing. A `pass`
9
14
  * or `ok` datum on the event overrides the default tone, so one gate row can read either way.
@@ -33,6 +33,12 @@ const RAIL_TONES = {
33
33
  active: { glyph: GLYPHS.pointer, paint: LIVE.running },
34
34
  neutral: { glyph: GLYPHS.neutral, paint: LIVE.chrome },
35
35
  };
36
+ /** Approval-close lifecycle labels extend the established closed rail vocabulary. */
37
+ export const APPROVAL_RAIL_ROWS = {
38
+ "approval-window-start": { label: "approval window", tone: "attention" },
39
+ "approval-window-expired": { label: "approval window expired", tone: "attention" },
40
+ "tip-verify-cancelled": { label: "tip verify cancelled", tone: "attention" },
41
+ };
36
42
  /** Closed retained set: the short operator label and the row's default tone. Labels are the rail's
37
43
  * own vocabulary — a raw journal event name is what this surface exists to stop printing. A `pass`
38
44
  * or `ok` datum on the event overrides the default tone, so one gate row can read either way.
@@ -116,6 +122,8 @@ export const RAIL_ROWS = {
116
122
  // gate starts and verdicts
117
123
  "phase-start": { label: "gate start", tone: "active" },
118
124
  "gate-result": { label: "gate", tone: "pass" },
125
+ "baseline-wait": { label: "baseline wait", tone: "active" },
126
+ "suite-budget": { label: "suite budget", tone: "attention" },
119
127
  "gate-reused": { label: "gate reused", tone: "neutral" },
120
128
  "judge-retry": { label: "judge retry", tone: "attention" },
121
129
  "review-no-verdict": { label: "review unavailable", tone: "attention" },
@@ -307,6 +315,11 @@ const RAIL_PROJECTION = {
307
315
  ? `gated ${d.gatedCommit.slice(0, 12)}, tip ${d.branchTip.slice(0, 12)}`
308
316
  : undefined,
309
317
  "trust-auto-answer": (d) => typeof d.adapter === "string" ? `${d.adapter}${typeof d.phase === "string" ? ` ${d.phase}` : ""}` : undefined,
318
+ // SB-1: the census the ceiling released beside, and the budget the round ran under versus the one it
319
+ // would have had on an empty census — none of `{count, occupancyCap, conservativeCap}` is on the ladder
320
+ "suite-budget": (d) => typeof d.count === "number" && typeof d.conservativeCap === "number" && typeof d.occupancyCap === "number"
321
+ ? `beside ${d.count}, cap ${d.conservativeCap} not ${d.occupancyCap}`
322
+ : undefined,
310
323
  };
311
324
  // The row's text: the formatter's detail first, then the salient fields that detail could not carry.
312
325
  // Appending (never prefixing) keeps the operator's prose at the front of the row, so the clip a
@@ -333,7 +346,7 @@ const clipCells = (text, cells) => cells <= 0 ? "" : cellWidth(text) <= cells ?
333
346
  export function narrationRow(event, runId, columns = process.stdout.columns ?? 80) {
334
347
  if (TTY_NOISE_EVENTS.includes(event.event))
335
348
  return null;
336
- const row = RAIL_ROWS[event.event];
349
+ const row = APPROVAL_RAIL_ROWS[event.event] ?? RAIL_ROWS[event.event];
337
350
  if (!row)
338
351
  return null;
339
352
  if (event.event === "phase-start" && typeof event.data.gate !== "string")
@@ -847,6 +847,9 @@ acceptance is required on every task (a nested list of observable outcomes).
847
847
  hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
848
848
  WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
849
849
 
850
+ - A millisecond ceiling in a test is a BUDGET for the SLOWEST RUNNER.
851
+ Any ceiling under one second must carry a SLOWEST-RUNNER note and a member that OVERRUNS it.
852
+
850
853
  PICK THE CRITERION FORM FROM WHO COULD BE WRONG:
851
854
  - When the WORKER could be wrong because it can choose the value, use "test:" and pin the exact
852
855
  literal it could otherwise choose; an example selected by its implementer proves only itself.
@@ -240,6 +240,7 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
240
240
  gates: z.ZodObject<{
241
241
  build: z.ZodOptional<z.ZodString>;
242
242
  test: z.ZodOptional<z.ZodString>;
243
+ tipTest: z.ZodOptional<z.ZodString>;
243
244
  lint: z.ZodOptional<z.ZodString>;
244
245
  diffCap: z.ZodOptional<z.ZodNumber>;
245
246
  byShape: z.ZodOptional<z.ZodOptional<z.ZodRecord<z.ZodEnum<{
@@ -339,6 +340,11 @@ export type InitConfigOverlay = {
339
340
  llm?: TickmarkrConfig["visibility"]["llm"];
340
341
  };
341
342
  };
343
+ /**
344
+ * Default config overlay template.
345
+ * Seam: gates.test defines the per-task gate and baseline test command;
346
+ * gates.tipTest optionally defines the integration tip verify test command (defaults to gates.test).
347
+ */
342
348
  export declare function configTemplate(overlay?: InitConfigOverlay): string;
343
349
  export { FLEET_OVERLAY_KEYS, fleetEditableEquals, fleetRepoOverlayFromDelta, renderFleetOverlayWrite, repoOverlayYaml, serializeFleetOverlay, unifiedYamlDiff, } from "./fleet-overlay.js";
344
350
  export type { FleetOverlayWrite } from "./fleet-overlay.js";
@@ -337,6 +337,7 @@ export const TickmarkrConfigSchema = z.object({
337
337
  gates: z.object({
338
338
  build: z.string(),
339
339
  test: z.string(),
340
+ tipTest: z.string(),
340
341
  lint: z.string(),
341
342
  diffCap: z.number().int().positive(),
342
343
  byShape: z.partialRecord(z.enum(SHAPES), ShapeGateParticipationSchema).optional(),
@@ -756,6 +757,11 @@ export function overlayBytesLoadError(repoRoot, bytes, opts = {}) {
756
757
  return e.message;
757
758
  }
758
759
  }
760
+ /**
761
+ * Default config overlay template.
762
+ * Seam: gates.test defines the per-task gate and baseline test command;
763
+ * gates.tipTest optionally defines the integration tip verify test command (defaults to gates.test).
764
+ */
759
765
  export function configTemplate(overlay) {
760
766
  const base = `# tickmarkr config overlay — merges over built-in defaults (repo beats global beats defaults)
761
767
  # concurrency: 3
@@ -3,9 +3,8 @@ export declare const MAX_BUF: number;
3
3
  export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH", "ORCA_TERMINAL_HANDLE", "ORCA_PANE_KEY", "ORCA_TAB_ID"];
4
4
  /**
5
5
  * Copy of worker env with the fork cap applied and host control-plane vars stripped.
6
- * The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
7
- * same machine the gate shells do, so both seams have to read the same run-owned number rather
8
- * than a flat constant. The operator's own export still wins.
6
+ * Workers retain the run's concurrency-derived cap (resolvedForkCap), independently of
7
+ * verification's suite-window budget. The operator's own export still wins.
9
8
  */
10
9
  export declare function sealHerdrEnv(env?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
11
10
  /** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
@@ -22,9 +22,8 @@ export const HERDR_CONTROL_VARS = [
22
22
  ];
23
23
  /**
24
24
  * Copy of worker env with the fork cap applied and host control-plane vars stripped.
25
- * The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
26
- * same machine the gate shells do, so both seams have to read the same run-owned number rather
27
- * than a flat constant. The operator's own export still wins.
25
+ * Workers retain the run's concurrency-derived cap (resolvedForkCap), independently of
26
+ * verification's suite-window budget. The operator's own export still wins.
28
27
  */
29
28
  export function sealHerdrEnv(env = process.env) {
30
29
  const out = { ...env };
@@ -31,6 +31,8 @@ export interface BaselineCommand {
31
31
  durationMs?: number;
32
32
  /** Sum of the per-file durations named by the runner; null when its output names none. */
33
33
  fileDurationSumMs?: number | null;
34
+ /** Total files reported by the runner summary; null when unavailable. */
35
+ fileCount?: number | null;
34
36
  /** fileDurationSumMs / durationMs — average implied file concurrency, not a configured fork count. */
35
37
  impliedParallelism?: number | null;
36
38
  /** The slowest per-file entry named by the runner; null when per-file timing is unavailable. */
@@ -158,6 +160,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
158
160
  }>): Promise<VacuousOracleWarning[]>;
159
161
  export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
160
162
  rerunOf?: HostStarvedRerun;
163
+ infraRerun?: HostStarvedRerun;
161
164
  }): Promise<GateResult[]>;
162
165
  export interface HostStarvedRerun {
163
166
  durationMs: number;
@@ -178,4 +181,10 @@ interface CalmWindow {
178
181
  }
179
182
  export declare function setCalmWindowForTests(over: Partial<CalmWindow>): void;
180
183
  export declare function resetCalmWindowForTests(): void;
184
+ export declare function waitForCalmWindow(): Promise<number>;
185
+ /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
186
+ export declare function runnerFileCount(raw: string): number | null;
187
+ export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string): string | undefined;
188
+ /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
189
+ export declare function classifyFreshRunnerOutput(entry: BaselineCommand | undefined, raw: string, code: number): FailureClassification | undefined;
181
190
  export {};
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
30
30
  // to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
31
31
  // Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
32
32
  // "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
33
- const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
33
+ const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
34
34
  const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
35
35
  const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
36
36
  const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
103
103
  // Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
104
104
  // and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
105
105
  const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
106
- // The stripped form is a second READ of the same line, for the recognition/headline paths that ask
107
- // "does anything here name a failure" — verdict classification (isInfraLine/namesRegression) keeps
108
- // reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
106
+ // The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
107
+ // Classification applies the same prefix stripping before its infra/regression vetoes.
109
108
  const namesFailureEitherForm = (l) => {
110
109
  if (namesFailure(l))
111
110
  return true;
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
163
162
  // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
164
163
  // to fingerprint(): test-owned output is never runner evidence about the work.
165
164
  const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
166
- if (lines.some(namesRegression))
165
+ const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
166
+ const infra = evidence.some(isInfraLine);
167
+ // A diagnostic section heading names no failing test. It cannot outvote the RPC death
168
+ // beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
169
+ if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
167
170
  return "regression";
168
- return lines.some(isInfraLine) ? "infra" : undefined;
171
+ return infra ? "infra" : undefined;
169
172
  }
170
173
  /**
171
174
  * Capture validity asks a different question from gate classification. At a gate, one genuine
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
338
341
  else if (scripts[name])
339
342
  out[name] = `${runPrefix} ${name}`;
340
343
  }
344
+ if (cfg.gates.tipTest)
345
+ out.tipTest = cfg.gates.tipTest;
341
346
  return out;
342
347
  }
343
348
  /**
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
452
457
  fingerprints: [],
453
458
  durationMs,
454
459
  fileDurationSumMs: null,
460
+ fileCount: null,
455
461
  impliedParallelism: null,
456
462
  longestFile: null,
457
463
  ceilingMs: effectiveCeilingMs({ durationMs }),
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
460
466
  export async function captureBaseline(cwd, commands) {
461
467
  const base = { commands: {} };
462
468
  for (const [name, cmd] of Object.entries(commands)) {
469
+ if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
470
+ continue;
471
+ }
463
472
  const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
464
473
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
465
474
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
466
475
  // scales the next ceiling up — the right direction for a suite that never finished once.
467
476
  const durationMs = r.durationMs ?? 0;
477
+ const combinedOutput = r.stdout + "\n" + r.stderr;
478
+ const raw = combinedOutput.split(cwd).join("");
468
479
  // OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
469
480
  // kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
470
481
  // exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
483
494
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
484
495
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
485
496
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
486
- base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
497
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
487
498
  continue;
488
499
  }
489
- const combinedOutput = r.stdout + "\n" + r.stderr;
490
- const raw = combinedOutput.split(cwd).join("");
491
500
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
492
501
  // discriminator correctly called the mixed output a regression. But the same output also said
493
502
  // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
499
508
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
500
509
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
501
510
  + `First invalidating line: ${invalidatingLines[0]}`);
502
- base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
511
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
503
512
  continue;
504
513
  }
505
- // OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
506
- // teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
514
+ // OBS-966: a worker RPC timeout is infra even beside an all-green summary.
515
+ // Capture and both gate readers share this discriminator; genuine test failures still dominate.
507
516
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
508
517
  if (runnerVerdict === "infra") {
509
- console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
510
- base.commands[name] = invalidCaptureEntry(durationMs, "infra");
518
+ console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
519
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
511
520
  continue;
512
521
  }
513
522
  base.commands[name] = {
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
519
528
  missingCommand: missingConfiguredCommand(cmd, r),
520
529
  durationMs,
521
530
  ...fileTiming(raw, durationMs),
531
+ fileCount: runnerFileCount(raw),
522
532
  ceilingMs: effectiveCeilingMs({ durationMs }),
523
533
  // T7: the world this measurement was taken in, so a later reader can ask whether its own world
524
534
  // is the same one. Recorded from THIS command's own shell result, never re-derived here.
525
535
  ...(r.capacity ? { capacity: r.capacity } : {}),
526
536
  };
527
537
  }
528
- const names = Object.keys(commands);
538
+ const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
529
539
  const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
530
540
  if (names.length > 0 && missing.length === names.length) {
531
541
  base.warnings = [{
@@ -609,10 +619,15 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
609
619
  const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
610
620
  const withRerun = rerunOf ? {
611
621
  ...withReapError,
612
- details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
622
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
613
623
  meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
614
624
  } : withReapError;
615
- results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
625
+ const final = opts.infraRerun ? {
626
+ ...withRerun,
627
+ details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
628
+ meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
629
+ } : withRerun;
630
+ results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
616
631
  };
617
632
  // …and whether the entry that would forgive this command was measured in the same world. A
618
633
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -641,11 +656,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
641
656
  });
642
657
  continue;
643
658
  }
659
+ const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
660
+ const deficit = fileCountDeficit(entry, raw);
661
+ if (deficit) {
662
+ record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
663
+ continue;
664
+ }
644
665
  if (r.code === 0) {
645
666
  record({ gate: name, pass: true, details: "exit 0" });
646
667
  continue;
647
668
  }
648
- const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
649
669
  // OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
650
670
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
651
671
  if (runnerVerdict === "green-teardown") {
@@ -665,12 +685,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
665
685
  // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
666
686
  // `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
667
687
  // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
668
- const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
669
- const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
670
- const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
688
+ const classification = classifyFreshRunnerOutput(entry, raw, r.code);
689
+ if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
690
+ const waitedMs = await waitForCalmWindow();
691
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
692
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
693
+ continue;
694
+ }
671
695
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
672
696
  // own baseline measurement. The first read buys one calm rerun here, never a worker repair.
673
- if (name === "test" && classification !== "infra" && failing.length && !rerunOf
697
+ if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
674
698
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
675
699
  const waitedMs = await waitForCalmWindow();
676
700
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
@@ -684,7 +708,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
684
708
  record({
685
709
  gate: name,
686
710
  pass: false,
687
- details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
711
+ details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
688
712
  meta: { classification, infra: true },
689
713
  });
690
714
  continue;
@@ -695,9 +719,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
695
719
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
696
720
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
697
721
  if (!failing.length && !baselineRed) {
698
- const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
699
- ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
700
- : "was killed at its ceiling";
722
+ const recordedCause = entry?.invalidCause === "infra"
723
+ ? "was invalidated by its recorded runner-infrastructure cause"
724
+ : entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
725
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
726
+ : "was killed at its ceiling";
701
727
  const closed = entry?.infra === true
702
728
  ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
703
729
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
@@ -742,7 +768,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
742
768
  return results;
743
769
  }
744
770
  const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
745
- const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
771
+ const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
746
772
  const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
747
773
  /** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
748
774
  export function classifyRunnerOutput(raw, code) {
@@ -750,6 +776,8 @@ export function classifyRunnerOutput(raw, code) {
750
776
  return undefined;
751
777
  const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
752
778
  const text = lines.join("\n");
779
+ if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
780
+ return classifyFailureOutput(text);
753
781
  const summary = SUMMARY_LINE_RE.exec(text);
754
782
  if (summary) {
755
783
  const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
@@ -780,14 +808,42 @@ export function hostStarved(fresh, durationMs, referenceMs) {
780
808
  const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
781
809
  return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
782
810
  }
783
- const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
811
+ const DEFAULT_CALM = {
812
+ pollMs: process.env.VITEST ? 10 : 5_000,
813
+ maxWaitMs: process.env.VITEST ? 50 : 600_000,
814
+ loadProvider: () => loadavg()[0] ?? 0,
815
+ calmLoad: () => availableParallelism() / 2,
816
+ };
784
817
  let calm = DEFAULT_CALM;
785
818
  export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
786
819
  export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
787
- async function waitForCalmWindow() {
820
+ export async function waitForCalmWindow() {
788
821
  const started = Date.now();
789
822
  while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
790
823
  await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
791
824
  }
792
825
  return Date.now() - started;
793
826
  }
827
+ /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
828
+ export function runnerFileCount(raw) {
829
+ const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
830
+ const clean = line.replace(ANSI_RE, "");
831
+ const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
832
+ return match ? [Number(match[1])] : [];
833
+ });
834
+ return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
835
+ }
836
+ export function fileCountDeficit(entry, raw) {
837
+ const actual = runnerFileCount(raw);
838
+ return entry?.fileCount != null && actual !== null && actual < entry.fileCount
839
+ ? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
840
+ : undefined;
841
+ }
842
+ /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
843
+ export function classifyFreshRunnerOutput(entry, raw, code) {
844
+ if (classifyRunnerOutput(raw, code) === "green-teardown")
845
+ return undefined;
846
+ const { failing } = freshFailures(entry, raw);
847
+ const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
848
+ return verdict === "infra" || verdict === "regression" ? verdict : undefined;
849
+ }
package/dist/gates/llm.js CHANGED
@@ -287,7 +287,7 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
287
287
  const pf = join(dir, "prompt.md");
288
288
  writeFileSync(pf, prompt);
289
289
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
290
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength(r.stdout + r.stderr) };
290
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
291
291
  }
292
292
  finally {
293
293
  rmSync(dir, { recursive: true, force: true });
@@ -427,7 +427,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
427
427
  }
428
428
  }
429
429
  timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
430
- if (timedOut)
430
+ if (timedOut && reviewing)
431
431
  forceClose = true;
432
432
  }
433
433
  const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
@@ -352,7 +352,12 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
352
352
  Approve iff no material finding remains and every prior material is resolved.
353
353
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
354
354
  `;
355
- const artifactId = `${task.id}-${nonce}`;
355
+ // Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
356
+ // must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
357
+ // make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
358
+ // disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
359
+ // so it can never collide with the attempt it replaces.
360
+ const artifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
356
361
  const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
357
362
  // Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
358
363
  let savedBrief;
@@ -378,6 +383,16 @@ The top-level comments array is optional. Use it only for actionable line-anchor
378
383
  // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
379
384
  cfg.review.timeoutMs);
380
385
  const raw = llm.output;
386
+ let saved;
387
+ if (artifactDir) {
388
+ try {
389
+ saved = join(artifactDir, `review-raw-${artifactId}.txt`);
390
+ writeFileSync(saved, redactSecrets(raw));
391
+ }
392
+ catch {
393
+ saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
394
+ }
395
+ }
381
396
  const provider = modelProvider(reviewer.model, reviewer.vendor);
382
397
  const v = extractVerdictJson(raw, nonce);
383
398
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
@@ -394,16 +409,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
394
409
  const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
395
410
  : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
396
411
  : classifyVerdictCause(raw, nonce, "approve", llm);
397
- let saved;
398
- if (artifactDir) {
399
- try {
400
- saved = join(artifactDir, `review-raw-${artifactId}.txt`);
401
- writeFileSync(saved, redactSecrets(raw));
402
- }
403
- catch {
404
- saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
405
- }
406
- }
407
412
  const failure = cause === "malformed-verdict"
408
413
  ? "review output unparseable"
409
414
  : "review dispatch failed — no structurally valid nonce-bound response";
@@ -455,6 +460,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
455
460
  ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
456
461
  ...reraised,
457
462
  ] } : {}),
463
+ ...(saved ? { rawPath: saved } : {}),
464
+ ...(savedBrief ? { briefPath: savedBrief } : {}),
458
465
  },
459
466
  };
460
467
  }
@@ -14,6 +14,8 @@ export interface RunOptions {
14
14
  graphChanged?: boolean;
15
15
  retryFailed?: boolean;
16
16
  concurrency?: number;
17
+ /** Bounded wait at a drain caused solely by parked tasks. */
18
+ approvalWindowMs?: number;
17
19
  driver?: ExecutorDriver;
18
20
  driverOverride?: DriverChoice;
19
21
  adapters?: WorkerAdapter[];
@@ -55,9 +57,8 @@ export interface RunSummary {
55
57
  * T14, amended by v2.2 T3: approvals the run accepted and never acted on. `approved` above is still
56
58
  * built ONCE at startup — replay determinism depends on it — but a live approval is no longer inert:
57
59
  * the boundary sweep in the task loop releases what lands while the daemon runs, so an approval
58
- * written mid-run is normally enacted by this run. ONE window survives, and it is the reason this
59
- * fold still exists: an approval accepted after the task loop exits — during tip verify, before the
60
- * run-end sample below — meets no further boundary, so nothing can release it before this run ends.
60
+ * written mid-run is enacted at a boundary, during the approval window, or by cancelling tip verify.
61
+ * This fold still exposes decisions that could not enact, including a failure before dispatch.
61
62
  * Without this the run-end record stated only buckets and tipVerify, both accurate, over a milestone
62
63
  * that was silently incomplete: run …230 ended tipVerify "passed" with two upheld approvals and zero
63
64
  * subsequent dispatches. Scored per task on its NEWEST approval: a later approval is the live
@@ -108,6 +109,7 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
108
109
  export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
109
110
  export declare const resetSuiteWaitCeilingForTests: () => void;
110
111
  export declare const APPROVAL_POLL_MS = 250;
112
+ export declare const APPROVAL_WINDOW_MS = 1000;
111
113
  export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
112
114
  /** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
113
115
  export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
@@ -151,6 +153,7 @@ export declare function commandsHash(commands: Record<string, string>): string;
151
153
  export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
152
154
  lastMergedTask?: string;
153
155
  baseline?: Baseline;
156
+ signal?: AbortSignal;
154
157
  }): Promise<boolean>;
155
158
  type SuitePidProbe = (pid: number) => number | undefined;
156
159
  /** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership