tickmarkr 1.96.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,8 +3,12 @@ import { extname, join, relative } from "node:path";
3
3
  import { filesGlob } from "../graph/files-glob.js";
4
4
  import { criticalPathHits, DEFAULT_CONFIG, DEFAULT_REVIEW_CRITICAL_PATHS, effectiveReviewPolicy, loadConfig, } from "../config/config.js";
5
5
  import { renderAcceptanceItem } from "../graph/schema.js";
6
- // Advisory plan-time scan only (OBS-12/13/14/21, OBS-76). NEVER expands files[], fails compile,
7
- // or feeds the scope gate — a warning the author acts on. Plain-text (no AST), capped + sorted.
6
+ // Advisory scan (OBS-12/13/14/21, OBS-76). NEVER expands files[] or fails compile — a warning the
7
+ // author acts on. Plain-text (no AST), capped + sorted for DISPLAY only.
8
+ // OBS-547: the collateral prediction is no longer plan-time-only. The daemon computes one full
9
+ // (uncapped) map per run and hands each task's slice to its scope gate, which classifies a red
10
+ // against it — the prediction never causes a refusal, it only says whether a red the gate already
11
+ // found was foreseeable (authoring defect) or not (chargeable quality failure).
8
12
  /** Max files walked per root (sorted walk; rest ignored). */
9
13
  const MAX_WALK_FILES = 400;
10
14
  /** Max collateral test paths listed per task. */
@@ -95,15 +99,17 @@ function makeReader(repoRoot) {
95
99
  };
96
100
  }
97
101
  /**
98
- * Return human-readable scope-lint lines for plan output (no `!` prefix — plan owns that).
99
- * Each line names the task id and at least one missing collateral test path.
102
+ * OBS-547: the FULL per-task collateral prediction, uncapped. ONE map is computed per run (daemon.ts,
103
+ * at run start) and handed whole to the scope gate, so a hit the plan display hides is still there
104
+ * when the gate asks about it. The lint lines below are a capped VIEW of this same map — never a
105
+ * second scan. Tasks with no hits are absent.
100
106
  */
101
- export function collateralLints(tasks, repoRoot) {
107
+ export function collateralHits(tasks, repoRoot) {
108
+ const map = new Map();
102
109
  const testFiles = walkCode(repoRoot, "tests");
103
110
  if (!testFiles.length)
104
- return [];
111
+ return map;
105
112
  const read = makeReader(repoRoot);
106
- const lines = [];
107
113
  for (const t of tasks) {
108
114
  // OBS-22: scopeGate accepts picomatch globs; advisory collateral warnings must agree.
109
115
  const scoped = filesGlob(t.files.map((f) => f.replace(/^\.\//, "")));
@@ -122,14 +128,43 @@ export function collateralLints(tasks, repoRoot) {
122
128
  if (mentions(text, needles))
123
129
  hits.push(tf);
124
130
  }
125
- if (!hits.length)
126
- continue;
127
131
  // deterministic: walk already sorted; stable list
132
+ if (hits.length)
133
+ map.set(t.id, hits);
134
+ }
135
+ return map;
136
+ }
137
+ /** The verbatim repair an authoring defect owes: the files[] lines the spec is missing. */
138
+ export function filesRepair(taskId, paths) {
139
+ return paths.map((p) => `add ${p} to ${taskId}.files[]`).join("\n");
140
+ }
141
+ /**
142
+ * OBS-547: cross-reference at the RED. The gate supplies specificity (the paths actually touched),
143
+ * the prediction supplies classification — so there are no false positives to fear and no refusal to
144
+ * author. Feed this the FULL map for the task: a classifier bounded by the display cap calls a
145
+ * predicted 21st hit a quality failure.
146
+ */
147
+ export function classifyScopeOffenders(taskId, hard, predicted) {
148
+ const named = new Set(predicted);
149
+ const hit = hard.filter((f) => named.has(f));
150
+ const missed = hard.filter((f) => !named.has(f));
151
+ return { authoring: hard.length > 0 && missed.length === 0, predicted: hit, missed, repair: filesRepair(taskId, hit) };
152
+ }
153
+ /**
154
+ * Return human-readable scope-lint lines for plan output (no `!` prefix — plan owns that).
155
+ * Each line names the task id and at least one missing collateral test path.
156
+ */
157
+ export function collateralLints(tasks, repoRoot) {
158
+ const lines = [];
159
+ for (const [id, hits] of collateralHits(tasks, repoRoot)) {
128
160
  const listed = hits.slice(0, MAX_HITS_PER_TASK).join(", ");
161
+ // OBS-547: the cap hides names, never predictions. Say so, and say where the hidden ones surface —
162
+ // a count with no route is exactly what left one run's victim unreadable.
129
163
  const tail = hits.length > MAX_HITS_PER_TASK
130
- ? ` (${hits.length} total; capped at ${MAX_HITS_PER_TASK} shown)`
164
+ ? ` (${hits.length} total; ${MAX_HITS_PER_TASK} shown, ${hits.length - MAX_HITS_PER_TASK} capped out of view`
165
+ + ` but RETAINED for the scope gate — a matching scope red prints the hidden path with its files[] repair)`
131
166
  : "";
132
- lines.push(`${t.id}: likely collateral tests not in files[]: ${listed}${tail}`);
167
+ lines.push(`${id}: likely collateral tests not in files[]: ${listed}${tail}`);
133
168
  }
134
169
  return lines;
135
170
  }
@@ -728,6 +728,16 @@ acceptance is required on every task (a nested list of observable outcomes).
728
728
  - Enumerating one axis exhaustively is what hides the others. A spec that guards PARTIAL coverage
729
729
  site-by-site, member-by-member, can be defeated wholesale by CONDITIONAL coverage, which leaves
730
730
  every enumeration satisfied. After you enumerate, ask what a single flag would do to the whole set.
731
+ - EVERY CRITERION NAMES THE PAIR IT DISCRIMINATES: the correct case that MUST PASS, and the
732
+ neighbouring plausible-wrong or false-clean case that MUST FAIL. Two easy examples that both pass are
733
+ not discrimination — they are two ways of being green, and the wrong half is the whole point: it is
734
+ what a reader would mistake for the mechanism, its VOCABULARY without its behaviour. Measured three
735
+ times in three days, each a criterion that pinned vocabulary instead of the discriminating case: a
736
+ suite that pinned an alarm's text while the alarm itself went unpinned; an export asserted SOMEWHERE,
737
+ which says nothing about the environment a process actually receives; and a gate green on "infra is
738
+ what execution says" while a signalled oracle still charged an attempt. Each was satisfied exactly as
739
+ written and each shipped the defect. If you cannot name the case that must FAIL, you have named a
740
+ topic, not a criterion — and the "so X fails" clause that ends a good criterion is where it goes.
731
741
  - A criterion that names a behaviour must name the VALUE AT WHICH IT WOULD BREAK. Every criterion is a
732
742
  claim about a variable — a status, a width, a count, an arrival time — and if it does not say which
733
743
  value of that variable is the hard one, THE TEST WILL CHOOSE THE EASY ONE AND BE GREEN. Measured: a
@@ -1,3 +1,4 @@
1
+ import { type JournalEvent } from "../run/journal.js";
1
2
  import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
2
3
  export declare const TRAILER_SAFE_FLOOR_COLS = 108;
3
4
  export declare const TRAILER_WIDTH_MARGIN = 2;
@@ -22,20 +23,20 @@ export declare class DeliveryCorruptedError extends Error {
22
23
  export type DriverJournal = (event: string, slotName: string, data: Record<string, unknown>) => void;
23
24
  /** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
24
25
  export declare function workerSplitDirection(paneCols: number | null, safeFloor?: number, margin?: number): "right" | "down";
25
- export declare const BOARD_TARGET_COLS = 110;
26
- export declare const BOARD_SEAT_FLOOR_COLS = 40;
27
- export interface BoardSplitPlan {
28
- direction: "right" | "down";
29
- /** herdr's split ratio is the FIRST child's share and a right split's first child is the caller,
30
- * so this is what the SEAT keeps. Absent on `down` — the board then takes the full width. */
31
- ratio?: number;
32
- /** Columns the board gets under this plan; null when it takes the caller's whole width. */
33
- boardCols: number | null;
26
+ export declare const BOARD_HEIGHT_SHARE = 0.72;
27
+ export interface BoardPlacement {
28
+ /** Always down: the split is vertical, so the board can own the caller's FULL width. */
29
+ direction: "down";
30
+ /** herdr's split ratio is the FIRST child's share, and a down split's first child is the TOP
31
+ * region — the region the board occupies once it is swapped above the caller. */
32
+ ratio: number;
33
+ /** The new pane is swapped ABOVE the caller; the split alone would leave the board underneath. */
34
+ swap: "above";
34
35
  }
35
- /** Board-first placement beside the supervising seat: right only while the caller can fund the board
36
- * its target AND leave the seat its floor; otherwise down at full width, never a squeezed board.
37
- * An unmeasurable caller falls back to down like every other placement here (fail closed). */
38
- export declare function boardSplitPlan(callerCols: number | null, boardCols?: number, seatFloor?: number): BoardSplitPlan;
36
+ /** The single approved vertical-stack record. The caller's columns are accepted and deliberately
37
+ * ignored: this signature is where width used to decide the arrangement, and the parameter stays
38
+ * so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
39
+ export declare function boardSplitPlan(_callerCols?: number | null): BoardPlacement;
39
40
  export declare function taskGroupOf(name: string): string | undefined;
40
41
  /** The operator-facing TITLE for a tab this driver creates when no stage label is supplied: a task
41
42
  * token for a worker (`T5`, `T5↻2` on a retry) and ROLE + task for a gate pane (`REVIEW T5`) — the
@@ -63,11 +64,14 @@ export declare class HerdrDriver implements ExecutorDriver {
63
64
  private inputBoxes;
64
65
  private bootstraps;
65
66
  private journalRoots;
67
+ private narrate?;
66
68
  private ws;
67
69
  private callerPane;
68
70
  private watches;
69
71
  constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource, journal?: DriverJournal | undefined);
70
72
  private appendDispatchRetry;
73
+ /** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
74
+ narrateWith(narrate: (event: JournalEvent) => void): void;
71
75
  private serial;
72
76
  private deliveryQueue;
73
77
  private reserveDispatch;
@@ -77,6 +81,7 @@ export declare class HerdrDriver implements ExecutorDriver {
77
81
  static available(): boolean;
78
82
  private herdr;
79
83
  private namedPaneId;
84
+ private paneStillOpen;
80
85
  private paneId;
81
86
  private paneWidth;
82
87
  slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
@@ -116,6 +121,7 @@ export declare class HerdrDriver implements ExecutorDriver {
116
121
  private closeGrouped;
117
122
  private ownedWatchPanes;
118
123
  private watchSlot;
124
+ private discardSplit;
119
125
  narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
120
126
  reconcile(desired: Set<string>, runId: string, opts?: {
121
127
  spareLiveLlm?: boolean;
@@ -69,19 +69,19 @@ export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_CO
69
69
  }
70
70
  // The watch board's geometry, deliberately NOT the trailer floor above. `workerSplitDirection`
71
71
  // halves the caller and refuses a right split under 108+2 — that bound protects WORKER panes, which
72
- // print a trailer; the supervising seat + board pair does not. Applied to that pair on 2026-08-18 it
73
- // sent a 189-column tab's board BELOW the seat and the operator corrected it (QUEUE-v194 criterion 1;
74
- // skills/tickmarkr-overseer/SKILL.md: "the side placement outranks the halving floor"). So the board
75
- // is allocated its measured width FIRST and the seat keeps the remainder.
76
- export const BOARD_TARGET_COLS = 110; // §14a measured clean-render bound for the board
77
- export const BOARD_SEAT_FLOOR_COLS = 40; // the seat beside it still has to be usable
78
- /** Board-first placement beside the supervising seat: right only while the caller can fund the board
79
- * its target AND leave the seat its floor; otherwise down at full width, never a squeezed board.
80
- * An unmeasurable caller falls back to down like every other placement here (fail closed). */
81
- export function boardSplitPlan(callerCols, boardCols = BOARD_TARGET_COLS, seatFloor = BOARD_SEAT_FLOOR_COLS) {
82
- if (callerCols == null || callerCols < boardCols + seatFloor)
83
- return { direction: "down", boardCols: null };
84
- return { direction: "right", ratio: Math.round(((callerCols - boardCols) / callerCols) * 1e4) / 1e4, boardCols };
72
+ // print a trailer; the supervising seat + board pair does not, and neither does the board's own
73
+ // placement any more. Every width-derived variant of this placement has been wrong in the operator's
74
+ // tab: the halving floor sent a 189-column board below the seat (2026-08-18), and the width-first
75
+ // side split that replaced it puts the board and the narration shoulder to shoulder when the board is
76
+ // the surface the operator reads and the narration is the rail beneath it. The placement is now ONE
77
+ // record — the board stacked ABOVE the caller at full width, taking 72% of the height — and it is
78
+ // invariant: no terminal width, measured or unmeasurable, can select a different arrangement.
79
+ export const BOARD_HEIGHT_SHARE = 0.72; // board 72 / narration 28, the operator's stack
80
+ /** The single approved vertical-stack record. The caller's columns are accepted and deliberately
81
+ * ignored: this signature is where width used to decide the arrangement, and the parameter stays
82
+ * so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
83
+ export function boardSplitPlan(_callerCols) {
84
+ return { direction: "down", ratio: BOARD_HEIGHT_SHARE, swap: "above" };
85
85
  }
86
86
  /** The tab a slot belongs to: its TASK — worker, judge, review and consult panes for one task share it.
87
87
  * Returns undefined for everything else, which keeps those on the dedicated-tab path.
@@ -147,6 +147,7 @@ export class HerdrDriver {
147
147
  // the caller launched tickmarkr elsewhere (process.cwd is not run identity). The repo itself is
148
148
  // also bound for judge/review/consult slots whose cwd is the root rather than a task worktree.
149
149
  journalRoots = new Map();
150
+ narrate;
150
151
  // VIS-10: the run's workspace id, captured once at construction (the daemon inherits it from the
151
152
  // operator's env before the driver is built). Required at slot() time, never in the constructor —
152
153
  // pickDriver and its unit test construct HerdrDriver without env, so slot() is the trust gate.
@@ -176,7 +177,14 @@ export class HerdrDriver {
176
177
  if (!repoRoot) {
177
178
  throw new Error(`cannot journal dispatch-retry: slot ${slot.name} has no daemon repo binding for ${slot.cwd}`);
178
179
  }
179
- Journal.open(repoRoot, owned.runId).append("dispatch-retry", owned.taskId, data);
180
+ // Bound to the live narration sink (`narrateWith`): this Journal is the driver's own — the
181
+ // daemon never appends this event and never sees it — so an unbound handle here persists the
182
+ // recovery to the file and the pipe while the operator's rail stays silent about it.
183
+ Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
184
+ }
185
+ /** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
186
+ narrateWith(narrate) {
187
+ this.narrate = narrate;
180
188
  }
181
189
  serial(fn) {
182
190
  const p = this.groupSerial.then(fn, fn);
@@ -274,6 +282,20 @@ export class HerdrDriver {
274
282
  return null;
275
283
  }
276
284
  }
285
+ // Is this pane id still in the listing? FAIL CLOSED: a listing we cannot read cannot prove a pane
286
+ // gone, and the caller uses this to decide whether a pane it tried to close is really off screen.
287
+ async paneStillOpen(paneId) {
288
+ const r = await this.herdr(`pane list`);
289
+ if (r.code !== 0)
290
+ return true;
291
+ try {
292
+ const panes = JSON.parse(r.stdout).result?.panes;
293
+ return !Array.isArray(panes) || panes.some((p) => p.pane_id === paneId);
294
+ }
295
+ catch {
296
+ return true;
297
+ }
298
+ }
277
299
  // Before delivery, resolve fresh via the durable name because pane ids can compact. After delivery,
278
300
  // pin the verified target so early liveness cannot drift to a label rebound onto another pane.
279
301
  async paneId(slot) {
@@ -1059,19 +1081,18 @@ export class HerdrDriver {
1059
1081
  return p.workspace_id === this.ws && typeof p.pane_id === "string" && owned?.role === "watch" && owned.taskId === "run";
1060
1082
  }).map((p) => p.pane_id);
1061
1083
  }
1062
- // T2: the watch is a sibling of the daemon's own pane, never a separate tab — beside it when the
1063
- // tab can fund the board its width, below it at full width when it cannot. Its durable owned name
1064
- // is how a later daemon RECOGNIZES the board it must retire, so a run never stacks a second one.
1084
+ // T2: the watch is a sibling of the daemon's own pane, never a separate tab — stacked ABOVE it at
1085
+ // the caller's full width, always, whatever the terminal measures. Its durable owned name is how a
1086
+ // later daemon RECOGNIZES the board it must retire, so a run never stacks a second one.
1065
1087
  async watchSlot(cwd, name) {
1066
1088
  if (!this.ws)
1067
1089
  throw new Error("herdr watch placement requires HERDR_WORKSPACE_ID — refusing unseeded pane");
1068
1090
  if (!this.callerPane)
1069
1091
  throw new Error("herdr watch placement requires HERDR_PANE_ID — refusing untargeted split");
1070
- // Board width first (boardSplitPlan), measured off the caller through the driver's own layout
1071
- // read — never an unconditional right split, and never the worker halving rule.
1072
- const plan = boardSplitPlan(await this.paneWidth(this.callerPane));
1073
- const ratio = plan.ratio == null ? "" : ` --ratio ${plan.ratio}`;
1074
- const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction}${ratio} --no-focus`);
1092
+ // One invariant placement (boardSplitPlan): split the caller down, then swap the new pane above
1093
+ // it. No layout read decides this — width chose the arrangement twice and was wrong twice.
1094
+ const plan = boardSplitPlan();
1095
+ const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction} --ratio ${plan.ratio} --no-focus`);
1075
1096
  if (sp.code !== 0)
1076
1097
  throw new Error(`herdr watch split failed: ${sp.stderr || sp.stdout}`);
1077
1098
  let pane;
@@ -1083,18 +1104,59 @@ export class HerdrDriver {
1083
1104
  }
1084
1105
  if (typeof pane !== "string" || !pane)
1085
1106
  throw new Error(`herdr watch split returned no pane id: ${sp.stdout}`);
1107
+ // The split leaves the board UNDER the caller; the swap is what makes the stack the requested
1108
+ // one. Verified, not assumed: a swap that failed would leave a board below the narration while
1109
+ // the daemon reported the geometry it asked for. Instead the split pane is closed and the failure
1110
+ // propagates — the daemon swallows it and runs boardless, which is honest about what is on screen.
1111
+ // `pane swap` answers a no-op with a ZERO exit and `changed:false` (herdr socket API: a swap it
1112
+ // declined is a non-error response), so an exit code alone proves nothing about the geometry —
1113
+ // that is exactly the path that would leave the board below the narration while the daemon
1114
+ // reported the stack. The documented `changed` flag is the verification; anything else — a
1115
+ // nonzero exit, `changed:false`, an unparseable result — fails closed.
1116
+ const swapped = await this.herdr(`pane swap --source-pane ${shq(pane)} --target-pane ${shq(this.callerPane)}`);
1117
+ let swapChanged;
1118
+ try {
1119
+ swapChanged = JSON.parse(swapped.stdout).result?.changed;
1120
+ }
1121
+ catch {
1122
+ /* fail closed below */
1123
+ }
1124
+ if (swapped.code !== 0 || swapChanged !== true) {
1125
+ await this.discardSplit(pane, `herdr watch swap ${plan.swap} failed: ${swapped.code !== 0
1126
+ ? swapped.stderr || swapped.stdout
1127
+ : `herdr reported no swap took place: ${swapped.stdout || swapped.stderr}`}`);
1128
+ }
1086
1129
  const renamed = await this.herdr(`pane rename ${shq(pane)} ${shq(name)}`);
1087
1130
  if (renamed.code !== 0 || await this.namedPaneId(name) !== pane) {
1088
- await this.herdr(`pane close ${shq(pane)}`);
1089
- throw new Error(`herdr watch rename failed: ${renamed.stderr || renamed.stdout}`);
1131
+ await this.discardSplit(pane, `herdr watch rename failed: ${renamed.stderr || renamed.stdout}`);
1090
1132
  }
1091
1133
  const seed = await this.herdr(`pane run ${shq(pane)} ${shq(`cd ${shq(cwd)}; export HERDR_WORKSPACE_ID=${shq(this.ws)}; ${herdrSealShellPrefix()}`)}`, cwd);
1092
1134
  if (seed.code !== 0) {
1093
- await this.herdr(`pane close ${shq(pane)}`);
1094
- throw new Error(`herdr watch seed failed: ${seed.stderr || seed.stdout}`);
1135
+ await this.discardSplit(pane, `herdr watch seed failed: ${seed.stderr || seed.stdout}`);
1095
1136
  }
1096
1137
  return { id: pane, name, cwd };
1097
1138
  }
1139
+ // A board that could not be placed costs the OPERATOR a stray pane unless the split is really
1140
+ // taken back, so every close is followed by a pane-list verification rather than issued and
1141
+ // forgotten. A nonzero close and a success that frees nothing are both diagnosed from that same
1142
+ // observation. The daemon swallows it either way and runs boardless — but never silently keeps a
1143
+ // split the geometry it asked for does not include.
1144
+ async discardSplit(pane, why) {
1145
+ const closed = await this.herdr(`pane close ${shq(pane)}`);
1146
+ const stillOpen = await this.paneStillOpen(pane);
1147
+ const orphan = stillOpen
1148
+ ? closed.code !== 0
1149
+ ? closed.stderr || closed.stdout || `exit ${closed.code}`
1150
+ : "close reported success but the pane could not be proven absent from pane list"
1151
+ : null;
1152
+ if (orphan !== null) {
1153
+ // The daemon swallows narrator failures whole (visibility is never a gate), so the thrown
1154
+ // error dies in its catch. This line is the operator's only notice that a pane they did not
1155
+ // ask for is still on their screen and that no process will take it back.
1156
+ console.error(`tickmarkr: the watch split ${pane} survived its close (${orphan}) — close it by hand; the run continues boardless`);
1157
+ }
1158
+ throw new Error(orphan === null ? why : `${why} — and the split pane ${pane} survived its close (${orphan})`);
1159
+ }
1098
1160
  // T6 narrator: the run's single live status surface, RUNNING THE COMMAND THIS CALL SUPPLIED. Only
1099
1161
  // a board this driver instance itself opened is reused (this.watches); any other surviving board —
1100
1162
  // a prior run's, or one already carrying this run's canonical name after a resume — is retired and
@@ -1,3 +1,4 @@
1
+ import type { JournalEvent } from "../run/journal.js";
1
2
  export interface Slot {
2
3
  id: string;
3
4
  name: string;
@@ -75,6 +76,7 @@ export interface ExecutorDriver {
75
76
  nudge?(slot: Slot, message: string): Promise<boolean>;
76
77
  notify(msg: string, opts?: NotifyOpts): Promise<void>;
77
78
  close(slot: Slot): Promise<void>;
79
+ narrateWith?(narrate: (event: JournalEvent) => void): void;
78
80
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
79
81
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
80
82
  reconcile?: (desired: Set<string>, runId: string, opts?: {
@@ -179,20 +179,30 @@ function tail(out, n = 8) {
179
179
  return "\n" + t.split("\n").slice(-n).join("\n");
180
180
  }
181
181
  /**
182
- * OBS-540: classify only bytes produced by the deterministic oracle process. A nonzero command with
183
- * an infra-only output shape returned no product verdict, so it remains fail-closed but cannot be
184
- * billed as a diff failure. Judge reasons never reach this helper: the judge path starts after the
185
- * deterministic loop and a parsed refusal is an evaluated quality verdict whatever its prose says.
182
+ * OBS-540/551: classify only execution evidence produced by the deterministic oracle process. That
183
+ * includes its output, exit status and runner-reported test count — never judge-authored prose. A
184
+ * nonzero command with an infra-only output shape, one of the enumerated signal exits, or a zero-run
185
+ * report without a failure identity returned no product verdict, so it remains fail-closed but cannot
186
+ * be billed as a diff failure. A parsed judge refusal is an evaluated quality verdict whatever its
187
+ * prose says. As in baseline classification, a named regression remains a verdict even if teardown
188
+ * later exits through a signal.
186
189
  */
187
190
  function oracleExecutionFailure(label, code, stdout, stderr) {
188
191
  const output = [stderr, stdout].filter((part) => part.length > 0).join("\n");
189
192
  const classification = classifyFailureOutput(output);
190
- if (classification === "infra") {
193
+ const signalName = code === 143 ? "SIGTERM" : code === 137 ? "SIGKILL" : undefined;
194
+ const signalKilled = signalName !== undefined && classification !== "regression";
195
+ const signalEvidence = signalKilled
196
+ ? `signal-shaped exit ${code} (${signalName}); `
197
+ : "";
198
+ const zeroRun = testsRan(output) === 0 && classification !== "regression";
199
+ const zeroRunEvidence = zeroRun ? "the runner reported zero tests run and no failure identity; " : "";
200
+ if (classification === "infra" || signalKilled || zeroRun) {
191
201
  return {
192
202
  gate: "acceptance",
193
203
  pass: false,
194
- details: `oracle failed: ${label} (exit ${code}) — infrastructure blocked execution before a verdict; this oracle verified nothing${tail(output)}`,
195
- meta: { cause: "oracle-execution", classification, infra: true, retryable: false },
204
+ details: `oracle failed: ${label} (exit ${code}) — ${signalEvidence || zeroRunEvidence}infrastructure blocked execution before a verdict; this oracle verified nothing${tail(output)}`,
205
+ meta: { cause: "oracle-execution", classification: "infra", infra: true, retryable: false },
196
206
  };
197
207
  }
198
208
  return {
@@ -33,6 +33,25 @@ export interface GateVia {
33
33
  nameFor: (role: "judge" | "review", adapter: string) => string;
34
34
  labelFor: (role: "judge" | "review") => string;
35
35
  }
36
+ export declare const GATE_INACTIVITY_WINDOW_MS: number;
37
+ /** Test seam — shrink only the calibrated inactivity window; production always reads 12 minutes. */
38
+ export declare function setGateInactivityWindowMsForTests(ms: number): void;
39
+ export declare function resetGateInactivityWindowMsForTests(): void;
40
+ export interface GateCpuAccountant {
41
+ start(): Promise<void>;
42
+ read(): {
43
+ cpu: {
44
+ ms: number;
45
+ resolutionMs: number;
46
+ } | undefined;
47
+ gaps: number;
48
+ };
49
+ stop(): Promise<void>;
50
+ }
51
+ export type GateCpuAccountantFactory = (marker: string, cwd: string) => GateCpuAccountant;
52
+ /** Test seam for deterministic measurable/activity/gap samples without platform-specific ps output. */
53
+ export declare function setGateCpuAccountantFactoryForTests(factory: GateCpuAccountantFactory): void;
54
+ export declare function resetGateCpuAccountantFactoryForTests(): void;
36
55
  export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
37
56
  value: T;
38
57
  outputs: string[];
package/dist/gates/llm.js CHANGED
@@ -6,6 +6,7 @@ import { join } from "node:path";
6
6
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
7
7
  import { bannerShell, paneDispatchCommand } from "../brand.js";
8
8
  import { sh } from "../run/git.js";
9
+ import { harvestCpuFlatWindowMs, normalizeStallSnapshot, WorkerTreeCpuAccountant, } from "../run/stall.js";
9
10
  export const GATE_PANE_SEP = " · ";
10
11
  // v1.64 gate-integrity (repo-scan Tier A·1): the concrete completion-faking shortcuts every
11
12
  // judge/review verdict must hunt for. Shared verbatim by the acceptance judge and review prompts.
@@ -94,6 +95,30 @@ export function rolePaneNameFromPrompt(prompt, fallback) {
94
95
  return fallback;
95
96
  }
96
97
  const llmOutputCapture = new AsyncLocalStorage();
98
+ // v2.0 T1 (OBS-555): the empirical healthy-duration p95 is 10.6 minutes. The smallest whole
99
+ // minute above it plus the specified one-minute margin is twelve minutes. This default remains
100
+ // strictly below BOTH unchanged 900_000ms production dispatch timeouts: JUDGE_TIMEOUT_MS in
101
+ // acceptance.ts and reviewGate's literal timeout in review.ts. Scope's 300_000ms call is not a
102
+ // verdict gate and retains its existing one-wait behavior.
103
+ export const GATE_INACTIVITY_WINDOW_MS = 12 * 60_000;
104
+ const GATE_WAIT_SLICE_MS = 30_000;
105
+ let gateInactivityWindowMs = GATE_INACTIVITY_WINDOW_MS;
106
+ /** Test seam — shrink only the calibrated inactivity window; production always reads 12 minutes. */
107
+ export function setGateInactivityWindowMsForTests(ms) {
108
+ gateInactivityWindowMs = ms;
109
+ }
110
+ export function resetGateInactivityWindowMsForTests() {
111
+ gateInactivityWindowMs = GATE_INACTIVITY_WINDOW_MS;
112
+ }
113
+ const productionGateCpuAccountant = (marker, cwd) => new WorkerTreeCpuAccountant(marker, cwd);
114
+ let gateCpuAccountantFactory = productionGateCpuAccountant;
115
+ /** Test seam for deterministic measurable/activity/gap samples without platform-specific ps output. */
116
+ export function setGateCpuAccountantFactoryForTests(factory) {
117
+ gateCpuAccountantFactory = factory;
118
+ }
119
+ export function resetGateCpuAccountantFactoryForTests() {
120
+ gateCpuAccountantFactory = productionGateCpuAccountant;
121
+ }
97
122
  // OBS-132: acceptance.ts owns verdict parsing and is deliberately byte-untouched. This async-scoped
98
123
  // recorder lets run-gates observe the exact output that acceptance parsed without changing runLlm's
99
124
  // return value or leaking concurrent tasks into one another. Callers retain output only when the
@@ -119,6 +144,8 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
119
144
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
120
145
  export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
121
146
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
147
+ let slot;
148
+ let accountant;
122
149
  try {
123
150
  const pf = join(dir, "prompt.md");
124
151
  writeFileSync(pf, prompt);
@@ -131,19 +158,90 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
131
158
  adapter.headlessCommand(pf, model),
132
159
  gateExitTrailer(nonce),
133
160
  ].join("\n"));
134
- const slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
161
+ slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
135
162
  via.onSlot?.(slot);
136
163
  await via.driver.run(slot, paneDispatchCommand(scriptPath));
137
164
  // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
138
165
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
139
- await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
140
- const out = await via.driver.read(slot, 400);
141
- if (!via.keep)
142
- await via.driver.close(slot);
166
+ const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
167
+ let out;
168
+ const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
169
+ if (!gatePrompt) {
170
+ await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
171
+ out = await via.driver.read(slot, 400);
172
+ }
173
+ else {
174
+ // Review and judge waits are sliced so the two-leg inactivity policy can observe the pane and
175
+ // the exact dispatch script's process tree between waits. The accountant retains short-lived
176
+ // descendants with the same semantics as daemon workers; llm.ts depends only on stall.ts.
177
+ accountant = gateCpuAccountantFactory(scriptPath, cwd);
178
+ await accountant.start();
179
+ const startedAt = Date.now();
180
+ out = await via.driver.read(slot, 400);
181
+ let priorSnapshot = normalizeStallSnapshot(out);
182
+ const anchoredAt = Date.now();
183
+ let quietSince = anchoredAt;
184
+ const initialCpu = accountant.read();
185
+ let priorCpuMs = initialCpu.cpu?.ms;
186
+ let priorGaps = initialCpu.gaps;
187
+ let cpuFlatSince = initialCpu.cpu === undefined ? undefined : anchoredAt;
188
+ while (Date.now() - startedAt < timeoutMs) {
189
+ const remaining = timeoutMs - (Date.now() - startedAt);
190
+ // The adaptive test-window arm keeps a seam-adjusted case sliced too; production stays 30s.
191
+ const sliceMs = Math.max(1, Math.min(GATE_WAIT_SLICE_MS, Math.ceil(gateInactivityWindowMs / 4), remaining));
192
+ const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
193
+ const raw = await via.driver.read(slot, 400);
194
+ out = raw;
195
+ // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
196
+ // wait timed out at the same boundary the marker landed; either way a trailer completes
197
+ // normally and is never mistaken for inactivity.
198
+ if (matched || new RegExp(exitPattern).test(raw))
199
+ break;
200
+ const now = Date.now();
201
+ const snapshot = normalizeStallSnapshot(raw);
202
+ if (snapshot !== priorSnapshot) {
203
+ priorSnapshot = snapshot;
204
+ quietSince = now;
205
+ }
206
+ const observation = accountant.read();
207
+ const cpu = observation.cpu;
208
+ if (observation.gaps !== priorGaps || cpu === undefined) {
209
+ // Missing evidence is a hold-open signal, never guessed inactivity. A later measurable
210
+ // sample starts a fresh complete window rather than inheriting quiet time across the gap.
211
+ priorGaps = observation.gaps;
212
+ priorCpuMs = undefined;
213
+ cpuFlatSince = undefined;
214
+ quietSince = now;
215
+ continue;
216
+ }
217
+ if (priorCpuMs === undefined || cpu.ms !== priorCpuMs) {
218
+ // Any process-tree CPU movement holds the call open and re-arms both clocks. Equality only
219
+ // becomes "flat" after the existing resolution-aware quantum window has elapsed.
220
+ priorCpuMs = cpu.ms;
221
+ cpuFlatSince = now;
222
+ quietSince = now;
223
+ continue;
224
+ }
225
+ const cpuFlatFor = now - (cpuFlatSince ?? now);
226
+ const snapshotQuietFor = now - quietSince;
227
+ if (cpuFlatFor >= harvestCpuFlatWindowMs(cpu.resolutionMs)
228
+ && snapshotQuietFor >= gateInactivityWindowMs) {
229
+ break;
230
+ }
231
+ }
232
+ }
143
233
  return dewrapPaneVerdict(out, nonce);
144
234
  }
145
235
  finally {
146
- rmSync(dir, { recursive: true, force: true });
236
+ try {
237
+ await accountant?.stop();
238
+ if (slot && !via.keep)
239
+ await via.driver.close(slot);
240
+ }
241
+ finally {
242
+ // Unconditional and synchronous: a stop/close failure must not leak this call's prompt and script.
243
+ rmSync(dir, { recursive: true, force: true });
244
+ }
147
245
  }
148
246
  }
149
247
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
@@ -4,6 +4,23 @@ import { type GateName, type Task } from "../graph/schema.js";
4
4
  import { type Baseline } from "./baseline.js";
5
5
  import { type GateVia } from "./llm.js";
6
6
  import type { GateResult } from "./types.js";
7
+ export type LoadProvider = () => number;
8
+ /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
9
+ export declare function setLoadProviderForTests(provider: LoadProvider): void;
10
+ export declare function resetLoadProviderForTests(): void;
11
+ /**
12
+ * One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
13
+ * intervals and nothing between them, so the composite `test` gate (a selected screen, then other
14
+ * gates, then the full suite) reports the two suites' cost rather than the span containing them —
15
+ * and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
16
+ * queue as well as the work. The load samples bracket the FIRST interval's start and the LAST
17
+ * interval's end: start is what a scheduler would have decided on, end is the state it left behind.
18
+ */
19
+ export interface GateTelemetry {
20
+ durationMs: number;
21
+ load1Start: number;
22
+ load1End: number;
23
+ }
7
24
  export type GateEvent = {
8
25
  phase: "start";
9
26
  gate: GateName;
@@ -31,6 +48,7 @@ export interface GateContext {
31
48
  artifactDir?: string;
32
49
  pipeline?: "v185" | "legacy";
33
50
  selectTests?: boolean;
51
+ collateral?: ReadonlyArray<string>;
34
52
  onGate?: (e: GateEvent) => void | Promise<void>;
35
53
  }
36
54
  /**