tickmarkr 2.1.7 → 2.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import { declaredInputBoxForWorkerName, matchesEmptyInputBox, matchesInputBox, m
6
6
  import { consumePaneLaunchIntent, PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
7
7
  import { createWorktree, sh } from "../run/git.js";
8
8
  import { Journal } from "../run/journal.js";
9
+ import { readSupervision } from "../run/supervision.js";
9
10
  import { herdrSealShellPrefix } from "./subprocess.js";
10
11
  import { canonicalizeLegacyName, formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
11
12
  // VIS-09 P43-03: adopted safety floor from 43-MEASUREMENT.md (narrowest safe 53 → floor 108).
@@ -194,6 +195,64 @@ export class HerdrDriver {
194
195
  // recovery to the file and the pipe while the operator's rail stays silent about it.
195
196
  Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
196
197
  }
198
+ openRunJournal(runId) {
199
+ const roots = new Set(this.journalRoots.values());
200
+ roots.add(process.cwd());
201
+ for (const repoRoot of roots) {
202
+ try {
203
+ return { repoRoot, journal: Journal.open(repoRoot, runId, this.narrate) };
204
+ }
205
+ catch {
206
+ /* try the next known root */
207
+ }
208
+ }
209
+ return undefined;
210
+ }
211
+ liveSupervisionSeats(repoRoot) {
212
+ const seats = new Set();
213
+ for (const tier of readSupervision(repoRoot)) {
214
+ if (tier.state === "ARMED" && tier.seat)
215
+ seats.add(tier.seat);
216
+ }
217
+ return seats;
218
+ }
219
+ journalReconcile(handle, event, taskId, data) {
220
+ try {
221
+ handle?.journal.append(event, taskId, data);
222
+ }
223
+ catch {
224
+ /* reconcile remains cosmetic even when its audit row cannot be written */
225
+ }
226
+ }
227
+ paneReconcileData(pane, label, ownedName, sweeperRunId) {
228
+ return {
229
+ paneId: pane.paneId,
230
+ ...(pane.tabId !== undefined ? { tabId: pane.tabId } : {}),
231
+ label,
232
+ ownedName,
233
+ ownedRunId: ownedName.runId,
234
+ runId: sweeperRunId,
235
+ sweeperRunId,
236
+ };
237
+ }
238
+ parsePaneList(handle, runId, stdout, stage) {
239
+ try {
240
+ const panes = JSON.parse(stdout).result?.panes;
241
+ if (!Array.isArray(panes))
242
+ throw new Error("pane list returned no panes array");
243
+ return panes;
244
+ }
245
+ catch (error) {
246
+ this.journalReconcile(handle, "pane-reconcile-list-failed", undefined, {
247
+ runId,
248
+ sweeperRunId: runId,
249
+ stage,
250
+ error: error instanceof Error ? error.message : String(error),
251
+ stdout,
252
+ });
253
+ return null;
254
+ }
255
+ }
197
256
  /** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
198
257
  narrateWith(narrate) {
199
258
  this.narrate = narrate;
@@ -1199,6 +1258,7 @@ export class HerdrDriver {
1199
1258
  // resume/end) run with nothing in flight and take them too. Cosmetic by contract: every failure —
1200
1259
  // herdr gone, pane vanished mid-sweep, unparseable listing — is swallowed; this method never throws.
1201
1260
  async reconcile(desired, runId, opts) {
1261
+ const journalHandle = this.openRunJournal(runId);
1202
1262
  try {
1203
1263
  if (!this.ws)
1204
1264
  return;
@@ -1206,25 +1266,96 @@ export class HerdrDriver {
1206
1266
  // panesToClose skips any pane whose label doesn't parse as tickmarkr-owned (orchestrator/operator
1207
1267
  // shells, undetected agents), so a fuller pane listing never widens the blast radius.
1208
1268
  const list = await this.herdr("pane list");
1209
- const panes = JSON.parse(list.stdout).result?.panes ?? [];
1210
- const toClose = panesToClose(panes.map((p) => ({ name: p.label, paneId: p.pane_id, tabId: p.tab_id, workspaceId: p.workspace_id })), desired, this.ws, runId, opts);
1269
+ if (list.code !== 0) {
1270
+ this.journalReconcile(journalHandle, "pane-reconcile-list-failed", undefined, {
1271
+ runId,
1272
+ sweeperRunId: runId,
1273
+ stage: "pre-close",
1274
+ exitCode: list.code,
1275
+ error: list.stderr || list.stdout || `exit ${list.code}`,
1276
+ });
1277
+ return;
1278
+ }
1279
+ const panes = this.parsePaneList(journalHandle, runId, list.stdout, "pre-close");
1280
+ if (panes === null)
1281
+ return;
1282
+ const liveSeats = journalHandle ? this.liveSupervisionSeats(journalHandle.repoRoot) : new Set();
1283
+ for (const seat of opts?.liveSeats ?? [])
1284
+ liveSeats.add(seat);
1285
+ const toClose = panesToClose(panes.map((p) => ({ name: p.label, paneId: p.pane_id, tabId: p.tab_id, workspaceId: p.workspace_id })), desired, this.ws, runId, { ...opts, liveSeats });
1286
+ const paneById = new Map(panes
1287
+ .filter((p) => typeof p.pane_id === "string")
1288
+ .map((p) => [p.pane_id, p]));
1211
1289
  const touched = new Set();
1212
1290
  for (const c of toClose) {
1213
- if (typeof c.tabId === "string")
1214
- touched.add(c.tabId);
1215
- await this.herdr(`pane close ${shq(c.paneId)}`); // best-effort — a vanished pane is already reconciled
1291
+ const listed = paneById.get(c.paneId);
1292
+ const label = typeof listed?.label === "string" ? listed.label : "";
1293
+ const ownedName = parseOwnedName(label);
1294
+ if (!ownedName) {
1295
+ this.journalReconcile(journalHandle, "pane-reconcile-close-failed", undefined, {
1296
+ paneId: c.paneId,
1297
+ ...(c.tabId !== undefined ? { tabId: c.tabId } : {}),
1298
+ label,
1299
+ runId,
1300
+ sweeperRunId: runId,
1301
+ error: "pane selected for reconcile no longer has a parseable owned label",
1302
+ });
1303
+ continue;
1304
+ }
1305
+ const data = this.paneReconcileData(c, label, ownedName, runId);
1306
+ const closed = await this.herdr(`pane close ${shq(c.paneId)}`);
1307
+ if (closed.code === 0) {
1308
+ if (typeof c.tabId === "string")
1309
+ touched.add(c.tabId);
1310
+ this.journalReconcile(journalHandle, "pane-reconcile-close", ownedName.taskId, data);
1311
+ }
1312
+ else {
1313
+ this.journalReconcile(journalHandle, "pane-reconcile-close-failed", ownedName.taskId, {
1314
+ ...data,
1315
+ exitCode: closed.code,
1316
+ error: closed.stderr || closed.stdout || `exit ${closed.code}`,
1317
+ });
1318
+ }
1216
1319
  }
1217
1320
  if (touched.size === 0)
1218
1321
  return;
1219
1322
  // a tab our closes emptied was ours by construction (a tab with operator panes still has panes)
1220
1323
  const pl = await this.herdr("pane list");
1221
- const alive = new Set((JSON.parse(pl.stdout).result?.panes ?? []).map((p) => p.tab_id));
1222
- for (const tab of touched)
1223
- if (!alive.has(tab))
1224
- await this.herdr(`tab close ${shq(tab)}`);
1324
+ if (pl.code !== 0) {
1325
+ this.journalReconcile(journalHandle, "pane-reconcile-list-failed", undefined, {
1326
+ runId,
1327
+ sweeperRunId: runId,
1328
+ stage: "post-close",
1329
+ exitCode: pl.code,
1330
+ error: pl.stderr || pl.stdout || `exit ${pl.code}`,
1331
+ });
1332
+ return;
1333
+ }
1334
+ const alivePanes = this.parsePaneList(journalHandle, runId, pl.stdout, "post-close");
1335
+ if (alivePanes === null)
1336
+ return;
1337
+ const alive = new Set(alivePanes.map((p) => p.tab_id));
1338
+ for (const tab of touched) {
1339
+ if (alive.has(tab))
1340
+ continue;
1341
+ const closed = await this.herdr(`tab close ${shq(tab)}`);
1342
+ if (closed.code !== 0) {
1343
+ this.journalReconcile(journalHandle, "tab-reconcile-close-failed", undefined, {
1344
+ tabId: tab,
1345
+ runId,
1346
+ sweeperRunId: runId,
1347
+ exitCode: closed.code,
1348
+ error: closed.stderr || closed.stdout || `exit ${closed.code}`,
1349
+ });
1350
+ }
1351
+ }
1225
1352
  }
1226
- catch {
1227
- /* cosmetic — visibility hygiene never fails the run */
1353
+ catch (error) {
1354
+ this.journalReconcile(journalHandle, "pane-reconcile-failed", undefined, {
1355
+ runId,
1356
+ sweeperRunId: runId,
1357
+ error: error instanceof Error ? error.message : String(error),
1358
+ });
1228
1359
  }
1229
1360
  }
1230
1361
  async worktree(repo, branch, baseRef) {
@@ -54,10 +54,12 @@ export interface FleetAgent {
54
54
  tabId?: string;
55
55
  workspaceId?: string;
56
56
  }
57
- export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: {
57
+ export interface PanesToCloseOpts {
58
58
  spareLiveLlm?: boolean;
59
59
  endedRunIds?: Set<string>;
60
- }): {
60
+ liveSeats?: Set<string>;
61
+ }
62
+ export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: PanesToCloseOpts): {
61
63
  paneId: string;
62
64
  tabId?: string;
63
65
  }[];
@@ -80,8 +82,5 @@ export interface ExecutorDriver {
80
82
  narrateWith?(narrate: (event: JournalEvent) => void): void;
81
83
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
82
84
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
83
- reconcile?: (desired: Set<string>, runId: string, opts?: {
84
- spareLiveLlm?: boolean;
85
- endedRunIds?: Set<string>;
86
- }) => Promise<void>;
85
+ reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
87
86
  }
@@ -20,54 +20,6 @@ export function parseOwnedName(name) {
20
20
  export function isForeignName(name) {
21
21
  return parseOwnedName(name) === null;
22
22
  }
23
- // v1.22b T1: workspace-aware fold over a fleet snapshot — decides which owned task panes are garbage
24
- // right now: the desired-set/spareLiveLlm sweep (OBS-17 T2), scoped to THIS RUN'S OWN panes (by runId,
25
- // OBS-772) in THIS RUN'S OWN WORKSPACE (OBS-769). Both conditions, and neither alone is the rule.
26
- // Watch panes are operator-owned after run end and are reclaimed by the next run; foreign names
27
- // (parseOwnedName fails) are never candidates.
28
- //
29
- // OBS-769 — WHY THE SWEEP STOPS AT THE WORKSPACE BOUNDARY. It used to close an owned pane carrying
30
- // any OTHER runId in any other workspace, unconditionally, as a "misplaced leftover". Two tickmarkr
31
- // runs in two repositories are lawful (the lock forbids two runs in ONE repository, not on one
32
- // machine) and herdr gives each its own workspace — so that branch made every pair of concurrent
33
- // runs kill each other's LIVE workers. Measured 2026-08-28: the run in w0 closed run ...2958's
34
- // panes at 23:42:40.351/.392, and 53s later ...2958's own task-human sweep closed w0's live codex
35
- // worker at 23:43:34.096. ...2958 ended 0/8. The death detector cannot see it: closing the pane
36
- // makes paneAbsent, processTree, confirmedProcessTree and worktreeDelta true by ONE cause, and a
37
- // closed pane can never accrue the CPU that the `cpu-accruing` hold reads.
38
- // The comment this replaces claimed "only run age marks a misplaced pane garbage" — there was no age
39
- // check in the code, and age is the wrong predicate anyway: w0's run STARTED EARLIER than ...2958,
40
- // so an age rule would have licensed exactly the kill that landed. Run age says nothing about
41
- // liveness, and a sweeping daemon cannot read another repository's run state. The workspace is the
42
- // only ownership boundary available without cross-repo I/O, so it is the one enforced.
43
- // Cost, named: an orphan pane from a dead run stranded in a workspace no later run opens is now left
44
- // for the operator. That is cosmetic (`reconcile` is cosmetic by contract — "visibility is never a
45
- // gate"), and a cosmetic cleanup must never be able to kill a live worker.
46
- // OBS-772 — WHY THE runId LINE EXISTS, AND WHY THE WORKSPACE LINE ALONE WAS NOT THE FIX. The first
47
- // repair was workspace-scoped only, and its own comment dismissed the residue — "two runs sharing one
48
- // workspace would still sweep each other" — as unreachable, on the reasoning that one workspace per run
49
- // is herdr's placement. That reasoned from ONE driver to the whole product. OrcaDriver has no workspace
50
- // dimension at all: orca.ts passes a single ORCA_SPACE as the workspaceId for EVERY checkout and as
51
- // `ws`, so `workspaceId !== ws` is never true there and every foreign pane fell straight through. Orca
52
- // users had zero protection while the defect read as fixed. The runId line is the real rule and it is
53
- // driver-agnostic: reconcile exists to clean up THIS RUN's panes, and a leftover from a dead run is
54
- // exactly what cannot be told from a live run's pane without liveness data this process does not have.
55
- // Both lines are kept — the workspace line preserves the pre-existing sparing of this run's own panes
56
- // in another workspace, which the runId line alone would not.
57
- // ⚠ WHAT THE runId LINE COST BEFORE OBS-777 — SUSPENDED, NOT NARROWED, and the price was larger than
58
- // it read. Sparing every other runId suspended OBS-17's FOUNDING use case: "a killed daemon can't
59
- // close its slots". This sweep was built to reclaim exactly those orphans, but could not reclaim ANY
60
- // previous run's panes. Three separate pins asserted the old behaviour (reconcile.test.ts,
61
- // orca-placement.test.ts,
62
- // reconcile-live.test.ts); all three were changed deliberately, and the third is why this paragraph
63
- // exists rather than a shorter one — two flipped pins is a trade, three is a pattern.
64
- // OBS-777 RESTORES that reclamation: the CALLER passes `opts.endedRunIds`, a Set the daemon computes
65
- // ONCE at run start from this repository's own `run-end` journals and dead lock holders. This fold
66
- // stays pure — it gains one optional field, not a repo root — a foreign repository's runId is never
67
- // resolvable and so stays spared by construction, and no driver learns about workspaces.
68
- // ponytail: two conditions, no geometry reasoning, nothing driver-specific. `reconcile` is cosmetic by
69
- // contract, and a cosmetic cleanup must never be able to kill a live worker — which is why the
70
- // ended-run authority is the only safe way to restore the sweep without reviving the cross-run kill.
71
23
  export function panesToClose(agents, desired, ws, runId, opts) {
72
24
  const out = [];
73
25
  for (const a of agents) {
@@ -76,6 +28,8 @@ export function panesToClose(agents, desired, ws, runId, opts) {
76
28
  const owned = parseOwnedName(a.name);
77
29
  if (!owned || owned.role === "watch")
78
30
  continue;
31
+ if (opts?.liveSeats?.has(a.name))
32
+ continue;
79
33
  if (owned.runId !== runId && !opts?.endedRunIds?.has(owned.runId))
80
34
  continue;
81
35
  if (a.workspaceId !== ws)
@@ -28,6 +28,19 @@ export interface VitestListedTest {
28
28
  file: string;
29
29
  projectName?: string;
30
30
  }
31
+ export type VitestListResult = {
32
+ status: "listed";
33
+ tests: VitestListedTest[];
34
+ } | {
35
+ status: "failed";
36
+ error: string;
37
+ };
38
+ export declare function listVitestTests(cwd: string): Promise<VitestListResult>;
39
+ export interface NamedTestAudit {
40
+ criterion: string;
41
+ matches: VitestListedTest[];
42
+ }
43
+ export declare function auditNamedTestOracles(items: readonly AcceptanceItem[], listedTests: readonly VitestListedTest[]): NamedTestAudit[];
31
44
  export type AcceptanceCorpusAuditResult = {
32
45
  specPath: string;
33
46
  status: "parse-failed";
@@ -98,6 +98,47 @@ export function testFiltered(testCmd, name) {
98
98
  const fwd = wrapped ? "-- " : "";
99
99
  return `${testCmd} ${fwd}-t ${shq(pattern)}`;
100
100
  }
101
+ const VitestListedTestsSchema = z.array(z.object({
102
+ name: z.string(),
103
+ file: z.string(),
104
+ projectName: z.string().optional(),
105
+ }));
106
+ export async function listVitestTests(cwd) {
107
+ const result = await sh(`${shq(join(cwd, "node_modules/.bin/vitest"))} list --json`, cwd);
108
+ if (result.code !== 0) {
109
+ return { status: "failed", error: (result.stderr || result.stdout || `exit ${result.code}`).trim() };
110
+ }
111
+ try {
112
+ const start = result.stdout.indexOf("[");
113
+ if (start < 0)
114
+ return { status: "failed", error: "runner emitted no JSON test listing" };
115
+ const parsed = VitestListedTestsSchema.safeParse(JSON.parse(result.stdout.slice(start)));
116
+ return parsed.success
117
+ ? { status: "listed", tests: parsed.data }
118
+ : { status: "failed", error: z.prettifyError(parsed.error) };
119
+ }
120
+ catch (error) {
121
+ return { status: "failed", error: error instanceof Error ? error.message : String(error) };
122
+ }
123
+ }
124
+ export function auditNamedTestOracles(items, listedTests) {
125
+ const runnerNames = listedTests.map((listed) => ({
126
+ listed,
127
+ fullName: listed.name.split(" > ").join(" "),
128
+ }));
129
+ return items.flatMap((item) => {
130
+ if (typeof item !== "object" || item.oracle !== "test")
131
+ return [];
132
+ return [{
133
+ criterion: item.test,
134
+ // OBS-511: mirror the gate's leaf-anchored suffix rule — this denominator must count
135
+ // exactly the tests the shipped -t filter would select.
136
+ matches: runnerNames
137
+ .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
138
+ .map(({ listed }) => listed),
139
+ }];
140
+ });
141
+ }
101
142
  function corpusSpecPaths(root) {
102
143
  const paths = [];
103
144
  const visit = (dir) => {
@@ -116,32 +157,18 @@ function corpusSpecPaths(root) {
116
157
  // listing. Every discovered path contributes either all parser-produced acceptance items or one named
117
158
  // parse failure; exceptions are evidence, never permission to shrink the corpus silently.
118
159
  export function auditAcceptanceCorpus(corpusRoot, listedTests) {
119
- const runnerNames = listedTests.map((listed) => ({
120
- listed,
121
- fullName: listed.name.split(" > ").join(" "),
122
- }));
123
160
  const results = [];
124
161
  for (const specPath of corpusSpecPaths(corpusRoot)) {
125
162
  try {
126
163
  const graph = compileNative(specPath);
127
164
  for (const task of graph.tasks) {
128
165
  for (const item of task.acceptance) {
166
+ const namedTest = auditNamedTestOracles([item], listedTests)[0];
129
167
  results.push({
130
168
  specPath,
131
169
  status: "parsed",
132
170
  item,
133
- ...(typeof item === "object" && item.oracle === "test"
134
- ? {
135
- namedTest: {
136
- criterion: item.test,
137
- // OBS-511: mirror the gate's leaf-anchored suffix rule — the audit's denominator
138
- // must count exactly the tests the -t filter would select, or doctor and gate disagree.
139
- matches: runnerNames
140
- .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
141
- .map(({ listed }) => listed),
142
- },
143
- }
144
- : {}),
171
+ ...(namedTest ? { namedTest } : {}),
145
172
  });
146
173
  }
147
174
  }
@@ -176,11 +176,11 @@ async function coveringTests(worktree, baseRef) {
176
176
  }
177
177
  const covering = tests.filter((t) => reachOf(t).has(file));
178
178
  if (!covering.length)
179
- return undefined; // nothing covers this file — only the full suite can speak for it
179
+ continue; // nothing covers this file — keep every attributable selection already accumulated
180
180
  for (const t of covering)
181
181
  selected.add(t);
182
182
  }
183
- return [...selected].sort();
183
+ return selected.size ? [...selected].sort() : undefined;
184
184
  }
185
185
  /**
186
186
  * The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
@@ -287,7 +287,7 @@ export async function runGates(task, ctx) {
287
287
  }
288
288
  return { results: sorted, commits };
289
289
  };
290
- const toolGates = ["build", "test", "lint"].filter(enabled);
290
+ const toolGates = ["build", "lint"].filter(enabled);
291
291
  /**
292
292
  * v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
293
293
  * the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
@@ -337,28 +337,28 @@ export async function runGates(task, ctx) {
337
337
  + `Uncommitted at round end:\n${dirt}`,
338
338
  meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
339
339
  });
340
- // build/test/lint vs the shared baseline
341
- const runBattery = async (commands, selected) => {
342
- if (!toolGates.length)
340
+ // shell tools vs the shared baseline
341
+ const runBattery = async (commands, selected, gates = toolGates) => {
342
+ if (!gates.length)
343
343
  return;
344
344
  if (!v185) {
345
- // ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
345
+ // ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
346
346
  // not at true execution start. They are collectively sub-second (measured), so the debounce
347
347
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
348
- // ponytail: legacy runs build/test/lint in ONE compareToBaseline call, so there is one interval
348
+ // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
349
349
  // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
350
350
  const batchAt = Date.now();
351
351
  const batchLoadStart = loadProvider();
352
- const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
352
+ const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
353
353
  const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
354
- for (const g of toolGates)
354
+ for (const g of gates)
355
355
  spans.set(g, batch);
356
356
  // The same refusal AFTER the commands, because a green command can dirty the tree the check
357
357
  // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
358
358
  // lands on the last gate that had one — the round dies there either way. A red battery is
359
359
  // reported as the red it is: the round already ends, and the command output is the better lead.
360
360
  const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
361
- const blame = dirt ? [...toolGates].reverse().find((g) => commands[g]) : undefined;
361
+ const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
362
362
  for (const r of toolResults) {
363
363
  await emitStart(r.gate);
364
364
  await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
@@ -366,8 +366,8 @@ export async function runGates(task, ctx) {
366
366
  return;
367
367
  }
368
368
  // T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
369
- // the full vitest suite before anyone reads its verdict.
370
- for (const g of toolGates) {
369
+ // any later tool before anyone reads its verdict.
370
+ for (const g of gates) {
371
371
  await emitStart(g);
372
372
  const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
373
373
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
@@ -437,9 +437,9 @@ export async function runGates(task, ctx) {
437
437
  * screen IS the round's verdict: what it produced is journaled (in the order it ran) and the round
438
438
  * ends there, so a drive-by out-of-scope edit costs <1s instead of the whole battery.
439
439
  *
440
- * A green screen changes nothing downstream. The recorded sequence stays GATE_NAMES order, so the
441
- * journal, `tickmarkr report`, the surfaces, and resume's GATE_NAMES walk over already-satisfied
442
- * gates all keep reading exactly one order.
440
+ * A green screen changes nothing downstream. The returned record stays in GATE_NAMES order while
441
+ * the event stream reports the order gates actually ran; resume's GATE_NAMES walk over satisfied
442
+ * records therefore keeps declaration order without making the live stream lie about execution.
443
443
  *
444
444
  * ponytail: the price of that is re-reading two git checks (~40ms) in their canonical positions
445
445
  * rather than teaching every consumer of the gate stream a second order. Both reads see the same
@@ -447,7 +447,7 @@ export async function runGates(task, ctx) {
447
447
  * for. Charge it only when there IS a battery command to protect.
448
448
  */
449
449
  const screenBlocks = async () => {
450
- if (!toolGates.some((g) => ctx.commands[g]))
450
+ if (!toolGates.some((g) => ctx.commands[g]) && !(enabled("test") && ctx.commands.test))
451
451
  return false;
452
452
  const screened = [];
453
453
  for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
@@ -626,12 +626,7 @@ export async function runGates(task, ctx) {
626
626
  }
627
627
  if (v185 && await screenBlocks())
628
628
  return done();
629
- // A non-final round may run only the tests covering its own diff; the merge-candidate round below
630
- // pays the full suite anyway, so a selection that misses costs a round and can never merge.
631
- const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
632
- ? await coveringTests(ctx.worktree, ctx.baseRef)
633
- : undefined;
634
- await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected);
629
+ await runBattery(ctx.commands);
635
630
  if (failed())
636
631
  return done();
637
632
  if (enabled("evidence")) {
@@ -644,6 +639,14 @@ export async function runGates(task, ctx) {
644
639
  if (failed())
645
640
  return done();
646
641
  }
642
+ // A non-final round may run only the tests covering its own diff; the merge-candidate round below
643
+ // pays the full suite anyway, so a selection that misses costs a round and can never merge.
644
+ const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
645
+ ? await coveringTests(ctx.worktree, ctx.baseRef)
646
+ : undefined;
647
+ await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
648
+ if (failed())
649
+ return done();
647
650
  if (v185 && (enabled("acceptance") || enabled("review"))) {
648
651
  // Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
649
652
  // unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.7",
3
+ "version": "2.1.9",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -83,7 +83,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
83
83
  1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
84
84
  2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
85
85
  3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
86
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal event rather than polling agents — a self-terminating poll (`until grep -q '"event":"run-end"' <state-dir>/runs/<runId>/journal.jsonl; do sleep 20; done`), never `tail -F | grep -m1` (wedges on the journal's final line) and never a pane-level done wait (turn-end flaps). Resolve blocked interactions in the relevant agent session.
86
+ 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
87
87
  5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
88
88
  6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records.
89
89
  7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
@@ -87,6 +87,6 @@ When spawning consultants (agents gathering synthesis input for decisions like S
87
87
  1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
88
88
  2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
89
89
  3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
90
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal event rather than repeatedly polling agents. Use a self-terminating poll — `until grep -q '"event":"run-end"' <state-dir>/runs/<runId>/journal.jsonl; do sleep 20; done` — never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). Resolve blocked interactions in the agent session; do not turn them into proxy questions.
90
+ 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
91
91
  5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
92
92
  6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).