tickmarkr 1.92.1 → 1.96.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -57,7 +57,7 @@ export declare function catalogModelAdvisory(cfg: TickmarkrConfig, catalog: Cata
57
57
  */
58
58
  export declare function hasWindowsConfig(cfg: TickmarkrConfig): boolean;
59
59
  export declare function declaredModelWindow(cfg: TickmarkrConfig, adapter: string, model: string): number | undefined;
60
- /** A numeric estimate exists only when every payload path is measurable from the worker-visible tree. */
60
+ /** A numeric estimate exists only when every context ref is measurable from the worker-visible tree. */
61
61
  export declare function estimateTaskPayloadTokens(task: Task, repoRoot: string, feedback?: string): number | undefined;
62
62
  export type RoutedAssignment = {
63
63
  taskId: string;
@@ -2,6 +2,7 @@ import { spawnSync } from "node:child_process";
2
2
  import { existsSync } from "node:fs";
3
3
  import { join } from "node:path";
4
4
  import { DEFAULT_CONFIG, TIER_RANK } from "../config/config.js";
5
+ import { filesGlob } from "../graph/files-glob.js";
5
6
  import { buildTaskPrompt } from "./prompt.js";
6
7
  import { channelKey, MODEL_ID_RE } from "./types.js";
7
8
  import { resolveCatalogModel } from "./catalog-remote.js";
@@ -194,80 +195,136 @@ export function hasWindowsConfig(cfg) {
194
195
  export function declaredModelWindow(cfg, adapter, model) {
195
196
  return cfg.tiers[adapter]?.windows?.[model];
196
197
  }
198
+ /** Named paths per lint line: an uncapped enumeration reached 4,924 columns on the P99 graph. */
199
+ const MAX_NAMED_PATHS = 3;
197
200
  const GLOB_CHARS = /[*?{[]/;
198
201
  // T13's visibility oracle is the committed base tree: workers are created from HEAD, not from the
199
202
  // author's checkout or index. Keep --full-tree load-bearing for callers below the repository root.
203
+ // `-l` prices every blob in the SAME call (a per-path `cat-file -s` spawn cannot price a pattern),
204
+ // and `-z` is load-bearing correctness, not economy: without it git quotes any non-ASCII path, so
205
+ // every Arabic/UTF-8 tracked file read as absent.
200
206
  function baseTreeAtHead(dir) {
201
207
  const git = (...args) => spawnSync("git", ["-C", dir, ...args], { encoding: "utf8", maxBuffer: 1 << 28 });
202
208
  const top = git("rev-parse", "--show-toplevel");
203
- const tree = git("ls-tree", "--full-tree", "-r", "--name-only", "HEAD");
209
+ const tree = git("ls-tree", "--full-tree", "-r", "-l", "-z", "HEAD");
204
210
  if (top.status !== 0 || tree.status !== 0 || typeof top.stdout !== "string" || typeof tree.stdout !== "string")
205
211
  return undefined;
206
- return { root: top.stdout.trim(), tracked: new Set(tree.stdout.split("\n").filter(Boolean)) };
212
+ const sizes = new Map();
213
+ for (const record of tree.stdout.split("\0")) {
214
+ const tab = record.indexOf("\t");
215
+ if (tab < 0)
216
+ continue;
217
+ // `<mode> <type> <object> <size>\t<path>`; a submodule entry carries `-` as its size and no bytes.
218
+ const bytes = Number.parseInt(record.slice(0, tab).split(/\s+/)[3] ?? "", 10);
219
+ sizes.set(record.slice(tab + 1), Number.isSafeInteger(bytes) && bytes >= 0 ? bytes : 0);
220
+ }
221
+ return { root: top.stdout.trim(), sizes };
207
222
  }
208
- function fileBytes(repoRoot, rel, tree = baseTreeAtHead(repoRoot)) {
209
- if (GLOB_CHARS.test(rel))
210
- return { status: "unreadable", reason: "glob" };
211
- if (!tree)
212
- return { status: "unreadable", reason: "base-tree-unavailable" };
223
+ /**
224
+ * Bytes a worker can read at that path in the base tree, or undefined when the tree carries nothing
225
+ * there. A files[]-shaped pattern and a directory both price as everything they cover — matching
226
+ * through src/graph/files-glob.ts so the estimate reads patterns exactly as the scope gate does.
227
+ */
228
+ function measureBytes(tree, rel) {
213
229
  const path = rel.replace(/\/+$/, "");
214
- if (!tree.tracked.has(path)) {
215
- if ([...tree.tracked].some((tracked) => tracked.startsWith(`${path}/`))) {
216
- return { status: "unreadable", reason: "directory" };
217
- }
218
- return { status: "unreadable", reason: existsSync(join(tree.root, path)) ? "untracked" : "absent" };
230
+ const exact = tree.sizes.get(path);
231
+ if (exact !== undefined)
232
+ return exact;
233
+ const match = GLOB_CHARS.test(path) ? filesGlob(path) : undefined;
234
+ let total = 0;
235
+ let covered = false;
236
+ for (const [tracked, bytes] of tree.sizes) {
237
+ if (!(match ? match(tracked) : tracked.startsWith(`${path}/`)))
238
+ continue;
239
+ total += bytes;
240
+ covered = true;
219
241
  }
220
- // Read the blob size from HEAD too: an unstaged/staged checkout edit is not what the worker receives.
221
- const size = spawnSync("git", ["-C", tree.root, "cat-file", "-s", `HEAD:${path}`], { encoding: "utf8" });
222
- const bytes = Number.parseInt(typeof size.stdout === "string" ? size.stdout.trim() : "", 10);
223
- return size.status === 0 && Number.isSafeInteger(bytes) && bytes >= 0
224
- ? { status: "measured", bytes }
225
- : { status: "unreadable", reason: "base-tree-read-failed" };
242
+ return covered ? total : undefined;
226
243
  }
227
- function taskPayloadResolution(task, repoRoot, feedback) {
228
- let bytes = buildTaskPrompt(task, feedback).length;
229
- const unreadable = [];
230
- const tree = baseTreeAtHead(repoRoot);
231
- const add = (field, path) => {
232
- const measured = fileBytes(repoRoot, path, tree);
233
- if (measured.status === "measured")
234
- bytes += measured.bytes;
235
- else
236
- unreadable.push({ field, path, reason: measured.reason });
244
+ function absentContextReason(root, path) {
245
+ if (spawnSync("git", ["-C", root, "check-ignore", "-q", "--", path]).status === 0)
246
+ return "gitignored";
247
+ return existsSync(join(root, path)) ? "untracked" : "absent";
248
+ }
249
+ function upstreamProducer(tasks) {
250
+ const byId = new Map(tasks.map((t) => [t.id, t]));
251
+ const cache = new Map();
252
+ const upstream = (taskId) => {
253
+ let entry = cache.get(taskId);
254
+ if (!entry) {
255
+ const seen = new Set();
256
+ const walk = (from) => {
257
+ for (const dep of byId.get(from)?.deps ?? [])
258
+ if (!seen.has(dep)) {
259
+ seen.add(dep);
260
+ walk(dep);
261
+ }
262
+ };
263
+ walk(taskId);
264
+ entry = [...seen].map((id) => {
265
+ const files = byId.get(id)?.files ?? [];
266
+ return { id, covers: files.length ? filesGlob(files) : () => false };
267
+ });
268
+ cache.set(taskId, entry);
269
+ }
270
+ return entry;
237
271
  };
238
- for (const path of task.context)
239
- add("context", path);
272
+ return (taskId, path) => upstream(taskId).find((dep) => dep.covers(path))?.id;
273
+ }
274
+ function taskPayloadResolution(task, repoRoot, feedback, tree = baseTreeAtHead(repoRoot), producerOf = () => undefined) {
275
+ let bytes = buildTaskPrompt(task, feedback).length;
276
+ if (!tree) {
277
+ return { tokens: Math.ceil(bytes / CHARS_PER_TOKEN), unmeasurable: [{ path: ".", reason: "base-tree-unavailable" }] };
278
+ }
279
+ // files[] is WRITE SCOPE, not payload: a task's own output (`99-04-SUMMARY.md`) is absent from the
280
+ // base tree BY CONSTRUCTION and a scope entry is a set, not a document. Price what the worker can
281
+ // already read there and charge nothing for the rest. Billing an output path as a measurement
282
+ // failure is what made the window comparison unreachable on every real graph — 41 of 41 tasks on
283
+ // the P99 graph reported "payload unreadable" and none was ever compared to its model's window.
240
284
  for (const path of task.files)
241
- add("files", path);
242
- return { tokens: Math.ceil(bytes / CHARS_PER_TOKEN), unreadable };
285
+ bytes += measureBytes(tree, path) ?? 0;
286
+ const unmeasurable = [];
287
+ for (const path of task.context) {
288
+ const measured = measureBytes(tree, path);
289
+ if (measured !== undefined) {
290
+ bytes += measured;
291
+ continue;
292
+ }
293
+ const producer = producerOf(task.id, path);
294
+ unmeasurable.push(producer
295
+ ? { path, reason: "produced-upstream", producer }
296
+ : { path, reason: absentContextReason(tree.root, path) });
297
+ }
298
+ return { tokens: Math.ceil(bytes / CHARS_PER_TOKEN), unmeasurable };
243
299
  }
244
- /** A numeric estimate exists only when every payload path is measurable from the worker-visible tree. */
300
+ /** A numeric estimate exists only when every context ref is measurable from the worker-visible tree. */
245
301
  export function estimateTaskPayloadTokens(task, repoRoot, feedback = "") {
246
302
  const estimate = taskPayloadResolution(task, repoRoot, feedback);
247
- return estimate.unreadable.length === 0 ? estimate.tokens : undefined;
303
+ return estimate.unmeasurable.length === 0 ? estimate.tokens : undefined;
248
304
  }
249
- function unreadablePayloadLint(taskId, unreadable) {
250
- const describe = ({ field, path, reason }) => {
251
- const prefix = `${field} ${JSON.stringify(path)}`;
252
- if (reason === "untracked")
253
- return `${prefix} is present in the checkout but not in the base tree`;
254
- if (reason === "absent")
255
- return `${prefix} is absent from the base tree and checkout`;
256
- if (reason === "glob")
257
- return `${prefix} is a glob and cannot be measured`;
258
- if (reason === "directory")
259
- return `${prefix} is a directory and cannot be measured as one file`;
305
+ function unreachableContextLint(taskId, dangling) {
306
+ const describe = ({ path, reason }) => {
260
307
  if (reason === "base-tree-unavailable")
261
- return `${prefix} cannot be checked because the base tree is unavailable`;
262
- return `${prefix} could not be read from the base tree`;
308
+ return "the base tree is unavailable, so no context ref can be checked";
309
+ const named = JSON.stringify(path);
310
+ if (reason === "gitignored")
311
+ return `${named} is gitignored — no commit can carry it into a worktree`;
312
+ if (reason === "untracked")
313
+ return `${named} is present in the checkout but not in the base tree`;
314
+ return `${named} is absent from the base tree and no task in this graph produces it`;
263
315
  };
264
- return `${taskId}: payload unreadable — ${unreadable.map(describe).join(", ")}; context-window comparison skipped`;
316
+ const rest = dangling.length - MAX_NAMED_PATHS;
317
+ return `${taskId}: context unreachable — ${dangling.slice(0, MAX_NAMED_PATHS).map(describe).join(", ")}`
318
+ + `${rest > 0 ? `, +${rest} more` : ""}; the worker's "read these first" list dangles`
319
+ + " and the context-window comparison is skipped";
265
320
  }
266
321
  /** Advisory only — absent windows config or undeclared model window ⇒ no lint. */
267
322
  export function contextWindowLints(tasks, assignments, cfg, repoRoot) {
268
323
  if (!hasAnyWindows(cfg))
269
324
  return [];
270
325
  const byId = new Map(assignments.map((a) => [a.taskId, a]));
326
+ const tree = baseTreeAtHead(repoRoot);
327
+ const producerOf = upstreamProducer(tasks);
271
328
  const lints = [];
272
329
  for (const t of tasks) {
273
330
  const a = byId.get(t.id);
@@ -276,13 +333,20 @@ export function contextWindowLints(tasks, assignments, cfg, repoRoot) {
276
333
  const window = declaredModelWindow(cfg, a.adapter, a.model);
277
334
  if (window === undefined)
278
335
  continue;
279
- const estimate = taskPayloadResolution(t, repoRoot, "");
280
- if (estimate.unreadable.length > 0) {
281
- lints.push(unreadablePayloadLint(t.id, estimate.unreadable));
336
+ const estimate = taskPayloadResolution(t, repoRoot, "", tree, producerOf);
337
+ const dangling = estimate.unmeasurable.filter((u) => u.reason !== "produced-upstream");
338
+ if (dangling.length > 0) {
339
+ lints.push(unreachableContextLint(t.id, dangling));
282
340
  continue;
283
341
  }
342
+ // An upstream-produced ref is unmeasurable but REACHABLE — the run writes it before this task
343
+ // branches — so it earns no lint of its own: silence on the benign class is what leaves a real
344
+ // dangling ref visible. The estimate is then a lower bound, and a lower bound over the window is
345
+ // still an overflow, so it is reported with what it excludes named.
284
346
  if (estimate.tokens > window) {
285
- lints.push(`${t.id}: payload ~${estimate.tokens} tokens exceeds ${a.adapter}:${a.model} window ${window}`);
347
+ const pending = estimate.unmeasurable.length;
348
+ lints.push(`${t.id}: payload ~${estimate.tokens} tokens exceeds ${a.adapter}:${a.model} window ${window}`
349
+ + (pending > 0 ? ` (lower bound — ${pending} context ref(s) are produced upstream and not yet measurable)` : ""));
286
350
  }
287
351
  }
288
352
  return lints;
@@ -0,0 +1 @@
1
+ export declare function beat(argv: string[], cwd?: string): Promise<string>;
@@ -0,0 +1,50 @@
1
+ import { mkdirSync, renameSync, rmSync, writeFileSync } from "node:fs";
2
+ import { dirname } from "node:path";
3
+ import { tickmarkrDir } from "../../graph/graph.js";
4
+ import { SUPERVISION_BEAT_MS, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSupervision, supervisionStandDownPath, } from "../../run/supervision.js";
5
+ // SUP-04: the writer side of supervision, as a VERB. `beatSupervision` and `SUPERVISION_BEAT_MS` shipped
6
+ // with exactly one in-repo caller — the daemon, on one tier — so `status` printed
7
+ // `orchestrator ARMED / overseer ABSENT / watch ABSENT` while a real overseer worked the run: two thirds
8
+ // of a supervision claim were constants dressed as measurements. A seat that supervises from a shell
9
+ // (the overseer skill's watcher loop) has no way into a node module; a command does.
10
+ //
11
+ // ONE-SHOT AND ERROR-PROPAGATING, deliberately not `armSupervision`. That helper is the right shape for
12
+ // a long-lived node process and the wrong shape here twice over. Its recurring interval would outlive
13
+ // the invocation — in a long-lived dispatcher a later `--stand-down` would disarm only its own timer
14
+ // while an earlier one kept writing, flipping DISARMED back to ARMED ten seconds later: the exact
15
+ // fail-open this instrument exists to catch. And it swallows write failures by design (an unwritten
16
+ // beat ages out, which is the truth for a watcher that must not be crashed by its own instrument) —
17
+ // but a CLI that swallows them PRINTS a state nothing recorded, so `beat` says ARMED while `status`
18
+ // reads UNREADABLE. Here the writes are unguarded: a failure leaves the process with a non-zero exit
19
+ // and a message instead of a claim. The LOOP belongs to the caller, which is what makes the beat
20
+ // evidence: stop calling it and the tier ages into STALE on its own.
21
+ export async function beat(argv, cwd = process.cwd()) {
22
+ const standDown = argv.includes("--stand-down");
23
+ const named = argv.find((a) => !a.startsWith("--"));
24
+ if (!isTier(named)) {
25
+ throw new Error(`usage: tickmarkr beat <${SUPERVISION_TIERS.join("|")}> [--stand-down] — got ${named ? `\`${named}\`` : "no tier"}`);
26
+ }
27
+ if (standDown)
28
+ return standDownTier(cwd, named);
29
+ // Clear a stand-down marker left by an earlier session BEFORE beating: a valid marker outranks every
30
+ // beat that does not strictly follow it, and two writes landing in the same millisecond do not. An
31
+ // uncleared marker would render DISARMED while this verb claimed ARMED, so the removal is unguarded
32
+ // too — `force` makes the ordinary "no marker" case a no-op, and anything else is a real failure.
33
+ rmSync(supervisionStandDownPath(cwd, named), { force: true, recursive: true });
34
+ beatSupervision(cwd, named);
35
+ return `${named} ARMED — beat again every ${SUPERVISION_BEAT_MS / 1_000}s; the tier reads STALE ${SUPERVISION_STALE_MS / 1_000}s after the last beat`;
36
+ }
37
+ // Stand-down is a RECORDED act, not a silence: the marker tells a reader this watcher left on purpose,
38
+ // so the tier reads DISARMED rather than ageing out as a death. Published atomically — written aside,
39
+ // renamed over — because a torn marker is rejected by the reader, and a rejected stand-down reports a
40
+ // deliberate hand-off as a death.
41
+ function standDownTier(repoRoot, tier) {
42
+ tickmarkrDir(repoRoot); // the write path DOES create — markers land inside the gitignored state dir
43
+ const p = supervisionStandDownPath(repoRoot, tier);
44
+ const tmp = `${p}.${process.pid}.tmp`;
45
+ mkdirSync(dirname(p), { recursive: true });
46
+ writeFileSync(tmp, JSON.stringify({ tier, pid: process.pid, disarmedAt: new Date().toISOString() }) + "\n");
47
+ renameSync(tmp, p);
48
+ return `${tier} DISARMED — hand-off recorded; status reads it stood down, not dead`;
49
+ }
50
+ const isTier = (v) => SUPERVISION_TIERS.includes(v ?? "");
@@ -8,6 +8,7 @@ import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../.
8
8
  import { GATE_NAMES } from "../../graph/schema.js";
9
9
  import { foldActivity } from "../../run/activity.js";
10
10
  import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFailureKind, runHasEnded, } from "../../run/journal.js";
11
+ import { normalizeGateOutcome } from "../../run/outcome.js";
11
12
  import { desiredPanes } from "../../run/reconcile.js";
12
13
  import { normalizeStallSnapshot } from "../../run/stall.js";
13
14
  import { readSupervision, supervisionText } from "../../run/supervision.js";
@@ -216,6 +217,18 @@ const livePhases = (events) => {
216
217
  const active = live.get(event.taskId);
217
218
  if (!active)
218
219
  continue;
220
+ // The daemon proves a silent worker alive by hashing its worktree and journals that proof as
221
+ // `worker-contact`. It is evidence ABOUT the live phase, never its outcome, so it decorates the
222
+ // entry and is deliberately absent from the `matched` set below.
223
+ if (event.event === "worker-contact" && active.phase === "worker") {
224
+ const at = Date.parse(event.ts);
225
+ if (Number.isFinite(at)) {
226
+ active.lastContactAt = at;
227
+ if (typeof event.data.evidence === "string")
228
+ active.contactEvidence = event.data.evidence;
229
+ }
230
+ continue;
231
+ }
219
232
  const gate = active.gate ?? phaseGate(active.phase);
220
233
  const matched = (active.phase === "worker" && event.event === "worker-result")
221
234
  || (active.phase === "gates" && event.event === "gate-result")
@@ -249,16 +262,44 @@ const fmtElapsed = (ms) => {
249
262
  const hours = Math.floor(minutes / 60);
250
263
  return `${hours}h${minutes % 60}m${seconds % 60}s`;
251
264
  };
252
- const workerOutputAge = (phase, now, worker) => now - (worker?.phaseStartedAt === phase.startedAt && worker.hasOutput ? worker.lastOutputAt : phase.startedAt);
265
+ /**
266
+ * The freshest evidence that this worker is alive, and what produced it. ONE derivation, because the
267
+ * two readers of worker silence drifted apart the first time this was fixed: OBS-538 taught the detail
268
+ * phrase to read the daemon's journaled `worker-contact` and left the stall ALARM (`staleWorker`)
269
+ * clocking pane bytes alone. For the exact population OBS-538 is about — a headless channel (codex
270
+ * `sub`, an API worker) that prints nothing to its pane all run, so `hasOutput` never becomes true —
271
+ * the alarm then warn-painted every live row from 60s onward while line 2 of the same card read
272
+ * "last contact 10s ago (worktree)". A card cannot contradict itself if both halves ask this.
273
+ */
274
+ const workerActivity = (phase, worker) => {
275
+ const outputAt = worker?.phaseStartedAt === phase.startedAt && worker.hasOutput ? worker.lastOutputAt : undefined;
276
+ const contactAt = phase.lastContactAt;
277
+ if (outputAt !== undefined && (contactAt === undefined || outputAt >= contactAt))
278
+ return { at: outputAt, source: "output" };
279
+ if (contactAt !== undefined)
280
+ return { at: contactAt, source: "contact" };
281
+ // Nothing has been observed since the phase began: silence is the whole phase, which is what both
282
+ // the phrase ("no output 14m51s") and the alarm's 60s threshold must measure.
283
+ return { at: phase.startedAt, source: "none" };
284
+ };
285
+ /** How long this worker has been silent by the freshest evidence available — the alarm's clock. */
286
+ const workerSilenceMs = (phase, now, worker) => now - workerActivity(phase, worker).at;
287
+ /**
288
+ * What a live worker row says about silence. Pane bytes are not more truthful than a worktree delta,
289
+ * only different, so the row names whichever evidence is fresher AND names its source.
290
+ */
253
291
  const phaseDetail = (phase, now, worker) => {
254
292
  const elapsed = fmtElapsed(now - phase.startedAt);
255
293
  if (phase.phase !== "worker")
256
294
  return `${phaseLabel(phase)} · ${elapsed} elapsed`;
257
- const age = fmtElapsed(workerOutputAge(phase, now, worker));
258
- const output = worker?.phaseStartedAt === phase.startedAt && worker.hasOutput
259
- ? `last output ${age} ago`
260
- : `no output ${age}`;
261
- return `${phaseLabel(phase.phase)} · ${elapsed} elapsed · ${output}`;
295
+ const head = `${phaseLabel(phase.phase)} · ${elapsed} elapsed`;
296
+ const activity = workerActivity(phase, worker);
297
+ const age = fmtElapsed(now - activity.at);
298
+ if (activity.source === "output")
299
+ return `${head} · last output ${age} ago`;
300
+ if (activity.source === "contact")
301
+ return `${head} · last contact ${age} ago (${phase.contactEvidence ?? "worktree"})`;
302
+ return `${head} · no output ${age}`;
262
303
  };
263
304
  const attemptStartIdx = (events, taskId) => {
264
305
  let idx = -1;
@@ -358,6 +399,8 @@ export const shortGoal = (goal, max) => {
358
399
  return fitCells(clause, cells).trimEnd();
359
400
  return `${fitCells(clause, cells - 3).trimEnd()}...`;
360
401
  };
402
+ /** Columns the machine row always spends on the task title, however long its machinery segment runs. */
403
+ const MACHINE_TITLE_FLOOR = 24;
361
404
  const detailText = (segments, budget) => {
362
405
  const shedOrder = LAYOUT_PRIORITY.slice(LAYOUT_PRIORITY.indexOf("journal") + 1).reverse();
363
406
  const shed = new Set();
@@ -378,6 +421,95 @@ const detailText = (segments, budget) => {
378
421
  * stay blank — `wrapCells` always answers at least one row.
379
422
  */
380
423
  const boardRows = (lines, columns) => lines.flatMap((line) => wrapCells(line, columns, { continuationPrefix: " " }));
424
+ /**
425
+ * A funded review round is a review that actually returned a verdict. `passed` and `failed` are the
426
+ * only two arms of the gate vocabulary that carry one (outcome.ts:23) — every other arm states why
427
+ * nothing ran, so counting it would bill the task for a reviewer it never spent.
428
+ */
429
+ const REVIEW_VERDICTS = new Set(["passed", "failed"]);
430
+ /**
431
+ * Fold the panel from parsed JournalEvent rows only. In particular, a review round is every
432
+ * gate-result whose typed gate field is `review` and whose outcome is a verdict, passed or failed.
433
+ * `review-raw-*` artifacts are unparseable-verdict captures, not round markers, and neither raw
434
+ * journal strings nor run-directory filenames participate in this count.
435
+ *
436
+ * A decline is not a round, and it is NOT readable off `data.skipped`: that field is one of four
437
+ * encodings the shipped record already writes, and the two it misses are both live on disk — a
438
+ * pre-R3 review decline states itself only as `{ pass: true, details: "skipped — …" }`, and a
439
+ * canonical row states it as `outcome.kind: "declined"`. Reading the flag alone bills both as funded
440
+ * rounds and can reorder the top four. `normalizeGateOutcome` is the ONE reader of all four
441
+ * encodings (outcome.ts:125), so the panel asks it rather than re-deriving a fifth answer; a row it
442
+ * cannot read stays `unavailable` and therefore uncounted, fail-closed. `replayMeasurement` is
443
+ * orthogonal to the outcome — a resume re-observing an interrupted round reports a real verdict —
444
+ * so it stays an explicit exclusion, the same clause journal.ts:110 uses.
445
+ */
446
+ const foldTaskEffort = (tasks, events) => {
447
+ const order = new Map(tasks.map((task, index) => [task.id, index]));
448
+ const effort = new Map(tasks.map((task) => [task.id, {
449
+ taskId: task.id,
450
+ dispatches: 0,
451
+ reviews: 0,
452
+ parks: 0,
453
+ total: 0,
454
+ }]));
455
+ for (const event of events) {
456
+ if (!event.taskId)
457
+ continue;
458
+ const task = effort.get(event.taskId);
459
+ if (!task)
460
+ continue;
461
+ if (event.event === "task-dispatch")
462
+ task.dispatches += 1;
463
+ else if (event.event === "gate-result" && event.data.gate === "review"
464
+ && event.data.replayMeasurement !== true
465
+ && REVIEW_VERDICTS.has(normalizeGateOutcome(event.data.outcome ?? event.data).kind))
466
+ task.reviews += 1;
467
+ else if (event.event === "task-human")
468
+ task.parks += 1;
469
+ task.total = task.dispatches + task.reviews + task.parks;
470
+ }
471
+ return [...effort.values()].sort((left, right) => right.total - left.total
472
+ || order.get(left.taskId) - order.get(right.taskId));
473
+ };
474
+ /**
475
+ * Prototype panel, fitted by the cockpit's display-cell authority before board-wide wrapping.
476
+ *
477
+ * `undefined` events mean this run's journal describes a DIFFERENT graph, so it funds no claim here.
478
+ * Folding it — or folding an empty array in its place — would state zero dispatches, reviews and
479
+ * parks for a run that recorded all three, which is unavailable evidence dressed as an answer. Same
480
+ * clause the verify claim uses (`verify unavailable`): no claim beats a false zero.
481
+ */
482
+ const effortPanel = (tasks, events, columns) => {
483
+ if (!events) {
484
+ return [
485
+ ` ${title("WHERE THE EFFORT WENT")}`,
486
+ legend(` dispatch · review · park counts unavailable — ${NOT_COMPARABLE_CLAIM}`),
487
+ ];
488
+ }
489
+ const top = foldTaskEffort(tasks, events).slice(0, 4);
490
+ const maxTotal = Math.max(1, ...top.map((task) => task.total));
491
+ const idColumns = Math.max(2, Math.min(16, Math.max(...top.map((task) => cellWidth(task.taskId)))));
492
+ const rows = top.map((task) => {
493
+ const counts = `${task.dispatches} dispatch · ${task.reviews} review · ${task.parks} park`;
494
+ const prefix = ` ${fitCells(task.taskId, idColumns)} `;
495
+ const barColumns = Math.min(46, Math.max(3, columns - cellWidth(prefix) - 2 - cellWidth(counts)));
496
+ const dispatchEnd = Math.round((task.dispatches / maxTotal) * barColumns);
497
+ const reviewEnd = Math.round(((task.dispatches + task.reviews) / maxTotal) * barColumns);
498
+ const parkEnd = Math.round((task.total / maxTotal) * barColumns);
499
+ const stack = ok("█".repeat(dispatchEnd))
500
+ + warn("█".repeat(Math.max(0, reviewEnd - dispatchEnd)))
501
+ + fail("█".repeat(Math.max(0, parkEnd - reviewEnd)));
502
+ return `${prefix}${fitCells(stack, barColumns)} ${counts}`;
503
+ });
504
+ return [
505
+ ` ${title("WHERE THE EFFORT WENT")}`,
506
+ legend(` dispatches · review rounds · human parks · top ${top.length} of ${tasks.length} tasks`),
507
+ "",
508
+ ...rows,
509
+ "",
510
+ ` ${ok("█")} ${dim("dispatch")} ${warn("█")} ${dim("review round")} ${fail("█")} ${dim("human park")}`,
511
+ ];
512
+ };
381
513
  // VIS-11 (v1.13): a liveness header for renderFrame — last journal event age + whether the recorded
382
514
  // daemon pid is still alive. Honest about unknowns: a pre-v1.13 journal with no pid renders "unknown",
383
515
  // never fabricated (garbage pid data fails toward unknown too). kill(pid,0) is a signal probe, not a
@@ -724,6 +856,17 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
724
856
  // (a recompiled graph's journal must not animate the wrong tasks); with no or stale journal the
725
857
  // dep-waiting cells still derive from the effective graph statuses.
726
858
  const activity = foldActivity(comparable ? events : [], effective.tasks);
859
+ // RULING-v189-scope §14c: a card answers BOTH supervisor questions. `foldActivity` names what a
860
+ // task waits FOR; this names what waits ON it. Read from `effective.tasks` — whose statuses are
861
+ // the journal's replay, never the compiled graph's — so a dependent the record says is done stops
862
+ // being named, and a task all of whose dependents are done acquires no entry at all.
863
+ const dependents = new Map();
864
+ for (const task of effective.tasks) {
865
+ if (task.status === "done")
866
+ continue;
867
+ for (const dep of task.deps)
868
+ dependents.set(dep, [...(dependents.get(dep) ?? []), task.id]);
869
+ }
727
870
  const unicode = visual();
728
871
  const divider = unicode ? " · " : " / ";
729
872
  // SUP-01: derived by stat() alone, from THIS frame's clock. status stays a reader — nothing here
@@ -783,7 +926,12 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
783
926
  const chain = gateChain(states, false);
784
927
  const prefix = livePhase ? ` ${ASCII_SPINNER[animationFrame % ASCII_SPINNER.length]} ${t.id} ` : ` ${surfaceTaskBox(st, merged)} ${t.id} `;
785
928
  const suffix = ` ${chain}${priorGraph ? ` ${PRIOR_GRAPH_MARKER}` : ""} ${livePhase ? "running" : surfaceStatusWord(st)}${label} ${assignCol}${pane ? ` pane ${pane}` : ""}`;
786
- return `${prefix}${shortGoal(t.title, Math.max(0, width - prefix.length - suffix.length))}${suffix}`;
929
+ // The machine row carries the cockpit's own law (see LAYOUT_PRIORITY and the two-line card):
930
+ // identity is never what a crowded row sheds. `dep-waiting on …` grows with fan-in and alone
931
+ // can exceed the terminal width, and the pre-floor math paid for it out of the TITLE — 20 of
932
+ // 41 rows in a live 41-task run named no task at all. These rows already overflow `width`
933
+ // (a pane name is 60 columns on its own), so the floor costs wrapping, never the graph.
934
+ return `${prefix}${shortGoal(t.title, Math.max(MACHINE_TITLE_FLOOR, width - prefix.length - suffix.length))}${suffix}`;
787
935
  });
788
936
  const zone = journalRowsOnly ? `${divider}zone ${localZoneLabel(zoneReference)}` : "";
789
937
  const header = runId
@@ -866,8 +1014,10 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
866
1014
  const rows = cells.map((c, i) => {
867
1015
  const { t, st, merged, failureKind, redTier, states, priorGraph, isStarved, phrase, channel, ctx, livePhase, pane } = c;
868
1016
  const word = statusWord(c);
1017
+ // The alarm reads the SAME evidence the detail phrase names — a worktree delta the daemon proved
1018
+ // counts, or the card would flag a worker it just called alive one line below (OBS-538 review).
869
1019
  const staleWorker = livePhase?.phase === "worker"
870
- && workerOutputAge(livePhase, now, workerLiveness.get(t.id)) >= 60_000;
1020
+ && workerSilenceMs(livePhase, now, workerLiveness.get(t.id)) >= 60_000;
871
1021
  const stWord = staleWorker ? warn(word) : merged ? ok(word) : redTier ? fail(word) : st === "failed" || st === "human" ? warn(word) : word;
872
1022
  // a fail names its gate in words right here — the one moment gate identity is needed on a row
873
1023
  const f = failedGates(states);
@@ -883,6 +1033,8 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
883
1033
  : ` ${statusRow(taskVerdict(c), taskLabel)}`;
884
1034
  // activity already names its channel for in-flight attempts — never repeat it
885
1035
  const chain = gateChain(states, true);
1036
+ // A finished task blocks nobody — the same journal-folded done-ness the map was built on.
1037
+ const blocking = graphTaskStatus(st, t.status) === "done" ? [] : dependents.get(t.id) ?? [];
886
1038
  const segments = [
887
1039
  ...(priorGraph ? [{ element: "progressCaption", text: PRIOR_GRAPH_MARKER }] : []),
888
1040
  ...(phrase
@@ -893,6 +1045,10 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
893
1045
  : [{ element: "secondaryHeader", text: channel }]),
894
1046
  ]
895
1047
  : [{ element: "secondaryHeader", text: channel }]),
1048
+ // The waiting phrase's reverse direction, and structure for the same reason: a supervisor
1049
+ // deciding what to unblock needs the names, so it wraps beside the blockers rather than being
1050
+ // shed or folded into a count. A finished task blocks nobody.
1051
+ ...(blocking.length ? [{ element: "journal", text: `blocks ${blocking.join(", ")}` }] : []),
896
1052
  ...(failureKind && !phrase?.includes(failureKind)
897
1053
  ? [{ element: "progressCaption", text: failureKind }]
898
1054
  : []),
@@ -909,8 +1065,9 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
909
1065
  }).flat();
910
1066
  if (!nowLine.length)
911
1067
  rows.pop(); // cards end blank-separated; drop the dangling one
1068
+ const effort = effortPanel(g.tasks, record && !comparable ? undefined : events, boardColumns);
912
1069
  return {
913
- content: boardRows([header, hr, gatesLegend, supervisionLegend, "", ...rows, ...nowLine], boardColumns).join("\n"),
1070
+ content: boardRows([header, hr, gatesLegend, supervisionLegend, "", ...rows, ...nowLine, "", hr, "", ...effort], boardColumns).join("\n"),
914
1071
  ...(hotPhase ? { hotPhase } : {}),
915
1072
  ...(runId ? { runId } : {}),
916
1073
  workerPhases: [...phases.values()].filter((phase) => phase.phase === "worker"),
@@ -42,6 +42,13 @@ export function parseCriteria(text) {
42
42
  // verify models the author as a "human" vendor channel — resolvable, excludes nothing real.
43
43
  export const HUMAN_CHANNEL = { adapter: "human", vendor: "human", model: "human", channel: "sub", tier: "frontier" };
44
44
  export const HUMAN_AUTHOR = { adapter: "human", model: "human", channel: "sub", tier: "frontier" };
45
+ // The battery's own dirty-worktree refusal, mirrored from run-gates.ts:220-232 — that check lives in
46
+ // a closure this command cannot reach, and verify may not reshape it. run-gates stays the authority:
47
+ // it re-checks at round entry and after every gate command, so a copy that ever drifted could only
48
+ // refuse EARLY with a stale sentence — never let a dirty tree through.
49
+ const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
50
+ + `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
51
+ + `and never merged (and the committed diff would never be run)`;
45
52
  // GATE-FIX-4 defect 1 (false-RED on macOS): os.tmpdir() returns /var/folders/…, a symlink into
46
53
  // /private/var — so a baseline captured under the repo path and a head battery run under the tmp
47
54
  // path disagree on every path-bearing fingerprint, and verify reds a green diff. graph.ts's
@@ -105,6 +112,51 @@ export async function verify(argv, cwd = process.cwd()) {
105
112
  evidence: { commits: [], artifacts: [], gateResults: [] },
106
113
  };
107
114
  const commands = detectGateCommands(cwd, cfg);
115
+ // PRECONDITIONS (OBS-541) — every check that can refuse this candidate, evaluated together and
116
+ // BEFORE the baseline capture below. Both read cheap local state (one `git status`, the doctor
117
+ // cache), and both used to be read AFTER the capture: the dirty tree by runGates' own round-entry
118
+ // refusal, the review seat by the resolution that sat under it. That cost one full capture per
119
+ // refusal — measured at 602s and 590s on two refusals of the same candidate — to learn something
120
+ // knowable in 50ms. Messages, exit taxonomy and fail-closed semantics are unchanged; only the
121
+ // order is. Anything else that can refuse before a gate runs belongs in this phase, above capture.
122
+ // `--untracked-files=all` is load-bearing twice over: it overrides a repo/user
123
+ // `status.showUntrackedFiles=no` (under which untracked work is INVISIBLE and a dirty tree would
124
+ // capture and gate GREEN), and it enumerates nested files individually instead of collapsing them
125
+ // to a bare `?? dir/`, so the refusal names every offending path. The `.tickmarkr-*` exemption is
126
+ // unaffected — the harness's droppings are root-level files, never directories.
127
+ const status = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain --untracked-files=all", cwd);
128
+ const dirt = status.code !== 0
129
+ ? `git status failed (exit ${status.code}) — the worktree cannot be proven clean`
130
+ : status.stdout.split("\n").map((l) => l.trimEnd())
131
+ .filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l)).join("\n");
132
+ if (dirt)
133
+ throw new Error(`${DIRTY_WHY}:\n${dirt}`);
134
+ // LLM seats only when a semantic gate will run.
135
+ let channels = [];
136
+ let judgeChannels;
137
+ let author = HUMAN_AUTHOR;
138
+ const adapters = allAdapters();
139
+ if (wantAcceptance || wantReview) {
140
+ const health = readDoctor(cwd) ?? (await probeAll(adapters));
141
+ const pools = rolePools(cfg, adapters, health);
142
+ judgeChannels = pools.judge;
143
+ channels = pools.review;
144
+ if (values.author && values.author !== "human") {
145
+ const [adapter, ...rest] = values.author.split(":");
146
+ const model = rest.join(":");
147
+ const c = channels.find((ch) => ch.adapter === adapter && ch.model === model);
148
+ if (!c) {
149
+ throw new Error(`--author ${values.author} does not name a discoverable review channel — one of: ${channels.map(channelKey).join(", ") || "(none)"}`);
150
+ }
151
+ author = { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier };
152
+ }
153
+ else {
154
+ channels = [...channels, HUMAN_CHANNEL];
155
+ }
156
+ if (wantReview && !channels.some((c) => c.vendor !== "human")) {
157
+ throw new Error("review gate needs at least one authed LLM channel (run `tickmarkr doctor`) — or pass --no-review");
158
+ }
159
+ }
108
160
  // Baseline: --baseline file > cached capture for this merge-base > fresh capture on a detached
109
161
  // temp worktree of the merge-base (so pre-existing failures on base are forgiven, exactly as a run).
110
162
  // ALL verify state (cache, base worktree, artifacts) lives OUTSIDE the repo: verify gates the repo
@@ -135,32 +187,6 @@ export async function verify(argv, cwd = process.cwd()) {
135
187
  await removeWorktree(cwd, baseDir);
136
188
  }
137
189
  }
138
- // LLM seats only when a semantic gate will run.
139
- let channels = [];
140
- let judgeChannels;
141
- let author = HUMAN_AUTHOR;
142
- const adapters = allAdapters();
143
- if (wantAcceptance || wantReview) {
144
- const health = readDoctor(cwd) ?? (await probeAll(adapters));
145
- const pools = rolePools(cfg, adapters, health);
146
- judgeChannels = pools.judge;
147
- channels = pools.review;
148
- if (values.author && values.author !== "human") {
149
- const [adapter, ...rest] = values.author.split(":");
150
- const model = rest.join(":");
151
- const c = channels.find((ch) => ch.adapter === adapter && ch.model === model);
152
- if (!c) {
153
- throw new Error(`--author ${values.author} does not name a discoverable review channel — one of: ${channels.map(channelKey).join(", ") || "(none)"}`);
154
- }
155
- author = { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier };
156
- }
157
- else {
158
- channels = [...channels, HUMAN_CHANNEL];
159
- }
160
- if (wantReview && !channels.some((c) => c.vendor !== "human")) {
161
- throw new Error("review gate needs at least one authed LLM channel (run `tickmarkr doctor`) — or pass --no-review");
162
- }
163
- }
164
190
  const artifactDir = join(stateDir, new Date().toISOString().replace(/[:.]/g, "-"));
165
191
  mkdirSync(artifactDir, { recursive: true });
166
192
  const { results } = await runGates(task, {