tickmarkr 2.5.6 → 2.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/adapters/prompt.js +21 -1
  2. package/dist/cli/commands/approve.js +5 -5
  3. package/dist/cli/commands/status.js +95 -34
  4. package/dist/cli/commands/verify.js +6 -4
  5. package/dist/gates/baseline.d.ts +8 -3
  6. package/dist/gates/baseline.js +6 -3
  7. package/dist/gates/review.js +12 -1
  8. package/dist/gates/run-gates.d.ts +4 -1
  9. package/dist/gates/run-gates.js +65 -9
  10. package/dist/gates/test-manifest.d.ts +3 -0
  11. package/dist/gates/test-manifest.js +20 -2
  12. package/dist/gates/test-reporter.js +6 -1
  13. package/dist/graph/graph.d.ts +2 -0
  14. package/dist/graph/graph.js +5 -0
  15. package/dist/run/activity.d.ts +28 -0
  16. package/dist/run/activity.js +194 -0
  17. package/dist/run/daemon.d.ts +9 -0
  18. package/dist/run/daemon.js +200 -46
  19. package/dist/run/git.d.ts +6 -1
  20. package/dist/run/git.js +63 -11
  21. package/dist/run/journal.d.ts +20 -4
  22. package/dist/run/journal.js +60 -9
  23. package/dist/run/operator-page-summary.d.ts +56 -0
  24. package/dist/run/operator-page-summary.js +68 -0
  25. package/dist/run/operator-summary.d.ts +69 -0
  26. package/dist/run/operator-summary.js +77 -0
  27. package/dist/run/protocol.d.ts +71 -0
  28. package/dist/run/protocol.js +32 -0
  29. package/dist/tui/cockpit/board.d.ts +9 -0
  30. package/dist/tui/cockpit/board.js +10 -0
  31. package/dist/tui/cockpit/derive.d.ts +35 -0
  32. package/dist/tui/cockpit/derive.js +152 -10
  33. package/dist/tui/cockpit/evidence-view.d.ts +2 -0
  34. package/dist/tui/cockpit/evidence-view.js +42 -12
  35. package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
  36. package/dist/tui/cockpit/run-cockpit.js +95 -1
  37. package/dist/tui/cockpit/run-view.d.ts +37 -0
  38. package/dist/tui/cockpit/run-view.js +189 -2
  39. package/package.json +1 -1
  40. package/skills/tickmarkr-overseer/SKILL.md +55 -3
@@ -83,6 +83,43 @@ export declare function applyRunViewKey(session: RunViewSession, event: RunViewK
83
83
  * a closed run whose record lacks a bucket says so rather than reading absence as empty.
84
84
  */
85
85
  export declare function runSummaryLine(s: OperatorSnapshot): string;
86
+ /** One projected reading and the physical journal line it came from; no line means no row says it. */
87
+ export interface ProjectionField {
88
+ readonly label: string;
89
+ readonly line?: number;
90
+ }
91
+ export interface TaskProjection {
92
+ readonly taskId: string;
93
+ readonly identity: ProjectionField;
94
+ readonly phase: ProjectionField;
95
+ readonly build: ProjectionField;
96
+ readonly blocker: ProjectionField;
97
+ readonly nextAction: ProjectionField;
98
+ /** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result. */
99
+ readonly stalled?: ProjectionField;
100
+ }
101
+ export declare const STALL_MARKER = "\u26A0 stalled harvest suspected";
102
+ /**
103
+ * The evidence-only projection status renders (run/activity.ts, run/operator-summary.ts), folded
104
+ * over the journal rows this view was handed, with the physical line each reading was taken from.
105
+ * Nothing is inferred from a title, a provider or a task id: an unrecorded field says so.
106
+ */
107
+ export declare function projectRunTasks(snapshot: OperatorSnapshot, rows: readonly RunEvidenceRow[], graph: RunGraph | undefined, decisions: readonly RunDecision[], runId: string): readonly TaskProjection[];
108
+ /** One wrapped line per task: every field with its line, and the stall marker when the harvest is suspect. */
109
+ export declare const projectionLine: (p: TaskProjection) => string;
110
+ export interface PaneLocator {
111
+ readonly status: "recorded" | "unavailable";
112
+ readonly reason: string;
113
+ /** The worker-launch row the reading stands on, when one exists. */
114
+ readonly line?: number;
115
+ }
116
+ /**
117
+ * Where the current attempt's pane is recorded — and only that. The reading is exact-name: the
118
+ * launch row's slot name must be the owned name of THIS task, attempt and run. No terminal is ever
119
+ * chosen by title, provider or task id; a missing, foreign, stale or ambiguous record reads
120
+ * unavailable with the rows it was read from, and reading never touches a host.
121
+ */
122
+ export declare function paneLocator(task: OperatorTask, rows: readonly RunEvidenceRow[], runId: string): PaneLocator;
86
123
  /** The board's footer keys merged with the cockpit's own — every one of them handled by this view or the shell. */
87
124
  export declare const RUN_VIEW_KEYS: readonly string[];
88
125
  export interface RunViewProps {
@@ -2,7 +2,11 @@ import { jsx as _jsx, jsxs as _jsxs, Fragment as _Fragment } from "react/jsx-run
2
2
  import { Box } from "ink";
3
3
  import { useContext } from "react";
4
4
  import { GLYPHS } from "../../brand.js";
5
+ import { formatOwnedName, parseOwnedName } from "../../drivers/types.js";
5
6
  import { GATE_NAMES } from "../../graph/schema.js";
7
+ import { projectActivity } from "../../run/activity.js";
8
+ import { projectOperatorSummary } from "../../run/operator-summary.js";
9
+ import { trackJournalRows } from "../../run/protocol.js";
6
10
  import { normalizeGateOutcome } from "../../run/outcome.js";
7
11
  import { BOARD_KEYS, boardFooter, clipBoard, renderBoardLines } from "./board.js";
8
12
  import { BodyText, Panel, ShellTheme } from "./components.js";
@@ -172,6 +176,185 @@ export function runSummaryLine(s) {
172
176
  }
173
177
  const attemptLabel = (t) => (t.attempt !== undefined ? `attempt ${t.attempt}` : t.dispatches === 0 ? "never dispatched" : "attempt unknown");
174
178
  const parkOf = (t) => (t.state === "human" ? `human · ${t.parkKind ?? "unknown kind"}` : t.state);
179
+ export const STALL_MARKER = "⚠ stalled harvest suspected";
180
+ const ordinal = (v) => (typeof v === "number" && Number.isInteger(v) && v >= 0 ? v : undefined);
181
+ const slotNameOf = (data) => {
182
+ const slot = data.slot;
183
+ return str(data.slotName) ?? (typeof slot === "object" && slot !== null ? str(slot.name) : str(slot));
184
+ };
185
+ const NO_ROW = "no journal row";
186
+ const readLabel = (f) => `${f.label} ${f.line === undefined ? `(${NO_ROW})` : `#L${f.line}`}`;
187
+ /**
188
+ * The evidence-only projection status renders (run/activity.ts, run/operator-summary.ts), folded
189
+ * over the journal rows this view was handed, with the physical line each reading was taken from.
190
+ * Nothing is inferred from a title, a provider or a task id: an unrecorded field says so.
191
+ */
192
+ export function projectRunTasks(snapshot, rows, graph, decisions, runId) {
193
+ const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
194
+ const graphTask = (id) => graph?.tasks.find((t) => t.id === id);
195
+ const activityTasks = snapshot.tasks.map((t) => ({ id: t.id, gates: graphTask(t.id)?.gates ?? [...GATE_NAMES], deps: graphTask(t.id)?.deps ?? [], status: t.state }));
196
+ const tracked = trackJournalRows(runId, ordered.map((r) => ({ raw: r.event, sourceIndex: r.line })));
197
+ const activity = projectActivity(runId, tracked, activityTasks);
198
+ const byTask = (id) => ordered.filter((r) => r.event?.taskId === id);
199
+ const summaries = new Map(projectOperatorSummary(snapshot.tasks.map((t) => {
200
+ const own = byTask(t.id);
201
+ const responsibility = [...own].reverse().find((r) => str(r.event.data.role) !== undefined || str(r.event.data.agent) !== undefined);
202
+ // The summary's own vocabulary (derive.ts reads the graph status for an unmentioned task): a task the
203
+ // operator fold holds as blocked, or a graph task the journal never mentions, is pending on its prerequisites.
204
+ const status = t.state === "blocked" || (t.state === "unknown" && graphTask(t.id) !== undefined) ? "pending" : t.state;
205
+ return {
206
+ id: t.id, status, deps: graphTask(t.id)?.deps ?? [],
207
+ ...(own.length === 0 ? {} : { phase: activity.get(t.id)?.state, lastEvidenceAt: own.at(-1).event.ts }),
208
+ ...(responsibility ? { responsible: { role: str(responsibility.event.data.role), agent: str(responsibility.event.data.agent) } } : {}),
209
+ };
210
+ }), decisions.map((d) => ({ taskId: d.taskId, park: { kind: d.park.kind, tombstone: d.park.tombstone, ...(d.park.reason === undefined ? {} : { reason: d.park.reason }) }, verbs: d.verbs, ...(d.diagnostic === undefined ? {} : { diagnostic: d.diagnostic }) }))).map((s) => [s.taskId, s]));
211
+ return snapshot.tasks.map((t) => {
212
+ const own = byTask(t.id);
213
+ const act = activity.get(t.id);
214
+ const summary = summaries.get(t.id);
215
+ // identity: the newest row that recorded a role, an agent, an owned slot name or a dispatch assignment
216
+ const identityRow = [...own].reverse().find((r) => { const d = r.event.data; if (act?.attempt !== undefined && ordinal(d.attempt) !== undefined && ordinal(d.attempt) !== act.attempt)
217
+ return false; return str(d.role) !== undefined || str(d.agent) !== undefined || slotNameOf(d) !== undefined || (r.event.event === "task-dispatch" && d.assignment !== undefined); });
218
+ let identity = { label: "identity unrecorded" };
219
+ if (identityRow) {
220
+ const d = identityRow.event.data;
221
+ const name = slotNameOf(d);
222
+ const owned = name === undefined ? null : parseOwnedName(name);
223
+ const role = str(d.role) ?? owned?.role ?? (identityRow.event.event === "worker-launch" ? "worker" : "role unrecorded");
224
+ const a = d.assignment;
225
+ const agent = str(d.agent) ?? name ?? (typeof a?.adapter === "string" && typeof a.model === "string" ? `${a.adapter}:${a.model}` : "agent unrecorded");
226
+ identity = { label: `${role} ${agent}`, line: identityRow.line };
227
+ }
228
+ // Finding 2: the reference is the row whose acceptance CHANGED the projected value — a rejected
229
+ // receipt or a delayed old-attempt result changes nothing and so is never cited. The task's rows
230
+ // and the run-level resets are re-folded prefix by prefix through projectActivity itself.
231
+ // ponytail: O(k²) in the task's own rows; memoize the fold if journals reach thousands of rows per task.
232
+ const changes = acceptedChanges(runId, tracked, t.id, activityTasks.find((a) => a.id === t.id));
233
+ const phase = { label: `phase ${summary?.phase ?? act?.state ?? "unrecorded"}`, ...(own.length > 0 && changes.state !== undefined ? { line: changes.state } : {}) };
234
+ const b = act?.build;
235
+ const receipt = b !== undefined && "receipt" in b ? b.receipt : undefined;
236
+ const build = { label: `build ${b?.state ?? "start-unrecorded"}${receipt ? ` exit ${receipt.exitCode ?? "-"}` : ""}`, ...(changes.build !== undefined ? { line: changes.build } : {}) };
237
+ const decision = decisions.find((d) => d.taskId === t.id);
238
+ const blockerRow = decision ? { line: decision.park.line } : [...own].reverse().find((r) => r.event.event === "task-human" || r.event.event === "task-failed");
239
+ const blk = summary?.blocker;
240
+ const blocker = { label: `blocker ${blk ? `${blk.kind}${blk.diagnostic ? ` · ${blk.diagnostic}` : ""}` : "none"}`, ...(blk && blockerRow ? { line: blockerRow.line } : {}) };
241
+ const nextAction = { label: `next ${blk?.nextAction ?? "none"}`, ...(blk?.nextAction && blockerRow ? { line: blockerRow.line } : {}) };
242
+ // OBS-1048 harvest, chronological: the newest worker-launch opens the attempt; a later worker-result retires it.
243
+ let launched, launchedAttempt, returned = false;
244
+ const nudges = [], pages = [];
245
+ // Finding 1: only rows of the launched attempt count; a row without an attempt belongs to it (derive.ts attemptHarvests).
246
+ const ofLaunched = (d) => launched !== undefined && (ordinal(d.attempt) ?? launchedAttempt) === launchedAttempt;
247
+ for (const r of own) {
248
+ const e = r.event;
249
+ switch (e.event) {
250
+ case "worker-launch":
251
+ launched = r.line;
252
+ launchedAttempt = ordinal(e.data.attempt);
253
+ returned = false;
254
+ nudges.length = 0;
255
+ pages.length = 0;
256
+ break;
257
+ case "worker-nudge-failed":
258
+ if (ofLaunched(e.data))
259
+ nudges.push(r.line);
260
+ break;
261
+ case "operator-page":
262
+ if (ofLaunched(e.data))
263
+ pages.push(r.line);
264
+ break;
265
+ case "worker-result":
266
+ if (ofLaunched(e.data))
267
+ returned = true;
268
+ break;
269
+ }
270
+ }
271
+ const stalled = launched !== undefined && !returned && nudges.length > 0 && pages.length > 0
272
+ ? { label: `${STALL_MARKER} · launch #L${launched} · nudge failed ${nudges.map((l) => `#L${l}`).join(",")} · paged ${pages.map((l) => `#L${l}`).join(",")} · no worker-result`, line: launched }
273
+ : undefined;
274
+ return { taskId: t.id, identity, phase, build, blocker, nextAction, ...(stalled ? { stalled } : {}) };
275
+ });
276
+ }
277
+ /** The line at which each projected value last changed, folding the task's rows (and run-level resets) prefix by prefix. */
278
+ function acceptedChanges(runId, tracked, taskId, task) {
279
+ const rows = tracked.filter((r) => { const raw = r.raw; return raw.taskId === taskId || (raw.taskId === undefined && (raw.event === "run-resume" || raw.event === "run-end")); });
280
+ const buildKey = (p) => `${p.build.state}|${"receipt" in p.build ? `${p.build.receipt.attribution.invocation}|${p.build.receipt.exitCode ?? ""}` : ""}`;
281
+ // Seeded from the empty fold: the initial reading is no row's doing, so only a change is cited.
282
+ const initial = projectActivity(runId, [], [task]).get(taskId);
283
+ let state = initial?.state, build = initial === undefined ? undefined : buildKey(initial);
284
+ const out = {};
285
+ for (let i = 0; i < rows.length; i++) {
286
+ const p = projectActivity(runId, rows.slice(0, i + 1), [task]).get(taskId);
287
+ if (!p)
288
+ continue;
289
+ const line = rows[i].sourceIndex;
290
+ if (p.state !== state) {
291
+ state = p.state;
292
+ out.state = line;
293
+ }
294
+ if (buildKey(p) !== build) {
295
+ build = buildKey(p);
296
+ out.build = line;
297
+ }
298
+ }
299
+ return out;
300
+ }
301
+ /** One wrapped line per task: every field with its line, and the stall marker when the harvest is suspect. */
302
+ export const projectionLine = (p) => [p.taskId, readLabel(p.identity), readLabel(p.phase), readLabel(p.build), readLabel(p.blocker), readLabel(p.nextAction), ...(p.stalled ? [p.stalled.label] : [])].join(" · ");
303
+ /**
304
+ * Where the current attempt's pane is recorded — and only that. The reading is exact-name: the
305
+ * launch row's slot name must be the owned name of THIS task, attempt and run. No terminal is ever
306
+ * chosen by title, provider or task id; a missing, foreign, stale or ambiguous record reads
307
+ * unavailable with the rows it was read from, and reading never touches a host.
308
+ */
309
+ export function paneLocator(task, rows, runId) {
310
+ const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
311
+ // The dispatch-recorded pane field stays reachable beside every reading, with the row it stands on.
312
+ const recorded = task.pane === undefined ? "" : ` · recorded pane ${task.pane}${task.evidence ? ` · evidence #L${task.evidence.line}` : ""}`;
313
+ if (task.attempt === undefined)
314
+ return { status: "unavailable", reason: `no recorded attempt${recorded}` };
315
+ const launches = ordered.filter((r) => r.event.event === "worker-launch" && r.event.taskId === task.id && ordinal(r.event.data.attempt) === task.attempt);
316
+ if (launches.length === 0)
317
+ return { status: "unavailable", reason: `no worker-launch for attempt ${task.attempt}${recorded}` };
318
+ const ids = new Set(launches.map((r) => { const s = r.event.data.slot; return str(s?.id) ?? `#L${r.line}`; }));
319
+ if (ids.size > 1)
320
+ return { status: "unavailable", reason: `ambiguous: ${launches.length} launches ${launches.map((r) => `#L${r.line}`).join(",")}${recorded}` };
321
+ const launch = launches.at(-1);
322
+ const d = launch.event.data;
323
+ const slot = d.slot;
324
+ const expected = formatOwnedName({ role: "worker", taskId: task.id, attempt: task.attempt, runId });
325
+ if (str(d.driver) === undefined || str(slot?.id) === undefined || str(slot?.cwd) === undefined)
326
+ return { status: "unavailable", reason: `launch #L${launch.line} lacks driver or slot identity${recorded}`, line: launch.line };
327
+ if (slot?.name !== expected)
328
+ return { status: "unavailable", reason: `launch #L${launch.line} names ${slotNameOf(d) ?? "no pane"}, not this attempt's owned pane${recorded}`, line: launch.line };
329
+ // Finding 3: only a host whose focus capability VERIFIES the recorded identity can ever confirm it —
330
+ // herdr needs the launch's workspace (HerdrDriver.focus rejects a target without one), orca verifies
331
+ // ownership but cannot focus, subprocess and unknown drivers verify nothing. Reading asks no host.
332
+ const driver = String(d.driver);
333
+ if (driver === "herdr" && str(d.workspace) === undefined)
334
+ return { status: "unavailable", reason: `launch #L${launch.line} records no workspace; herdr cannot verify the pane${recorded}`, line: launch.line };
335
+ if (driver !== "herdr" && driver !== "orca")
336
+ return { status: "unavailable", reason: `launch #L${launch.line} driver ${driver} cannot verify a pane${recorded}`, line: launch.line };
337
+ const stale = ordered.find((r) => {
338
+ if (r.line <= launch.line)
339
+ return false;
340
+ const event = r.event;
341
+ if (event.event === "run-resume" || event.event === "run-end")
342
+ return true;
343
+ if (event.taskId !== task.id)
344
+ return false;
345
+ if (event.event === "worker-result")
346
+ return (ordinal(event.data.attempt) ?? task.attempt) === task.attempt;
347
+ if (event.event === "pane-close")
348
+ return str(event.data.paneId) === str(slot?.id);
349
+ return event.event === "merge";
350
+ });
351
+ if (stale)
352
+ return { status: "unavailable", reason: `launch #L${launch.line} is stale: ${stale.event.event} #L${stale.line} followed it${recorded}`, line: launch.line };
353
+ // A journal launch is evidence of what was opened, not a live-host observation. This pure read
354
+ // deliberately performs no focus/list operation, so it cannot upgrade the locator to recorded:
355
+ // the pane may have been closed outside the journal. The explicit `o` action owns verification.
356
+ return { status: "unavailable", reason: `launch #L${launch.line} records ${driver} ${String(slot.id)}, but live host verification is required${recorded}`, line: launch.line };
357
+ }
175
358
  /** The board's footer keys merged with the cockpit's own — every one of them handled by this view or the shell. */
176
359
  export const RUN_VIEW_KEYS = [...BOARD_KEYS.slice(0, 2), "←→ verdict gate", "PgUp/PgDn page", "x outcome", "a actions", "o pane", "Tab focus", "? keys", BOARD_KEYS[2]];
177
360
  /** The Run body: the approved board (BD-1) over the fold, then the selected task's detail panels. */
@@ -182,6 +365,10 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
182
365
  const fit = (text) => fitCells(text, inner);
183
366
  const decisionLines = (lines) => lines.flatMap((line) => wrapCells(line, inner));
184
367
  const lookup = evidenceLookup(rows, page);
368
+ const wrap = (text) => wrapCells(text, inner);
369
+ // The whole journal, not the retained tail: the projection and the locator read every row.
370
+ const allRows = lookup.later(0);
371
+ const projections = projectRunTasks(snapshot, allRows, graph, decisions, run.runId);
185
372
  const tasks = snapshot.tasks;
186
373
  const selection = Math.min(session.selection, Math.max(0, tasks.length - 1));
187
374
  const task = tasks[selection];
@@ -195,9 +382,9 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
195
382
  const { menu, confirming, receipt, notice } = session.decisions;
196
383
  const blocked = task === undefined ? [] : graph?.tasks.filter((g) => g.deps.includes(task.id)).map((g) => g.id) ?? decision?.blocks ?? [];
197
384
  const board = renderBoardLines({ runId: run.runId, snapshot, graph, now: now(), live: run.live, selection: task?.id, keys: false, colour }, width);
198
- return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [_jsx(BodyText, { children: fit(`state ${parkOf(task)} · ${attemptLabel(task)} · dispatches ${task.dispatches} · path ${task.path ?? "unknown"} · pane ${task.pane ?? "unknown"} · alarm ${task.alarmMs === undefined ? "unknown" : `${task.alarmMs}ms`}${task.mergeEvidence ? ` · merged #L${task.mergeEvidence.line}` : ""}`) }), _jsx(BodyText, { emphasis: "dim", children: fit(`outcome filter ${session.outcomeFilter} (x cycles) · ${shown.length} of ${cells.length} gates shown`) }), shown.map((c) => (_jsx(BodyText, { children: fit(`${c.letter} ${c.gate.padEnd(10)} ${c.labels.join(" · ")}${c.line === undefined ? "" : ` · #L${c.line}`}`) }, c.gate)))] })), verdictCell !== undefined && (_jsxs(Panel, { title: `VERDICT / ${verdictCell.gate}${verdictCell.line === undefined ? "" : ` #L${verdictCell.line}`} · ${verdict.length === 0 ? "no verdict text" : `lines ${from + 1}–${from + window.length} of ${verdict.length}`}`, children: [window.map((line, i) => _jsx(BodyText, { children: fit(`${String(from + i + 1).padStart(4)} ${line}`) }, `${from + i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit("←→ gate · PageUp/PageDown page") })] })), task?.state === "human" && (_jsx(Panel, { title: `PARK / ${task.id}`, children: decision === undefined
385
+ return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [wrap(`state ${parkOf(task)} · ${attemptLabel(task)} · dispatches ${task.dispatches} · path ${task.path ?? "unknown"} · pane ${task.pane ?? "unknown"} · alarm ${task.alarmMs === undefined ? "unknown" : `${task.alarmMs}ms`}${task.mergeEvidence ? ` · merged #L${task.mergeEvidence.line}` : ""} · locator ${(() => { const l = paneLocator(task, allRows, run.runId); return l.status === "recorded" ? `recorded ${l.reason}` : `unavailable — ${l.reason}`; })()}`).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit(`outcome filter ${session.outcomeFilter} (x cycles) · ${shown.length} of ${cells.length} gates shown`) }), shown.map((c) => (_jsx(BodyText, { children: fit(`${c.letter} ${c.gate.padEnd(10)} ${c.labels.join(" · ")}${c.line === undefined ? "" : ` · #L${c.line}`}`) }, c.gate)))] })), projections.length > 0 && (_jsx(Panel, { title: "PROJECTION / every task \u00B7 identity \u00B7 phase \u00B7 build \u00B7 blocker \u00B7 next action, each with its journal line", children: projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => _jsx(BodyText, { emphasis: p.stalled && i === 0 ? "strong" : "normal", children: line }, `${p.taskId}:${i}`))) })), verdictCell !== undefined && (_jsxs(Panel, { title: `VERDICT / ${verdictCell.gate}${verdictCell.line === undefined ? "" : ` #L${verdictCell.line}`} · ${verdict.length === 0 ? "no verdict text" : `lines ${from + 1}–${from + window.length} of ${verdict.length}`}`, children: [window.map((line, i) => _jsx(BodyText, { children: fit(`${String(from + i + 1).padStart(4)} ${line}`) }, `${from + i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit("←→ gate · PageUp/PageDown page") })] })), task?.state === "human" && (_jsx(Panel, { title: `PARK / ${task.id}`, children: decision === undefined
199
386
  ? _jsx(BodyText, { children: fit("park evidence unavailable — refresh before deciding") })
200
- : (_jsxs(_Fragment, { children: [_jsx(BodyText, { children: fit(`#L${decision.park.line} ${decision.park.kind ?? "unknown kind"}${decision.park.failedGate ? ` · failed gate ${decision.park.failedGate}` : ""} · ${decision.park.reason ?? "no reason recorded"}`) }), _jsx(BodyText, { children: fit(`blocks ${blocked.length === 0 ? "nothing" : blocked.join(", ")} · attempts ${decision.attempts}`) }), _jsx(BodyText, { children: fit(decision.verbs.length === 0 ? `diagnostic: ${decision.diagnostic ?? "no verb"}` : `decisions: ${decision.verbs.join(" · ")} (a Actions)`) }), _jsx(BodyText, { emphasis: "dim", children: fit(run.live ? "matching live daemon enacts a release at its next task boundary" : run.blockingRunId ? `other live run ${run.blockingRunId} holds the lock; resume after it ends` : "no live owner: a release records permission; resume dispatches") })] })) })), menu !== null && (_jsx(Panel, { title: `ACTIONS / ${menu.taskId}`, focused: true, children: menu.verbs.length === 0
387
+ : (_jsxs(_Fragment, { children: [wrap(`#L${decision.park.line} ${decision.park.kind ?? "unknown kind"}${decision.park.failedGate ? ` · failed gate ${decision.park.failedGate}` : ""} · ${decision.park.reason ?? "no reason recorded"}`).map((line, i) => _jsx(BodyText, { children: line }, `park:${i}`)), _jsx(BodyText, { children: fit(`blocks ${blocked.length === 0 ? "nothing" : blocked.join(", ")} · attempts ${decision.attempts}`) }), wrap(decision.verbs.length === 0 ? `diagnostic: ${decision.diagnostic ?? "no verb"}` : `decisions: ${decision.verbs.join(" · ")} (a Actions)`).map((line, i) => _jsx(BodyText, { children: line }, `verbs:${i}`)), _jsx(BodyText, { emphasis: "dim", children: fit(run.live ? "matching live daemon enacts a release at its next task boundary" : run.blockingRunId ? `other live run ${run.blockingRunId} holds the lock; resume after it ends` : "no live owner: a release records permission; resume dispatches") })] })) })), menu !== null && (_jsx(Panel, { title: `ACTIONS / ${menu.taskId}`, focused: true, children: menu.verbs.length === 0
201
388
  ? _jsx(BodyText, { children: fit(`no decision — ${decision?.diagnostic ?? "no verb for this park"}`) })
202
389
  : menu.verbs.map((verb, i) => _jsx(BodyText, { emphasis: i === menu.selection ? "strong" : "normal", children: fit(`${i === menu.selection ? `${GLYPHS.pointer} ` : " "}${verb}`) }, verb)) })), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.command.verb.toUpperCase()}`, focused: true, children: decisionLines(decisionConfirmLines(confirming)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), receipt !== null && (_jsx(Panel, { title: receipt.ok ? "RECEIPT · appended" : "RECEIPT · refused", children: decisionLines(decisionReceiptLines(receipt)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), notice !== null && _jsx(BodyText, { children: notice }), _jsx(BodyText, { emphasis: "dim", children: clipBoard(boardFooter([...RUN_VIEW_KEYS, decisionKeybar(session.decisions, decision)].filter(Boolean), colour), width) })] }));
203
390
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.5.6",
3
+ "version": "2.5.7",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -139,7 +139,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
139
139
  fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
140
140
  with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
141
141
  updates it). Never long context strings or ✓-chains.
142
- - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
142
+ - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal, name it in the same act — `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (tab title; see the seat-name law under Seat-spawn recipes) — and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
143
143
  2. **Orchestrator**: Launch the orchestrator with your agent host.
144
144
  - **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify a model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
145
145
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create` on a path worktree selector with a command:
@@ -176,6 +176,12 @@ through brief lineage. **An executor choice nobody made is still an executor cho
176
176
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Re-list terminals, re-read the
177
177
  recorded terminal handles and evidence files, and re-arm only file/journal watchers. Do not use a Herdr
178
178
  pane id, `agent start`, `pane run`, tab canon, or a name-keyed Herdr watcher on an Orca launch.
179
+ **Handles do not survive an Orca restart (OBS-1050, three gates lost to it in v2.5.6).** The restart
180
+ re-issues every terminal handle; a recorded one answers `terminal_handle_stale` and any watcher or gate
181
+ keyed to it is already dead. So: record every seat's handle in `pids/<seat>.handle` beside its title at
182
+ spawn; on `terminal_handle_stale`, re-list terminals and re-resolve the seat by TITLE, then rewrite the
183
+ handle file; and treat a stale handle as a dead seat until the read-back of the new handle proves
184
+ otherwise — a gate that was running in that seat has to be re-launched, never assumed to continue.
179
185
  When briefing a seat on either host, take its current address from the handoff file; never hardcode a pane id
180
186
  or terminal handle in a brief or command template.
181
187
 
@@ -199,7 +205,45 @@ Inventories retain the **full suite log**, not a tail or summary, and no one run
199
205
  - **On herdr (`HERDR_ENV=1`)**: Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
200
206
  verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
201
207
  `herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`.
202
- - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`. Brief delivery travels as a file announced by `orca terminal send --terminal <handle> --text "<announcement>" --enter --wait-submit 15 --json`. Read `result.send.prompt.stages` and treat the brief as delivered only when a `turn_started` stage is present; `accepted: true` proves input acceptance, not a started turn; never resend on `accepted` alone.
208
+ - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
209
+ **`--worktree path:` resolves only an Orca-MANAGED worktree** (`orca worktree list`); on any other checkout —
210
+ a `git worktree add` the overseer made for a spec branch, a throwaway clone — `terminal create` hangs and
211
+ returns "Timed out waiting for terminal handle after creation". The lawful form for such a checkout is a
212
+ terminal in a managed worktree with a cwd change in the command:
213
+ `orca terminal create --worktree path:<managed-repo> --title "ORCH · <version>" --command "cd <checkout> && <agent-cmd>" --json`.
214
+ **An Orca-hosted run's checkout is created by Orca** (`orca worktree create --repo id:<repo> --base-branch <spec-branch>`),
215
+ never by `git worktree add`: the orca driver asks `orca worktree current` from inside every task worktree and refuses
216
+ (`selector_not_found`) on an unmanaged checkout — every task fails pre-launch and the board cannot place (OBS-1061,
217
+ one abandoned run). **The overseer sits in the RUN's workspace**, in its own tab beside the orchestrator's, so the
218
+ run's card shows its supervision; records may live on another branch and are committed there by path.
219
+ **The orchestrator NEVER shares the overseer's tab.** Never spawn it with `orca terminal split` off the
220
+ overseer's own handle, even as a fallback: the daemon self-places its live board as a split of the
221
+ LAUNCHING terminal, so an orchestrator in the overseer's tab drags the board and every worker pane
222
+ into it (measured 2026-09-20: `terminal create` timed out on an unmanaged spec worktree, the overseer
223
+ split instead, and the run had to be re-seated before GO). Before delivering the brief, prove the seat
224
+ is in its own tab: `orca terminal list --json`, the new handle's `tabId` must differ from the
225
+ overseer's; if it does not, close the seat and create it again the lawful way. `split` is for seats
226
+ that BELONG beside another seat (consultant pairs, a reviewer beside its surgeon), never for the
227
+ orchestrator or a gate seat.
228
+ **Seat names on Orca are TAB titles, and a tab title HOLDS; the pane title is the agent's and drifts.**
229
+ Orca keeps two titles per seat. The TAB title is owned: `orca terminal create --title` and
230
+ `orca terminal rename --terminal <handle> --title "<seat>"` set it, and nothing a process writes to
231
+ its pty changes it — measured 2026-09-20 (OBS-1063) on a scratch terminal writing an OSC title every
232
+ 2 s: the PANE title flipped to the OSC text within 8 s, the TAB title stayed, and the operator's tab
233
+ strip renders the TAB title. The pane title is whatever the process last wrote (oh-my-zsh, Claude
234
+ Code's turn summary, Codex's last prompt) and is NOT the seat's name. This is vendor-neutral by
235
+ construction — no per-vendor env var, no agent command, no launch-form ritual — which is what makes
236
+ it the same law as Herdr's tab label. So: **the overseer names its OWN seat at Setup 1** with
237
+ `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (the seat
238
+ it was invoked in is the one seat nobody else launches), every seat it spawns is named by
239
+ `create --title`, and a live-label change (progress fraction, hot-state token) is `rename` again.
240
+ **Prove a name by the field that holds:** `tabs[].title` in
241
+ `orca terminal list --include-visual-layouts --json`, never the plain `list`'s `title` — that one is
242
+ the pane title and reads as drift on every agent seat, which is how three sessions "proved" a
243
+ failure that was a misread (`src/drivers/orca.ts:27-42` keys worker identity on the owned tab title
244
+ for the same reason). The earlier three-part launch form (`DISABLE_AUTO_TITLE` + `printf` +
245
+ `CLAUDE_CODE_DISABLE_TERMINAL_TITLE=1`) is withdrawn: it silenced one vendor's pane title and named
246
+ nothing the tab did not already hold. Brief delivery travels as a file announced by `orca terminal send --terminal <handle> --text "<announcement>" --enter --wait-submit 15 --json`. Read `result.send.prompt.stages` and treat the brief as delivered only when a `turn_started` stage is present; `accepted: true` proves input acceptance, not a started turn; never resend on `accepted` alone.
203
247
  For Leg-2 (both hosts), a Codex reviewer under `workspace-write` must be briefed with an in-worktree verdict path such as
204
248
  `<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
205
249
  ## Supervising tickmarkr as the executor — WHO DOES WHAT
@@ -656,6 +700,14 @@ they are left implicit:
656
700
  fail-closed verdict, no daemon, no retries. A per-lane grep gate re-implements a weaker version of
657
701
  this and passes on source text the screen never renders, which is exactly the class the acceptance
658
702
  judge exists to reject. One command per lane, named in the lane's own brief.
703
+ Fix-leg laws learned the hard way (v2.5.6 R25/R26/R34):
704
+ - A `test:` criterion for a fix leg is written FROM the landed test's title, verbatim, never before the
705
+ test exists — the acceptance oracle runs the sentence as vitest `-t` and a paraphrase matches zero tests.
706
+ - Every fix-leg verify carries `--author <the seat that wrote it>`; without it the author is HUMAN, every
707
+ vendor stays eligible, and a same-vendor approval is not the cross-vendor review the criterion names.
708
+ - A grader change is proved on a captured REAL job log, never on a synthetic one.
709
+ - A public-CI red is classified in a throttled Linux container first (`docker run --cpus=2 node:22`),
710
+ by measurement: starvation on the base tree is not a defect on the branch.
659
711
 
660
712
  ## Pane mechanics that bite
661
713
 
@@ -1029,7 +1081,7 @@ orchestrator turn boundary.
1029
1081
 
1030
1082
  ### On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`) — supervision instruments
1031
1083
 
1032
- At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion.
1084
+ At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion. A seat-liveness watcher on the ORCHESTRATOR handle is mandatory, not optional: poll `orca terminal read --terminal <handle> --screen --json` on a bounded interval, treat `terminal_handle_stale` or a missing terminal as seat death, and re-resolve by title before re-arming (OBS-1050 — the orchestrator seat exited silently and the daemon ran unsupervised to a PARTIAL run-end).
1033
1085
 
1034
1086
  ## Specialist pipeline rules
1035
1087