tickmarkr 2.5.9 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/adapters/claude-code.js +19 -6
  2. package/dist/adapters/prompt.d.ts +2 -1
  3. package/dist/adapters/prompt.js +11 -1
  4. package/dist/adapters/types.d.ts +4 -0
  5. package/dist/adapters/types.js +10 -0
  6. package/dist/cli/commands/fleet.js +26 -1
  7. package/dist/cli/commands/init.js +1 -1
  8. package/dist/cli/commands/plan.js +7 -3
  9. package/dist/cli/commands/report.js +11 -1
  10. package/dist/cli/commands/status.js +13 -3
  11. package/dist/cli/commands/verify.js +2 -0
  12. package/dist/cli/help.d.ts +2 -0
  13. package/dist/cli/help.js +2 -0
  14. package/dist/compile/native.js +39 -4
  15. package/dist/drivers/orca.d.ts +1 -1
  16. package/dist/drivers/orca.js +27 -5
  17. package/dist/gates/baseline.d.ts +2 -0
  18. package/dist/gates/baseline.js +9 -1
  19. package/dist/gates/cache.d.ts +11 -1
  20. package/dist/gates/cache.js +35 -15
  21. package/dist/gates/llm.js +4 -1
  22. package/dist/gates/review.d.ts +2 -3
  23. package/dist/gates/review.js +28 -35
  24. package/dist/gates/run-gates.d.ts +4 -0
  25. package/dist/gates/run-gates.js +17 -4
  26. package/dist/gates/test-manifest.d.ts +17 -0
  27. package/dist/gates/test-manifest.js +109 -9
  28. package/dist/gates/test-reporter.js +4 -0
  29. package/dist/graph/graph.d.ts +7 -3
  30. package/dist/graph/graph.js +21 -5
  31. package/dist/graph/schema.d.ts +2 -0
  32. package/dist/graph/schema.js +2 -0
  33. package/dist/route/preference.d.ts +1 -1
  34. package/dist/route/preference.js +10 -39
  35. package/dist/route/role-pick.d.ts +16 -0
  36. package/dist/route/role-pick.js +15 -0
  37. package/dist/route/router.d.ts +14 -0
  38. package/dist/route/router.js +9 -1
  39. package/dist/run/consult.js +5 -9
  40. package/dist/run/daemon.d.ts +16 -1
  41. package/dist/run/daemon.js +554 -77
  42. package/dist/run/git.d.ts +46 -1
  43. package/dist/run/git.js +138 -6
  44. package/dist/run/host-health.d.ts +20 -0
  45. package/dist/run/host-health.js +64 -0
  46. package/dist/run/journal.d.ts +1 -1
  47. package/dist/run/journal.js +51 -6
  48. package/dist/run/merge.d.ts +1 -1
  49. package/dist/run/merge.js +30 -6
  50. package/dist/run/operator-state.d.ts +24 -2
  51. package/dist/run/operator-state.js +41 -5
  52. package/dist/run/stall.d.ts +38 -2
  53. package/dist/run/stall.js +276 -6
  54. package/dist/tui/cockpit/board.d.ts +1 -1
  55. package/dist/tui/cockpit/board.js +27 -19
  56. package/dist/tui/cockpit/derive.d.ts +2 -0
  57. package/dist/tui/cockpit/derive.js +4 -0
  58. package/dist/tui/cockpit/live-runtime.d.ts +4 -0
  59. package/dist/tui/cockpit/live-runtime.js +37 -3
  60. package/dist/tui/cockpit/live-store.d.ts +3 -0
  61. package/dist/tui/cockpit/live-store.js +31 -8
  62. package/dist/tui/cockpit/run-cockpit.js +2 -1
  63. package/dist/tui/cockpit/run-view.d.ts +2 -4
  64. package/dist/tui/cockpit/run-view.js +9 -8
  65. package/package.json +1 -1
  66. package/schema/rungraph.schema.json +7 -0
  67. package/skills/tickmarkr-overseer/SKILL.md +76 -14
@@ -132,10 +132,42 @@ export async function runConsolidatedCockpit(options) {
132
132
  let committedTargets = [];
133
133
  let revision = 0;
134
134
  const listeners = new Set();
135
- const publish = () => { revision++; for (const listener of listeners)
135
+ // Every publish resyncs the key so a view switch never earns a second frame on the next tick.
136
+ const publish = () => { revision++; observed = observationKey(source.snapshot()); for (const listener of listeners)
136
137
  listener(); };
137
- const unsubscribe = source.subscribe(publish);
138
+ let observed = "";
138
139
  const geometry = () => planShell(output.columns ?? 80, output.rows ?? 24, state.shortcutColumns);
140
+ // OBS-1132: the store publishes on every observation tick; the board only re-derives and
141
+ // re-renders when what it would draw changed. Lock and supervision observation continue
142
+ // regardless. Excluded on purpose: sequence, observedAt, metrics and beat ages, which move
143
+ // every tick without changing a visible cell.
144
+ // ponytail: mirrors board.ts `ago` buckets and its ≥118-cell "wide" header — the only clock reader,
145
+ // and only the Run view draws the board, so Home and Evidence never invalidate on the age.
146
+ // The lock is compared by what it says (owner, state, liveness), never by its file stamp: the
147
+ // daemon heartbeat re-touches graph.lock every 10 s without changing a visible cell. Its clock-derived
148
+ // `expired` flag is not drawn by any view either, so a lock ageing past STALE_MS earns no frame.
149
+ const visibleAge = (ms) => ms < 6e4 ? "just now" : ms < 3.6e6 ? `${Math.round(ms / 6e4)}m` : `${Math.round(ms / 3.6e6)}h`;
150
+ const observationKey = (snap) => {
151
+ const last = snap.operator.lastEventAt;
152
+ const clock = state.view === "run" && geometry().bodyColumns >= 118 && last ? visibleAge((options.now ?? Date.now)() - Date.parse(last)) : "";
153
+ return JSON.stringify([
154
+ snap.freshness, snap.actionsEnabled, snap.errors, snap.viewport, snap.inputSequence, clock,
155
+ snap.journal.generation, snap.journal.offset, snap.journal.status, snap.journal.malformedCount, snap.journal.backlogBytes, snap.journal.pending, snap.journal.error,
156
+ snap.graph.identity, snap.graph.status, snap.config.identity, snap.config.status, snap.cache.identity, snap.cache.status,
157
+ snap.lock.value, snap.lock.status, snap.lock.state, snap.lock.alive,
158
+ // D-383: supervision tier state is not drawn by any view; an ARMED→STALE transition earns no
159
+ // frame. Its UNREADABLE case already surfaces through snap.errors above.
160
+ ]);
161
+ };
162
+ observed = observationKey(source.snapshot());
163
+ const unsubscribe = source.subscribe(() => {
164
+ const next = observationKey(source.snapshot());
165
+ if (next !== observed) {
166
+ observed = next;
167
+ publish();
168
+ }
169
+ });
170
+ let frames = 0, derivations = 0;
139
171
  const openEvidence = (identity) => {
140
172
  evidenceNavigation++;
141
173
  selectedEvidence = identity;
@@ -484,7 +516,7 @@ export async function runConsolidatedCockpit(options) {
484
516
  stop(e);
485
517
  return false;
486
518
  }
487
- }, diagnostics: source.diagnostics, stage: edits => { state = { ...state, staged: [...edits] }; publish(); } };
519
+ }, diagnostics: source.diagnostics, stage: edits => { state = { ...state, staged: [...edits] }; publish(); }, frames: () => frames, derivations: () => derivations };
488
520
  options.onDelivery?.(delivery);
489
521
  options.onShellDelivery?.(delivery);
490
522
  resize = () => {
@@ -530,6 +562,7 @@ export async function runConsolidatedCockpit(options) {
530
562
  function App() {
531
563
  useSyncExternalStore(listener => { listeners.add(listener); return () => { listeners.delete(listener); }; }, () => revision);
532
564
  const stdinContext = useStdin();
565
+ useLayoutEffect(() => { frames++; });
533
566
  useLayoutEffect(() => {
534
567
  const emitter = stdinContext.internal_eventEmitter;
535
568
  leafInput = emitter;
@@ -565,6 +598,7 @@ export async function runConsolidatedCockpit(options) {
565
598
  stop(error);
566
599
  }
567
600
  });
601
+ derivations++;
568
602
  const snap = source.snapshot();
569
603
  const p = geometry();
570
604
  const nextDecisionsKey = `${snap.journal.generation}:${snap.journal.offset}:${snap.graph.identity}`;
@@ -2,6 +2,7 @@ import { type JournalEvent } from "../../run/journal.js";
2
2
  import { type OperatorRecord } from "../../run/operator-state.js";
3
3
  export declare const OBSERVATION_INTERVAL_MS = 1000;
4
4
  export declare const STORE_LIMITS: {
5
+ readonly graphBytes: number;
5
6
  readonly history: 256;
6
7
  readonly historyBytes: number;
7
8
  readonly recordBytes: number;
@@ -147,6 +148,7 @@ export declare function createLiveStore(options: LiveStoreOptions): {
147
148
  kind: "fixture";
148
149
  paths: string[];
149
150
  })[] | undefined;
151
+ outOfScope?: string[] | undefined;
150
152
  routingHints?: {
151
153
  pin?: {
152
154
  via: string;
@@ -248,6 +250,7 @@ export declare function createLiveStore(options: LiveStoreOptions): {
248
250
  kind: "fixture";
249
251
  paths: string[];
250
252
  })[] | undefined;
253
+ outOfScope?: string[] | undefined;
251
254
  routingHints?: {
252
255
  pin?: {
253
256
  via: string;
@@ -1,4 +1,4 @@
1
- import { closeSync, fstatSync, openSync, readFileSync, readSync, statSync } from "node:fs";
1
+ import { closeSync, fstatSync, openSync, readSync, statSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { graphPath, stateDirName } from "../../graph/graph.js";
4
4
  import { validateGraph } from "../../graph/schema.js";
@@ -7,7 +7,8 @@ import { isPidLive, STALE_MS } from "../../run/lock.js";
7
7
  import { readTierLiveness, SUPERVISION_TIERS } from "../../run/supervision.js";
8
8
  import { OperatorStateFold } from "../../run/operator-state.js";
9
9
  export const OBSERVATION_INTERVAL_MS = 1_000;
10
- export const STORE_LIMITS = { history: 256, historyBytes: 2 * 1024 * 1024, recordBytes: 1024 * 1024, readBytes: 1024 * 1024, subscribers: 64, metrics: 12, errors: 32 };
10
+ // Graph declarations have a separate 16 MiB bound; journal retention remains unchanged.
11
+ export const STORE_LIMITS = { graphBytes: 16 * 1024 * 1024, history: 256, historyBytes: 2 * 1024 * 1024, recordBytes: 1024 * 1024, readBytes: 1024 * 1024, subscribers: 64, metrics: 12, errors: 32 };
11
12
  const errorText = (e) => e instanceof Error ? e.message : String(e);
12
13
  const fileIdentity = (st) => `${st.dev}:${st.ino}`;
13
14
  const stamp = (st) => `${fileIdentity(st)}:${st.size}:${st.mtimeNs}:${st.ctimeNs}`;
@@ -192,10 +193,12 @@ export class JournalTail {
192
193
  class JsonSource {
193
194
  path;
194
195
  parse;
196
+ maxBytes;
195
197
  cached;
196
- constructor(path, parse) {
198
+ constructor(path, parse, maxBytes = STORE_LIMITS.recordBytes) {
197
199
  this.path = path;
198
200
  this.parse = parse;
201
+ this.maxBytes = maxBytes;
199
202
  }
200
203
  read(now) {
201
204
  try {
@@ -203,9 +206,29 @@ class JsonSource {
203
206
  const id = stamp(st);
204
207
  if (this.cached?.identity === id && this.cached.status === "readable")
205
208
  return this.cached = { ...this.cached, observedAt: now };
206
- if (!st.isFile() || st.size > BigInt(STORE_LIMITS.recordBytes))
207
- throw new Error("source is not a bounded regular file");
208
- const value = this.parse(readFileSync(this.path, "utf8"));
209
+ if (!st.isFile())
210
+ throw new Error("source is not a regular file");
211
+ if (st.size > BigInt(this.maxBytes))
212
+ throw new Error(`source exceeds ${this.maxBytes} byte cap`);
213
+ // Bound the read itself too: a writer can grow the file after stat.
214
+ const fd = openSync(this.path, "r");
215
+ let value;
216
+ try {
217
+ const bytes = Buffer.alloc(Number(st.size) + 1);
218
+ let length = 0;
219
+ while (length < bytes.length) {
220
+ const n = readSync(fd, bytes, length, bytes.length - length, length);
221
+ if (!n)
222
+ break;
223
+ length += n;
224
+ }
225
+ if (length !== Number(st.size) || stamp(fstatSync(fd, { bigint: true })) !== id)
226
+ throw new Error("source changed during read; retry observation");
227
+ value = this.parse(new TextDecoder("utf-8", { fatal: true }).decode(bytes.subarray(0, length)));
228
+ }
229
+ finally {
230
+ closeSync(fd);
231
+ }
209
232
  return this.cached = { source: this.path, identity: id, value, status: "readable", observedAt: now };
210
233
  }
211
234
  catch (e) {
@@ -222,7 +245,7 @@ export function createLiveStore(options) {
222
245
  const tail = new JournalTail(join(state, "runs", parseRunId(options.runId), "journal.jsonl"), {
223
246
  reset: () => { fold = new OperatorStateFold(); }, record: record => fold.apply(record),
224
247
  });
225
- const graph = new JsonSource(graphPath(options.cwd), raw => validateGraph(JSON.parse(raw)));
248
+ const graph = new JsonSource(graphPath(options.cwd), raw => validateGraph(JSON.parse(raw)), STORE_LIMITS.graphBytes);
226
249
  const config = new JsonSource(options.configPath ?? join(state, "config.yaml"), raw => raw);
227
250
  const cache = new JsonSource(options.cachePath ?? join(state, "doctor.json"), JSON.parse);
228
251
  const lock = new JsonSource(join(state, "graph.lock"), raw => {
@@ -256,7 +279,7 @@ export function createLiveStore(options) {
256
279
  const readable = journal.status === "readable" && journal.backlogBytes === 0;
257
280
  return {
258
281
  sequence: ++sequence, observedAt, delayed, freshness: errors.length ? "failed" : delayed ? "delayed" : "fresh",
259
- operator: fold.snapshot({ graph: graphReading.status === "readable" ? graphReading.value : undefined, sequence, observedAt, readable }),
282
+ operator: fold.snapshot({ graph: graphReading.status === "readable" ? graphReading.value : undefined, sequence, observedAt, readable, graphAvailability: { status: graphReading.status, error: graphReading.error } }),
260
283
  journal, graph: graphReading, config: configReading, cache: cacheReading,
261
284
  lock: { ...owner, state: lockState, alive, expired: owner.identity ? observedAt - Number(owner.identity.split(":")[3]) / 1e6 > STALE_MS : undefined },
262
285
  supervision, errors, actionsEnabled: readable && !errors.length && !delayed,
@@ -3,6 +3,7 @@ import { Box, useApp, useStdout } from "ink";
3
3
  import { cloneElement, createContext, useContext, useEffect, useInsertionEffect, useMemo, useRef, useSyncExternalStore, } from "react";
4
4
  import { GLYPHS, PLAIN_COMPACT_LOCKUP, } from "../../brand.js";
5
5
  import { allocateBandColumns, BandLines, BodyText, CockpitGrid, composeBandLine, JournalRowPanel, keyRosterLines, Panel, PANEL_CHROME_ROWS, ProgressMeter, Sparkline, StatTile, StatusStrip, } from "./components.js";
6
+ import { authorsNote } from "../../run/operator-state.js";
6
7
  import { fieldReading } from "./derive.js";
7
8
  import { STALL_MARKER } from "./run-view.js";
8
9
  import { initialRunInteractionState, projectRunKeyEntries, reconcileRunInteraction, RUN_SIDE_RAIL_COLUMN_FLOOR, runPanelFocusOrder, runSideRailVisible, selectableRunViewRowIds, } from "./keys.js";
@@ -280,7 +281,7 @@ function promotedViewRows(data, viewId) {
280
281
  time: fieldReading(row.lastEventTime),
281
282
  ...(row.lastEventTimestamp === undefined ? {} : { timestamp: row.lastEventTimestamp }),
282
283
  state: taskRowState(row),
283
- text: `${row.taskId} · ${fieldReading(row.state)} · ${runAttemptLabel(row.attempts)} · ${fieldReading(row.actor)}${row.title === undefined ? "" : ` · ${row.title}`} · ${taskProjectionText(row, data.journalRows)}`,
284
+ text: `${row.taskId} · ${fieldReading(row.state)} · ${runAttemptLabel(row.attempts)} · ${fieldReading(row.actor)}${authorsNote(row.authors, row.actor) === undefined ? "" : ` · ${authorsNote(row.authors, row.actor)}`}${row.title === undefined ? "" : ` · ${row.title}`} · ${taskProjectionText(row, data.journalRows)}`,
284
285
  }));
285
286
  }
286
287
  if (viewId === "gates") {
@@ -35,10 +35,8 @@ export interface RunGateCell {
35
35
  /** The verdict text the evidence row carries, split into lines for paging. */
36
36
  readonly verdict: readonly string[];
37
37
  }
38
- /**
39
- * The current attempt's seven cells in declaration order. `rows` supplies the later journal rows
40
- * that give a cell its inherited/satisfied label; only rows for this task after the evidence line count.
41
- */
38
+ /** Current-attempt cells in declaration order; later task rows supply inherited/satisfied labels.
39
+ * Only rows after each cell's own evidence line count. */
42
40
  export declare function runGateCells(task: OperatorTask, evidence: EvidenceLookup, rows?: readonly RunEvidenceRow[]): readonly RunGateCell[];
43
41
  export declare const VERDICT_WINDOW = 10;
44
42
  export interface RunViewSession {
@@ -53,7 +53,7 @@ export function evidenceLookup(rows, page) {
53
53
  return lookup;
54
54
  }
55
55
  export const GATE_CELL_LETTERS = {
56
- passed: "P", failed: "F", running: "R", "not-run": "-", disabled: "D", unknown: "?",
56
+ passed: "P", failed: "F", queued: "Q", running: "R", "not-run": "-", disabled: "D", unknown: "?",
57
57
  };
58
58
  /** The outcome selector's vocabulary — a classification of the row, never a word search. */
59
59
  export const OUTCOME_FILTERS = ["all", "infra failure", "work failure", "pass", "unknown"];
@@ -81,10 +81,8 @@ function outcomeLabel(outcome) {
81
81
  case "unavailable": return `unknown — ${outcome.reason}`;
82
82
  }
83
83
  }
84
- /**
85
- * The current attempt's seven cells in declaration order. `rows` supplies the later journal rows
86
- * that give a cell its inherited/satisfied label; only rows for this task after the evidence line count.
87
- */
84
+ /** Current-attempt cells in declaration order; later task rows supply inherited/satisfied labels.
85
+ * Only rows after each cell's own evidence line count. */
88
86
  export function runGateCells(task, evidence, rows = []) {
89
87
  return GATE_NAMES.map((gate) => {
90
88
  const cell = task.gates[gate] ?? { state: "unknown" };
@@ -94,12 +92,15 @@ export function runGateCells(task, evidence, rows = []) {
94
92
  const labels = [];
95
93
  let outcome;
96
94
  if (data === undefined) {
97
- labels.push(line === undefined ? { "not-run": "not run", disabled: "disabled by policy", running: "running", unknown: "unknown", passed: "passed", failed: "failed" }[cell.state] : `evidence #L${line} unavailable`);
95
+ labels.push(line === undefined ? { "not-run": "not run", disabled: "disabled by policy", queued: "queued", running: "running", unknown: "unknown", passed: "passed", failed: "failed" }[cell.state] : `evidence #L${line} unavailable`);
96
+ }
97
+ else if (cell.state === "queued") {
98
+ labels.push(`queued — ${row?.event?.event ?? "wait"}${typeof data.count === "number" ? ` (${data.count} suites)` : ""}`);
98
99
  }
99
100
  else if (data.disabled === true) {
100
101
  labels.push("disabled by policy");
101
102
  }
102
- else if (row?.event?.event === "gate-start") {
103
+ else if (["gate-start", "gate-phase-start", "phase-start"].includes(row?.event?.event ?? "")) {
103
104
  labels.push("running");
104
105
  }
105
106
  else {
@@ -124,7 +125,7 @@ export function runGateCells(task, evidence, rows = []) {
124
125
  if (e.event === "task-approved" && e.data.release === "gate-satisfied" && e.data.gate === gate)
125
126
  labels.push(`satisfied by approval #L${later.line}`);
126
127
  }
127
- const verdict = typeof data?.details === "string" ? data.details.split("\n") : [];
128
+ const verdict = row?.event?.event === "gate-result" && typeof data?.details === "string" ? data.details.split("\n") : [];
128
129
  return { gate, state: cell.state, letter: GATE_CELL_LETTERS[cell.state], ...(line === undefined ? {} : { line }), ...(outcome ? { outcome } : {}), outcomeClass: outcomeClassOf(outcome, cell.state), labels, verdict };
129
130
  });
130
131
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.5.9",
3
+ "version": "2.6.1",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -255,6 +255,13 @@
255
255
  ]
256
256
  }
257
257
  },
258
+ "outOfScope": {
259
+ "type": "array",
260
+ "items": {
261
+ "type": "string",
262
+ "minLength": 1
263
+ }
264
+ },
258
265
  "gates": {
259
266
  "default": [
260
267
  "build",
@@ -62,7 +62,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
62
62
  - **On herdr (`HERDR_ENV=1`)**: Every bundled `watch-*.sh` arm, including `watch-artifacts.sh`, sets `TKR_ARMING_SEAT=<seat>` and writes its own pid under
63
63
  `<state-dir>/overseer/pids/<arming-seat>-<script>-<pid>.pid`. A seat retires only watchers it armed,
64
64
  by reading those files and killing the exact recorded pids; it never uses `pkill -f`, `pgrep -f`, or
65
- any argv/path pattern. A journal path is shared by partner tiers and therefore cannot prove ownership.
65
+ any argv/path pattern. The pre-arm beat probe lists unowned writers and is reconciled in the beat section; retiring a watcher this seat armed stays kill-by-recorded-pid and never that probe. A journal path is shared by partner tiers and therefore cannot prove ownership.
66
66
  Verify each executing pid in two process-table reads before acting, kill-by-pid, arm the replacement,
67
67
  then verify its new pid in two process-table reads. A stand-down order inventories both sets: the
68
68
  ordering seat's recorded pids to retire, and the partner's watchers armed on the ordering seat that
@@ -72,7 +72,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
72
72
  and nothing had been watching either file.
73
73
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: File and journal watchers record
74
74
  owner pid and arm id beside the evidence files; a seat retires only watchers it armed, by those
75
- recorded pids. It never uses `pkill -f`, `pgrep -f`, or any argv/path pattern. Verify each executing
75
+ recorded pids. It never uses `pkill -f`, `pgrep -f`, or any argv/path pattern. The pre-arm beat probe lists unowned writers and is reconciled in the beat section; retiring a watcher this seat armed stays kill-by-recorded-pid and never that probe. Verify each executing
76
76
  pid in two process-table reads before acting. A stand-down inventories this seat's recorded pids
77
77
  and the partner's watchers that must survive it.
78
78
  **An adopted seat ANNOUNCES itself, in the same act as re-arming:** tell the adopted orchestrator the
@@ -140,11 +140,11 @@ through brief lineage. **An executor choice nobody made is still an executor cho
140
140
  with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
141
141
  updates it). Never long context strings or ✓-chains.
142
142
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal, name it in the same act — `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (tab title; see the seat-name law under Seat-spawn recipes) — and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
143
- 2. **Orchestrator**: Launch the orchestrator with your agent host.
144
- - **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify a model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
143
+ 2. **Orchestrator**: First consume a successful `tickmarkr fleet --pick consult` in the seat repository, per the Fleet selection contract below; then launch the returned adapter/model with your agent host.
144
+ - **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <picked-model>` after the `--` from the Fleet receipt). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <picked-model>` from the Fleet receipt). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
145
145
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create` on a path worktree selector with a command:
146
146
  `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "<agent-cmd>" --json`
147
- For Claude Code: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "claude --permission-mode bypassPermissions" --json`. For Codex: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "codex --dangerously-bypass-approvals-and-sandbox" --json`. Parse `result.terminal.handle` from the create receipt.
147
+ For Claude Code: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "claude --model <picked-model> --permission-mode bypassPermissions --settings '{\"promptSuggestionEnabled\":false}'" --json`. For Codex: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "codex --model <picked-model> --dangerously-bypass-approvals-and-sandbox" --json`. Parse `result.terminal.handle` from the create receipt.
148
148
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
149
149
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
150
150
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then announce the brief file:
@@ -202,9 +202,46 @@ Inventories retain the **full suite log**, not a tail or summary, and no one run
202
202
 
203
203
  ### Seat-spawn and Leg-2 recipes
204
204
 
205
+ **Fleet selection contract (OBS-1165), on BOTH hosts:** Before every supervisor-opened seat,
206
+ including every respawn, replacement, delegated spawn and Leg-2 dispatch, run the applicable
207
+ `tickmarkr fleet --pick <role>` in that seat's repository. Never reuse a prior pick at respawn.
208
+ The purpose-to-role mapping is explicit:
209
+
210
+ | Seat purpose | Fleet role | Vendor exclusions |
211
+ | --- | --- | --- |
212
+ | orchestrator | consult | Any vendors excluded for this mission |
213
+ | records | consult | Any vendors excluded for this mission |
214
+ | author (including planner, executor and scout) | consult | Any vendors excluded for this mission |
215
+ | consultant | consult | Any vendors excluded for this mission or consultation round |
216
+ | lab-rater | consult | Any vendors excluded for this rating round |
217
+ | independent reviewer (including checker and verifier) | review | Author vendor plus every vendor already used by independent reviewers in this round, and mission exclusions |
218
+
219
+ For example, an independent reviewer runs
220
+ `tickmarkr fleet --pick review --exclude-vendor <author-vendor> --exclude-vendor <prior-reviewer-vendor>`;
221
+ omit the prior-reviewer argument only for the first reviewer. Repeat `--exclude-vendor <vendor>`
222
+ for every applicable exclusion on either role. Track the author's actual vendor and each reviewer's
223
+ returned vendor in the seat record; if the author vendor is unknown, stop before review selection.
224
+ Unknown seat purposes require an explicit mapping decision; never silently map them to consult.
225
+
226
+ Consume only exit status zero and one complete JSON identity with `role`, `adapter`, `model`, `vendor`
227
+ and `channel`. Validate the role and exclusions against the request and record this receipt with the
228
+ seat. A nonzero refusal (including missing `<role>.prefer`, exhausted eligible preferences, or stale
229
+ probe data) stops the spawn: report the named reason, never invent a default, auto-write preferences,
230
+ or fall back to a remembered model. `fleet --print` and a preference list are not a resolved pick.
231
+
232
+ Build the launch command from the returned `adapter` and `model`, shell-quoting the model as one
233
+ argument: `claude-code` maps to executable/kind `claude` with `--model <picked-model>`, `codex` to
234
+ `codex --model <picked-model>`, `kimi` to `kimi --model <picked-model>`, and `grok` to
235
+ `grok -m <picked-model>`. Select only the matching adapter recipe below; examples do not choose
236
+ models. If the returned adapter has no documented visible TUI recipe and approval flags, stop and
237
+ report that transport limitation rather than substitute another adapter. Preserve the existing
238
+ visible-seat transport safeguards: named interactive TUI seats, host-specific create receipts,
239
+ orchestrator tab isolation, role-specific sandbox/approval flags, Claude prompt suggestions disabled,
240
+ Kimi `-y`, artifact plus terminal marker, and verified file-brief delivery. Fleet pick never starts a seat.
241
+
205
242
  - **On herdr (`HERDR_ENV=1`)**: Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
206
243
  verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
207
- `herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`.
244
+ `herdr agent start <seat> --kind grok --pane <pane> -- -m <picked-model>`.
208
245
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
209
246
  **`--worktree path:` resolves only an Orca-MANAGED worktree** (`orca worktree list`); on any other checkout —
210
247
  a `git worktree add` the overseer made for a spec branch, a throwaway clone — `terminal create` hangs and
@@ -790,14 +827,14 @@ they are left implicit:
790
827
  stall watcher to catch — silent-time equals lifetime. Standing operator rule since 2026-07-13:
791
828
  consults and one-off LLM calls run as the CLI's real interactive TUI in a visible named pane.
792
829
  Headless is for exit-code probes — a quota check that wants `rc`, never work anyone must watch.
793
- 3. **Buy seat diversity from the live capability matrix, at every dispatch.** When one vendor's model
830
+ 3. **Buy seat diversity through Fleet, at every dispatch and respawn.** Resolve authors with `tickmarkr fleet --pick consult` and independent checkers/verifiers with `tickmarkr fleet --pick review --exclude-vendor <author-vendor>`, adding every applicable vendor exclusion under the Fleet selection contract. Consume the successful JSON adapter/model before using the visible-seat recipes; never select directly from doctor data. When one vendor's model
794
831
  quota collapses, the reflex is to collapse every seat onto the surviving model and hold the
795
832
  cross-vendor CLI back for a late probe — P97 ran planner, checker and verifier as one family that
796
833
  way, and three same-family passes confirmed one wrong anchored conclusion with the refuting fact in
797
834
  the room. `<state-dir>/doctor.json` already lists every installed+authed adapter and its models (nine
798
835
  were authed on 2026-08-17 while every seat ran claude). Priority when independence is scarce:
799
836
  **verifier > checker > planner > executors**.
800
- - **On herdr (`HERDR_ENV=1`)** the independent seat goes cross-vendor with `herdr agent start … --kind codex`;
837
+ - **On herdr (`HERDR_ENV=1`)** the independent seat uses its Fleet-selected adapter kind and model with `herdr agent start … --kind <picked-kind> -- …`;
801
838
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)** it uses the Orca terminal-create
802
839
  seat recipe. The choice is ruled at dispatch, never debated under time pressure.
803
840
  **A codex seat inside a git WORKTREE cannot commit and cannot write outside the worktree** (OBS-824, measured
@@ -806,6 +843,12 @@ they are left implicit:
806
843
  report INSIDE the worktree and to commit nothing — the overseer commits from the main checkout — or give the work to a
807
844
  claude seat, or to a throwaway CLONE (a real `.git` directory). A brief that tells a codex-in-worktree seat to commit
808
845
  buys a stall, not a commit.
846
+
847
+ **Every one-shot `codex exec` from an agent shell closes stdin (D-307-pre).** This skill has no
848
+ `codex exec` recipe — cross-vendor seats are spawned interactive — and the seat's ad-hoc one-shot
849
+ did not close stdin on that command. The agent shell left stdin open, and `codex exec` then
850
+ blocks on reading additional input from stdin (`Reading additional input from stdin…`) until the
851
+ pipe ends (42 minutes, measured). Run it as `codex exec … < /dev/null`.
809
852
  4. **Gate every exec lane with the shipped battery, not hand-rolled greps.**
810
853
  `tickmarkr verify --base <ref> --criteria <file>` is the standalone form of the engine's own gates —
811
854
  build/test/lint diffed against a recorded baseline, evidence, scope, plus the semantic judges — one
@@ -826,8 +869,13 @@ they are left implicit:
826
869
  - **Verified send protocol**:
827
870
  - **On herdr (`HERDR_ENV=1`)**: `herdr agent send` writes WITHOUT Enter, and `pane run`'s Enter can be swallowed
828
871
  by bracketed-paste on long payloads. Robust sequence: read the pane (bare prompt required) → send-text →
829
- sleep 2–3s → send-keys Enter → read back (input empty / agent `working`). Never report "briefed" without
830
- the read-back. Long content goes in a brief file, never pane text. `scripts/seat-send.sh` encodes
872
+ sleep 2–3s → read back. Send `send-keys Enter` only when that read-back shows the staged text on the
873
+ composer and no permission prompt or numbered choice holds focus — an Enter sent onto an active prompt
874
+ approves it or picks a choice instead of submitting the brief (OBS-1119). When a prompt or choice holds
875
+ focus instead, resolve it first, then re-read before retrying. After a submitted Enter, read back once
876
+ more and confirm the composer is empty or the agent shows `working` — the pre-Enter read only proved
877
+ Enter was safe to send, not that the brief was submitted. Never report "briefed" without
878
+ both read-backs. Long content goes in a brief file, never pane text. `scripts/seat-send.sh` encodes
831
879
  this whole path — size guard, atomic prompt, prompt-line read-back, optional interrupt — and never
832
880
  auto-resends. Each adapter declares its prompt glyph beside its input-box matchers; `seat-send.sh` reads
833
881
  that declaration rather than assuming Claude's `❯`.
@@ -1009,7 +1057,7 @@ the repo root as its own `run_in_background` Bash call:
1009
1057
 
1010
1058
  ```bash
1011
1059
  cd <repo> && tickmarkr beat overseer --seat <overseer-agent-or-pane> --loop
1012
- tickmarkr beat overseer --seat <overseer-agent-or-pane> --stand-down # deliberately hand off; --loop exits
1060
+ cd <repo> && tickmarkr beat overseer --seat <overseer-agent-or-pane> --stand-down # deliberately hand off; --loop exits
1013
1061
  ```
1014
1062
 
1015
1063
  **The legacy wrapper loop, `while :; do tickmarkr beat overseer --seat <pane>; sleep 10; done`, is the
@@ -1043,9 +1091,10 @@ had three beat writers, one owned by an unrelated session. The shipped `--loop`
1043
1091
  but its process ownership still needs a live check. So:
1044
1092
 
1045
1093
  - **Split the liveness reads.** A tier's liveness is read from beat freshness in the repository status
1046
- path; the loop's liveness is read from the live process payload (`tickmarkr beat <tier> --seat <seat>
1047
- --loop` in this repo). Neither liveness claim is read from a recorded pid: a pid
1048
- recorded earlier can be stale, reused, or detached from the beat now holding the tier green.
1094
+ path. The loop's liveness is read from the live process payload of both arm forms: the loop arm
1095
+ (`tickmarkr beat <tier> --seat <seat> --loop` in this repo) and the legacy wrapper
1096
+ (`while :; do tickmarkr beat <tier> --seat <seat>; sleep 10; done`). Neither liveness claim is read from a recorded pid:
1097
+ a pid recorded earlier can be stale, reused, or detached from the beat now holding the tier green.
1049
1098
  - **At every adopt, clear, or re-brief, sweep for pre-existing beat writers on YOUR tier before
1050
1099
  arming one** (`pgrep -f "tickmarkr beat <tier>"`, **read twice and intersected** — this exact probe
1051
1100
  returned its own shell as pid 14680 on 2026-08-31). **The PRIMARY target is the legacy
@@ -1053,6 +1102,11 @@ but its process ownership still needs a live check. So:
1053
1102
  a `--loop` exits by itself once your new arm replaces its own, but the wrapper's bare tick beats
1054
1103
  whatever arm is on disk, so it survives your re-arm and never exits on its own. Trace each survivor
1055
1104
  to its parent session. Stop an unowned writer only — never its parent — then verify the parent survived.
1105
+ **Reconcile this probe with the recorded-pid ownership rule.** That rule retires watchers this seat
1106
+ armed by the exact recorded pid, and it never uses `pkill -f`, `pgrep -f`, or an argv pattern. This
1107
+ probe is the listing of unowned legacy writers that have no pid file; the pattern stays
1108
+ `pgrep -f "tickmarkr beat <tier>"` with no `--loop` qualifier. A stop is kill-by-pid of an unowned
1109
+ survivor after the two reads. The probe does not retire a recorded pid.
1056
1110
  - **`ARMED (<seat>)` is an attributable claim, not proof that the named seat is still alive.** Before
1057
1111
  trusting it, ask whose session owns the beater; a still-running `--loop` can keep naming a departed
1058
1112
  seat (rule 11's outliving-its-trigger failure, in beat form).
@@ -1547,6 +1601,14 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
1547
1601
  the same root read the other way: a DETACHED loop outlives its seat and holds a tier `ARMED` with
1548
1602
  nobody home (OBS-583). Neither direction may be assumed; the lifetime is a property of how the watcher
1549
1603
  was launched, and it belongs in writing next to every claim that one is armed.
1604
+ **A seat's background tasks are bound to the seat's life (D-302 add.3).** An ad-hoc background task —
1605
+ an analysis heredoc, a one-shot shell — dies with the seat that launched it. Record its pid and kill
1606
+ that pid at stand-down, the same recorded-pid act as any other task this seat owns. Leave it attached
1607
+ to the seat so it is not reparented to pid 1 and left running after the seat is gone. A watcher this
1608
+ skill explicitly labels detached, with its heartbeat file, is the process that may outlive the seat.
1609
+ **Forbid the home-wide recursive glob by name:** `glob.glob('~/**/.tickmarkr/runs/*/journal.jsonl', recursive=True)`.
1610
+ That heredoc walked the whole home (`~/**`, every worktree, `node_modules`, Library) and ran orphaned
1611
+ for 30 hours at 100% CPU after its seat was gone. No seat glob starts at `~` or `$HOME`.
1550
1612
  **And EVERY process-table probe has an idiom that defeats it, so the rule above needs the one test
1551
1613
  that survives all of them. Lead with this; it is not the last resort, it is the first move:**
1552
1614