tickmarkr 2.5.6 → 2.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +5 -5
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +6 -4
- package/dist/gates/baseline.d.ts +8 -3
- package/dist/gates/baseline.js +6 -3
- package/dist/gates/review.js +12 -1
- package/dist/gates/run-gates.d.ts +4 -1
- package/dist/gates/run-gates.js +65 -9
- package/dist/gates/test-manifest.d.ts +3 -0
- package/dist/gates/test-manifest.js +20 -2
- package/dist/gates/test-reporter.js +6 -1
- package/dist/graph/graph.d.ts +2 -0
- package/dist/graph/graph.js +5 -0
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/daemon.d.ts +9 -0
- package/dist/run/daemon.js +200 -46
- package/dist/run/git.d.ts +6 -1
- package/dist/run/git.js +63 -11
- package/dist/run/journal.d.ts +20 -4
- package/dist/run/journal.js +60 -9
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +55 -3
|
@@ -83,6 +83,43 @@ export declare function applyRunViewKey(session: RunViewSession, event: RunViewK
|
|
|
83
83
|
* a closed run whose record lacks a bucket says so rather than reading absence as empty.
|
|
84
84
|
*/
|
|
85
85
|
export declare function runSummaryLine(s: OperatorSnapshot): string;
|
|
86
|
+
/** One projected reading and the physical journal line it came from; no line means no row says it. */
|
|
87
|
+
export interface ProjectionField {
|
|
88
|
+
readonly label: string;
|
|
89
|
+
readonly line?: number;
|
|
90
|
+
}
|
|
91
|
+
export interface TaskProjection {
|
|
92
|
+
readonly taskId: string;
|
|
93
|
+
readonly identity: ProjectionField;
|
|
94
|
+
readonly phase: ProjectionField;
|
|
95
|
+
readonly build: ProjectionField;
|
|
96
|
+
readonly blocker: ProjectionField;
|
|
97
|
+
readonly nextAction: ProjectionField;
|
|
98
|
+
/** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result. */
|
|
99
|
+
readonly stalled?: ProjectionField;
|
|
100
|
+
}
|
|
101
|
+
export declare const STALL_MARKER = "\u26A0 stalled harvest suspected";
|
|
102
|
+
/**
|
|
103
|
+
* The evidence-only projection status renders (run/activity.ts, run/operator-summary.ts), folded
|
|
104
|
+
* over the journal rows this view was handed, with the physical line each reading was taken from.
|
|
105
|
+
* Nothing is inferred from a title, a provider or a task id: an unrecorded field says so.
|
|
106
|
+
*/
|
|
107
|
+
export declare function projectRunTasks(snapshot: OperatorSnapshot, rows: readonly RunEvidenceRow[], graph: RunGraph | undefined, decisions: readonly RunDecision[], runId: string): readonly TaskProjection[];
|
|
108
|
+
/** One wrapped line per task: every field with its line, and the stall marker when the harvest is suspect. */
|
|
109
|
+
export declare const projectionLine: (p: TaskProjection) => string;
|
|
110
|
+
export interface PaneLocator {
|
|
111
|
+
readonly status: "recorded" | "unavailable";
|
|
112
|
+
readonly reason: string;
|
|
113
|
+
/** The worker-launch row the reading stands on, when one exists. */
|
|
114
|
+
readonly line?: number;
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Where the current attempt's pane is recorded — and only that. The reading is exact-name: the
|
|
118
|
+
* launch row's slot name must be the owned name of THIS task, attempt and run. No terminal is ever
|
|
119
|
+
* chosen by title, provider or task id; a missing, foreign, stale or ambiguous record reads
|
|
120
|
+
* unavailable with the rows it was read from, and reading never touches a host.
|
|
121
|
+
*/
|
|
122
|
+
export declare function paneLocator(task: OperatorTask, rows: readonly RunEvidenceRow[], runId: string): PaneLocator;
|
|
86
123
|
/** The board's footer keys merged with the cockpit's own — every one of them handled by this view or the shell. */
|
|
87
124
|
export declare const RUN_VIEW_KEYS: readonly string[];
|
|
88
125
|
export interface RunViewProps {
|
|
@@ -2,7 +2,11 @@ import { jsx as _jsx, jsxs as _jsxs, Fragment as _Fragment } from "react/jsx-run
|
|
|
2
2
|
import { Box } from "ink";
|
|
3
3
|
import { useContext } from "react";
|
|
4
4
|
import { GLYPHS } from "../../brand.js";
|
|
5
|
+
import { formatOwnedName, parseOwnedName } from "../../drivers/types.js";
|
|
5
6
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
7
|
+
import { projectActivity } from "../../run/activity.js";
|
|
8
|
+
import { projectOperatorSummary } from "../../run/operator-summary.js";
|
|
9
|
+
import { trackJournalRows } from "../../run/protocol.js";
|
|
6
10
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
7
11
|
import { BOARD_KEYS, boardFooter, clipBoard, renderBoardLines } from "./board.js";
|
|
8
12
|
import { BodyText, Panel, ShellTheme } from "./components.js";
|
|
@@ -172,6 +176,185 @@ export function runSummaryLine(s) {
|
|
|
172
176
|
}
|
|
173
177
|
const attemptLabel = (t) => (t.attempt !== undefined ? `attempt ${t.attempt}` : t.dispatches === 0 ? "never dispatched" : "attempt unknown");
|
|
174
178
|
const parkOf = (t) => (t.state === "human" ? `human · ${t.parkKind ?? "unknown kind"}` : t.state);
|
|
179
|
+
export const STALL_MARKER = "⚠ stalled harvest suspected";
|
|
180
|
+
const ordinal = (v) => (typeof v === "number" && Number.isInteger(v) && v >= 0 ? v : undefined);
|
|
181
|
+
const slotNameOf = (data) => {
|
|
182
|
+
const slot = data.slot;
|
|
183
|
+
return str(data.slotName) ?? (typeof slot === "object" && slot !== null ? str(slot.name) : str(slot));
|
|
184
|
+
};
|
|
185
|
+
const NO_ROW = "no journal row";
|
|
186
|
+
const readLabel = (f) => `${f.label} ${f.line === undefined ? `(${NO_ROW})` : `#L${f.line}`}`;
|
|
187
|
+
/**
|
|
188
|
+
* The evidence-only projection status renders (run/activity.ts, run/operator-summary.ts), folded
|
|
189
|
+
* over the journal rows this view was handed, with the physical line each reading was taken from.
|
|
190
|
+
* Nothing is inferred from a title, a provider or a task id: an unrecorded field says so.
|
|
191
|
+
*/
|
|
192
|
+
export function projectRunTasks(snapshot, rows, graph, decisions, runId) {
|
|
193
|
+
const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
|
|
194
|
+
const graphTask = (id) => graph?.tasks.find((t) => t.id === id);
|
|
195
|
+
const activityTasks = snapshot.tasks.map((t) => ({ id: t.id, gates: graphTask(t.id)?.gates ?? [...GATE_NAMES], deps: graphTask(t.id)?.deps ?? [], status: t.state }));
|
|
196
|
+
const tracked = trackJournalRows(runId, ordered.map((r) => ({ raw: r.event, sourceIndex: r.line })));
|
|
197
|
+
const activity = projectActivity(runId, tracked, activityTasks);
|
|
198
|
+
const byTask = (id) => ordered.filter((r) => r.event?.taskId === id);
|
|
199
|
+
const summaries = new Map(projectOperatorSummary(snapshot.tasks.map((t) => {
|
|
200
|
+
const own = byTask(t.id);
|
|
201
|
+
const responsibility = [...own].reverse().find((r) => str(r.event.data.role) !== undefined || str(r.event.data.agent) !== undefined);
|
|
202
|
+
// The summary's own vocabulary (derive.ts reads the graph status for an unmentioned task): a task the
|
|
203
|
+
// operator fold holds as blocked, or a graph task the journal never mentions, is pending on its prerequisites.
|
|
204
|
+
const status = t.state === "blocked" || (t.state === "unknown" && graphTask(t.id) !== undefined) ? "pending" : t.state;
|
|
205
|
+
return {
|
|
206
|
+
id: t.id, status, deps: graphTask(t.id)?.deps ?? [],
|
|
207
|
+
...(own.length === 0 ? {} : { phase: activity.get(t.id)?.state, lastEvidenceAt: own.at(-1).event.ts }),
|
|
208
|
+
...(responsibility ? { responsible: { role: str(responsibility.event.data.role), agent: str(responsibility.event.data.agent) } } : {}),
|
|
209
|
+
};
|
|
210
|
+
}), decisions.map((d) => ({ taskId: d.taskId, park: { kind: d.park.kind, tombstone: d.park.tombstone, ...(d.park.reason === undefined ? {} : { reason: d.park.reason }) }, verbs: d.verbs, ...(d.diagnostic === undefined ? {} : { diagnostic: d.diagnostic }) }))).map((s) => [s.taskId, s]));
|
|
211
|
+
return snapshot.tasks.map((t) => {
|
|
212
|
+
const own = byTask(t.id);
|
|
213
|
+
const act = activity.get(t.id);
|
|
214
|
+
const summary = summaries.get(t.id);
|
|
215
|
+
// identity: the newest row that recorded a role, an agent, an owned slot name or a dispatch assignment
|
|
216
|
+
const identityRow = [...own].reverse().find((r) => { const d = r.event.data; if (act?.attempt !== undefined && ordinal(d.attempt) !== undefined && ordinal(d.attempt) !== act.attempt)
|
|
217
|
+
return false; return str(d.role) !== undefined || str(d.agent) !== undefined || slotNameOf(d) !== undefined || (r.event.event === "task-dispatch" && d.assignment !== undefined); });
|
|
218
|
+
let identity = { label: "identity unrecorded" };
|
|
219
|
+
if (identityRow) {
|
|
220
|
+
const d = identityRow.event.data;
|
|
221
|
+
const name = slotNameOf(d);
|
|
222
|
+
const owned = name === undefined ? null : parseOwnedName(name);
|
|
223
|
+
const role = str(d.role) ?? owned?.role ?? (identityRow.event.event === "worker-launch" ? "worker" : "role unrecorded");
|
|
224
|
+
const a = d.assignment;
|
|
225
|
+
const agent = str(d.agent) ?? name ?? (typeof a?.adapter === "string" && typeof a.model === "string" ? `${a.adapter}:${a.model}` : "agent unrecorded");
|
|
226
|
+
identity = { label: `${role} ${agent}`, line: identityRow.line };
|
|
227
|
+
}
|
|
228
|
+
// Finding 2: the reference is the row whose acceptance CHANGED the projected value — a rejected
|
|
229
|
+
// receipt or a delayed old-attempt result changes nothing and so is never cited. The task's rows
|
|
230
|
+
// and the run-level resets are re-folded prefix by prefix through projectActivity itself.
|
|
231
|
+
// ponytail: O(k²) in the task's own rows; memoize the fold if journals reach thousands of rows per task.
|
|
232
|
+
const changes = acceptedChanges(runId, tracked, t.id, activityTasks.find((a) => a.id === t.id));
|
|
233
|
+
const phase = { label: `phase ${summary?.phase ?? act?.state ?? "unrecorded"}`, ...(own.length > 0 && changes.state !== undefined ? { line: changes.state } : {}) };
|
|
234
|
+
const b = act?.build;
|
|
235
|
+
const receipt = b !== undefined && "receipt" in b ? b.receipt : undefined;
|
|
236
|
+
const build = { label: `build ${b?.state ?? "start-unrecorded"}${receipt ? ` exit ${receipt.exitCode ?? "-"}` : ""}`, ...(changes.build !== undefined ? { line: changes.build } : {}) };
|
|
237
|
+
const decision = decisions.find((d) => d.taskId === t.id);
|
|
238
|
+
const blockerRow = decision ? { line: decision.park.line } : [...own].reverse().find((r) => r.event.event === "task-human" || r.event.event === "task-failed");
|
|
239
|
+
const blk = summary?.blocker;
|
|
240
|
+
const blocker = { label: `blocker ${blk ? `${blk.kind}${blk.diagnostic ? ` · ${blk.diagnostic}` : ""}` : "none"}`, ...(blk && blockerRow ? { line: blockerRow.line } : {}) };
|
|
241
|
+
const nextAction = { label: `next ${blk?.nextAction ?? "none"}`, ...(blk?.nextAction && blockerRow ? { line: blockerRow.line } : {}) };
|
|
242
|
+
// OBS-1048 harvest, chronological: the newest worker-launch opens the attempt; a later worker-result retires it.
|
|
243
|
+
let launched, launchedAttempt, returned = false;
|
|
244
|
+
const nudges = [], pages = [];
|
|
245
|
+
// Finding 1: only rows of the launched attempt count; a row without an attempt belongs to it (derive.ts attemptHarvests).
|
|
246
|
+
const ofLaunched = (d) => launched !== undefined && (ordinal(d.attempt) ?? launchedAttempt) === launchedAttempt;
|
|
247
|
+
for (const r of own) {
|
|
248
|
+
const e = r.event;
|
|
249
|
+
switch (e.event) {
|
|
250
|
+
case "worker-launch":
|
|
251
|
+
launched = r.line;
|
|
252
|
+
launchedAttempt = ordinal(e.data.attempt);
|
|
253
|
+
returned = false;
|
|
254
|
+
nudges.length = 0;
|
|
255
|
+
pages.length = 0;
|
|
256
|
+
break;
|
|
257
|
+
case "worker-nudge-failed":
|
|
258
|
+
if (ofLaunched(e.data))
|
|
259
|
+
nudges.push(r.line);
|
|
260
|
+
break;
|
|
261
|
+
case "operator-page":
|
|
262
|
+
if (ofLaunched(e.data))
|
|
263
|
+
pages.push(r.line);
|
|
264
|
+
break;
|
|
265
|
+
case "worker-result":
|
|
266
|
+
if (ofLaunched(e.data))
|
|
267
|
+
returned = true;
|
|
268
|
+
break;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
const stalled = launched !== undefined && !returned && nudges.length > 0 && pages.length > 0
|
|
272
|
+
? { label: `${STALL_MARKER} · launch #L${launched} · nudge failed ${nudges.map((l) => `#L${l}`).join(",")} · paged ${pages.map((l) => `#L${l}`).join(",")} · no worker-result`, line: launched }
|
|
273
|
+
: undefined;
|
|
274
|
+
return { taskId: t.id, identity, phase, build, blocker, nextAction, ...(stalled ? { stalled } : {}) };
|
|
275
|
+
});
|
|
276
|
+
}
|
|
277
|
+
/** The line at which each projected value last changed, folding the task's rows (and run-level resets) prefix by prefix. */
|
|
278
|
+
function acceptedChanges(runId, tracked, taskId, task) {
|
|
279
|
+
const rows = tracked.filter((r) => { const raw = r.raw; return raw.taskId === taskId || (raw.taskId === undefined && (raw.event === "run-resume" || raw.event === "run-end")); });
|
|
280
|
+
const buildKey = (p) => `${p.build.state}|${"receipt" in p.build ? `${p.build.receipt.attribution.invocation}|${p.build.receipt.exitCode ?? ""}` : ""}`;
|
|
281
|
+
// Seeded from the empty fold: the initial reading is no row's doing, so only a change is cited.
|
|
282
|
+
const initial = projectActivity(runId, [], [task]).get(taskId);
|
|
283
|
+
let state = initial?.state, build = initial === undefined ? undefined : buildKey(initial);
|
|
284
|
+
const out = {};
|
|
285
|
+
for (let i = 0; i < rows.length; i++) {
|
|
286
|
+
const p = projectActivity(runId, rows.slice(0, i + 1), [task]).get(taskId);
|
|
287
|
+
if (!p)
|
|
288
|
+
continue;
|
|
289
|
+
const line = rows[i].sourceIndex;
|
|
290
|
+
if (p.state !== state) {
|
|
291
|
+
state = p.state;
|
|
292
|
+
out.state = line;
|
|
293
|
+
}
|
|
294
|
+
if (buildKey(p) !== build) {
|
|
295
|
+
build = buildKey(p);
|
|
296
|
+
out.build = line;
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
return out;
|
|
300
|
+
}
|
|
301
|
+
/** One wrapped line per task: every field with its line, and the stall marker when the harvest is suspect. */
|
|
302
|
+
export const projectionLine = (p) => [p.taskId, readLabel(p.identity), readLabel(p.phase), readLabel(p.build), readLabel(p.blocker), readLabel(p.nextAction), ...(p.stalled ? [p.stalled.label] : [])].join(" · ");
|
|
303
|
+
/**
|
|
304
|
+
* Where the current attempt's pane is recorded — and only that. The reading is exact-name: the
|
|
305
|
+
* launch row's slot name must be the owned name of THIS task, attempt and run. No terminal is ever
|
|
306
|
+
* chosen by title, provider or task id; a missing, foreign, stale or ambiguous record reads
|
|
307
|
+
* unavailable with the rows it was read from, and reading never touches a host.
|
|
308
|
+
*/
|
|
309
|
+
export function paneLocator(task, rows, runId) {
|
|
310
|
+
const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
|
|
311
|
+
// The dispatch-recorded pane field stays reachable beside every reading, with the row it stands on.
|
|
312
|
+
const recorded = task.pane === undefined ? "" : ` · recorded pane ${task.pane}${task.evidence ? ` · evidence #L${task.evidence.line}` : ""}`;
|
|
313
|
+
if (task.attempt === undefined)
|
|
314
|
+
return { status: "unavailable", reason: `no recorded attempt${recorded}` };
|
|
315
|
+
const launches = ordered.filter((r) => r.event.event === "worker-launch" && r.event.taskId === task.id && ordinal(r.event.data.attempt) === task.attempt);
|
|
316
|
+
if (launches.length === 0)
|
|
317
|
+
return { status: "unavailable", reason: `no worker-launch for attempt ${task.attempt}${recorded}` };
|
|
318
|
+
const ids = new Set(launches.map((r) => { const s = r.event.data.slot; return str(s?.id) ?? `#L${r.line}`; }));
|
|
319
|
+
if (ids.size > 1)
|
|
320
|
+
return { status: "unavailable", reason: `ambiguous: ${launches.length} launches ${launches.map((r) => `#L${r.line}`).join(",")}${recorded}` };
|
|
321
|
+
const launch = launches.at(-1);
|
|
322
|
+
const d = launch.event.data;
|
|
323
|
+
const slot = d.slot;
|
|
324
|
+
const expected = formatOwnedName({ role: "worker", taskId: task.id, attempt: task.attempt, runId });
|
|
325
|
+
if (str(d.driver) === undefined || str(slot?.id) === undefined || str(slot?.cwd) === undefined)
|
|
326
|
+
return { status: "unavailable", reason: `launch #L${launch.line} lacks driver or slot identity${recorded}`, line: launch.line };
|
|
327
|
+
if (slot?.name !== expected)
|
|
328
|
+
return { status: "unavailable", reason: `launch #L${launch.line} names ${slotNameOf(d) ?? "no pane"}, not this attempt's owned pane${recorded}`, line: launch.line };
|
|
329
|
+
// Finding 3: only a host whose focus capability VERIFIES the recorded identity can ever confirm it —
|
|
330
|
+
// herdr needs the launch's workspace (HerdrDriver.focus rejects a target without one), orca verifies
|
|
331
|
+
// ownership but cannot focus, subprocess and unknown drivers verify nothing. Reading asks no host.
|
|
332
|
+
const driver = String(d.driver);
|
|
333
|
+
if (driver === "herdr" && str(d.workspace) === undefined)
|
|
334
|
+
return { status: "unavailable", reason: `launch #L${launch.line} records no workspace; herdr cannot verify the pane${recorded}`, line: launch.line };
|
|
335
|
+
if (driver !== "herdr" && driver !== "orca")
|
|
336
|
+
return { status: "unavailable", reason: `launch #L${launch.line} driver ${driver} cannot verify a pane${recorded}`, line: launch.line };
|
|
337
|
+
const stale = ordered.find((r) => {
|
|
338
|
+
if (r.line <= launch.line)
|
|
339
|
+
return false;
|
|
340
|
+
const event = r.event;
|
|
341
|
+
if (event.event === "run-resume" || event.event === "run-end")
|
|
342
|
+
return true;
|
|
343
|
+
if (event.taskId !== task.id)
|
|
344
|
+
return false;
|
|
345
|
+
if (event.event === "worker-result")
|
|
346
|
+
return (ordinal(event.data.attempt) ?? task.attempt) === task.attempt;
|
|
347
|
+
if (event.event === "pane-close")
|
|
348
|
+
return str(event.data.paneId) === str(slot?.id);
|
|
349
|
+
return event.event === "merge";
|
|
350
|
+
});
|
|
351
|
+
if (stale)
|
|
352
|
+
return { status: "unavailable", reason: `launch #L${launch.line} is stale: ${stale.event.event} #L${stale.line} followed it${recorded}`, line: launch.line };
|
|
353
|
+
// A journal launch is evidence of what was opened, not a live-host observation. This pure read
|
|
354
|
+
// deliberately performs no focus/list operation, so it cannot upgrade the locator to recorded:
|
|
355
|
+
// the pane may have been closed outside the journal. The explicit `o` action owns verification.
|
|
356
|
+
return { status: "unavailable", reason: `launch #L${launch.line} records ${driver} ${String(slot.id)}, but live host verification is required${recorded}`, line: launch.line };
|
|
357
|
+
}
|
|
175
358
|
/** The board's footer keys merged with the cockpit's own — every one of them handled by this view or the shell. */
|
|
176
359
|
export const RUN_VIEW_KEYS = [...BOARD_KEYS.slice(0, 2), "←→ verdict gate", "PgUp/PgDn page", "x outcome", "a actions", "o pane", "Tab focus", "? keys", BOARD_KEYS[2]];
|
|
177
360
|
/** The Run body: the approved board (BD-1) over the fold, then the selected task's detail panels. */
|
|
@@ -182,6 +365,10 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
|
|
|
182
365
|
const fit = (text) => fitCells(text, inner);
|
|
183
366
|
const decisionLines = (lines) => lines.flatMap((line) => wrapCells(line, inner));
|
|
184
367
|
const lookup = evidenceLookup(rows, page);
|
|
368
|
+
const wrap = (text) => wrapCells(text, inner);
|
|
369
|
+
// The whole journal, not the retained tail: the projection and the locator read every row.
|
|
370
|
+
const allRows = lookup.later(0);
|
|
371
|
+
const projections = projectRunTasks(snapshot, allRows, graph, decisions, run.runId);
|
|
185
372
|
const tasks = snapshot.tasks;
|
|
186
373
|
const selection = Math.min(session.selection, Math.max(0, tasks.length - 1));
|
|
187
374
|
const task = tasks[selection];
|
|
@@ -195,9 +382,9 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
|
|
|
195
382
|
const { menu, confirming, receipt, notice } = session.decisions;
|
|
196
383
|
const blocked = task === undefined ? [] : graph?.tasks.filter((g) => g.deps.includes(task.id)).map((g) => g.id) ?? decision?.blocks ?? [];
|
|
197
384
|
const board = renderBoardLines({ runId: run.runId, snapshot, graph, now: now(), live: run.live, selection: task?.id, keys: false, colour }, width);
|
|
198
|
-
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [
|
|
385
|
+
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [wrap(`state ${parkOf(task)} · ${attemptLabel(task)} · dispatches ${task.dispatches} · path ${task.path ?? "unknown"} · pane ${task.pane ?? "unknown"} · alarm ${task.alarmMs === undefined ? "unknown" : `${task.alarmMs}ms`}${task.mergeEvidence ? ` · merged #L${task.mergeEvidence.line}` : ""} · locator ${(() => { const l = paneLocator(task, allRows, run.runId); return l.status === "recorded" ? `recorded ${l.reason}` : `unavailable — ${l.reason}`; })()}`).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit(`outcome filter ${session.outcomeFilter} (x cycles) · ${shown.length} of ${cells.length} gates shown`) }), shown.map((c) => (_jsx(BodyText, { children: fit(`${c.letter} ${c.gate.padEnd(10)} ${c.labels.join(" · ")}${c.line === undefined ? "" : ` · #L${c.line}`}`) }, c.gate)))] })), projections.length > 0 && (_jsx(Panel, { title: "PROJECTION / every task \u00B7 identity \u00B7 phase \u00B7 build \u00B7 blocker \u00B7 next action, each with its journal line", children: projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => _jsx(BodyText, { emphasis: p.stalled && i === 0 ? "strong" : "normal", children: line }, `${p.taskId}:${i}`))) })), verdictCell !== undefined && (_jsxs(Panel, { title: `VERDICT / ${verdictCell.gate}${verdictCell.line === undefined ? "" : ` #L${verdictCell.line}`} · ${verdict.length === 0 ? "no verdict text" : `lines ${from + 1}–${from + window.length} of ${verdict.length}`}`, children: [window.map((line, i) => _jsx(BodyText, { children: fit(`${String(from + i + 1).padStart(4)} ${line}`) }, `${from + i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit("←→ gate · PageUp/PageDown page") })] })), task?.state === "human" && (_jsx(Panel, { title: `PARK / ${task.id}`, children: decision === undefined
|
|
199
386
|
? _jsx(BodyText, { children: fit("park evidence unavailable — refresh before deciding") })
|
|
200
|
-
: (_jsxs(_Fragment, { children: [
|
|
387
|
+
: (_jsxs(_Fragment, { children: [wrap(`#L${decision.park.line} ${decision.park.kind ?? "unknown kind"}${decision.park.failedGate ? ` · failed gate ${decision.park.failedGate}` : ""} · ${decision.park.reason ?? "no reason recorded"}`).map((line, i) => _jsx(BodyText, { children: line }, `park:${i}`)), _jsx(BodyText, { children: fit(`blocks ${blocked.length === 0 ? "nothing" : blocked.join(", ")} · attempts ${decision.attempts}`) }), wrap(decision.verbs.length === 0 ? `diagnostic: ${decision.diagnostic ?? "no verb"}` : `decisions: ${decision.verbs.join(" · ")} (a Actions)`).map((line, i) => _jsx(BodyText, { children: line }, `verbs:${i}`)), _jsx(BodyText, { emphasis: "dim", children: fit(run.live ? "matching live daemon enacts a release at its next task boundary" : run.blockingRunId ? `other live run ${run.blockingRunId} holds the lock; resume after it ends` : "no live owner: a release records permission; resume dispatches") })] })) })), menu !== null && (_jsx(Panel, { title: `ACTIONS / ${menu.taskId}`, focused: true, children: menu.verbs.length === 0
|
|
201
388
|
? _jsx(BodyText, { children: fit(`no decision — ${decision?.diagnostic ?? "no verb for this park"}`) })
|
|
202
389
|
: menu.verbs.map((verb, i) => _jsx(BodyText, { emphasis: i === menu.selection ? "strong" : "normal", children: fit(`${i === menu.selection ? `${GLYPHS.pointer} ` : " "}${verb}`) }, verb)) })), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.command.verb.toUpperCase()}`, focused: true, children: decisionLines(decisionConfirmLines(confirming)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), receipt !== null && (_jsx(Panel, { title: receipt.ok ? "RECEIPT · appended" : "RECEIPT · refused", children: decisionLines(decisionReceiptLines(receipt)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), notice !== null && _jsx(BodyText, { children: notice }), _jsx(BodyText, { emphasis: "dim", children: clipBoard(boardFooter([...RUN_VIEW_KEYS, decisionKeybar(session.decisions, decision)].filter(Boolean), colour), width) })] }));
|
|
203
390
|
}
|
package/package.json
CHANGED
|
@@ -139,7 +139,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
139
139
|
fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
|
|
140
140
|
with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
|
|
141
141
|
updates it). Never long context strings or ✓-chains.
|
|
142
|
-
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
|
|
142
|
+
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal, name it in the same act — `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (tab title; see the seat-name law under Seat-spawn recipes) — and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
|
|
143
143
|
2. **Orchestrator**: Launch the orchestrator with your agent host.
|
|
144
144
|
- **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify a model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
|
|
145
145
|
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create` on a path worktree selector with a command:
|
|
@@ -176,6 +176,12 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
176
176
|
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Re-list terminals, re-read the
|
|
177
177
|
recorded terminal handles and evidence files, and re-arm only file/journal watchers. Do not use a Herdr
|
|
178
178
|
pane id, `agent start`, `pane run`, tab canon, or a name-keyed Herdr watcher on an Orca launch.
|
|
179
|
+
**Handles do not survive an Orca restart (OBS-1050, three gates lost to it in v2.5.6).** The restart
|
|
180
|
+
re-issues every terminal handle; a recorded one answers `terminal_handle_stale` and any watcher or gate
|
|
181
|
+
keyed to it is already dead. So: record every seat's handle in `pids/<seat>.handle` beside its title at
|
|
182
|
+
spawn; on `terminal_handle_stale`, re-list terminals and re-resolve the seat by TITLE, then rewrite the
|
|
183
|
+
handle file; and treat a stale handle as a dead seat until the read-back of the new handle proves
|
|
184
|
+
otherwise — a gate that was running in that seat has to be re-launched, never assumed to continue.
|
|
179
185
|
When briefing a seat on either host, take its current address from the handoff file; never hardcode a pane id
|
|
180
186
|
or terminal handle in a brief or command template.
|
|
181
187
|
|
|
@@ -199,7 +205,45 @@ Inventories retain the **full suite log**, not a tail or summary, and no one run
|
|
|
199
205
|
- **On herdr (`HERDR_ENV=1`)**: Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
|
|
200
206
|
verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
|
|
201
207
|
`herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`.
|
|
202
|
-
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
|
|
208
|
+
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
|
|
209
|
+
**`--worktree path:` resolves only an Orca-MANAGED worktree** (`orca worktree list`); on any other checkout —
|
|
210
|
+
a `git worktree add` the overseer made for a spec branch, a throwaway clone — `terminal create` hangs and
|
|
211
|
+
returns "Timed out waiting for terminal handle after creation". The lawful form for such a checkout is a
|
|
212
|
+
terminal in a managed worktree with a cwd change in the command:
|
|
213
|
+
`orca terminal create --worktree path:<managed-repo> --title "ORCH · <version>" --command "cd <checkout> && <agent-cmd>" --json`.
|
|
214
|
+
**An Orca-hosted run's checkout is created by Orca** (`orca worktree create --repo id:<repo> --base-branch <spec-branch>`),
|
|
215
|
+
never by `git worktree add`: the orca driver asks `orca worktree current` from inside every task worktree and refuses
|
|
216
|
+
(`selector_not_found`) on an unmanaged checkout — every task fails pre-launch and the board cannot place (OBS-1061,
|
|
217
|
+
one abandoned run). **The overseer sits in the RUN's workspace**, in its own tab beside the orchestrator's, so the
|
|
218
|
+
run's card shows its supervision; records may live on another branch and are committed there by path.
|
|
219
|
+
**The orchestrator NEVER shares the overseer's tab.** Never spawn it with `orca terminal split` off the
|
|
220
|
+
overseer's own handle, even as a fallback: the daemon self-places its live board as a split of the
|
|
221
|
+
LAUNCHING terminal, so an orchestrator in the overseer's tab drags the board and every worker pane
|
|
222
|
+
into it (measured 2026-09-20: `terminal create` timed out on an unmanaged spec worktree, the overseer
|
|
223
|
+
split instead, and the run had to be re-seated before GO). Before delivering the brief, prove the seat
|
|
224
|
+
is in its own tab: `orca terminal list --json`, the new handle's `tabId` must differ from the
|
|
225
|
+
overseer's; if it does not, close the seat and create it again the lawful way. `split` is for seats
|
|
226
|
+
that BELONG beside another seat (consultant pairs, a reviewer beside its surgeon), never for the
|
|
227
|
+
orchestrator or a gate seat.
|
|
228
|
+
**Seat names on Orca are TAB titles, and a tab title HOLDS; the pane title is the agent's and drifts.**
|
|
229
|
+
Orca keeps two titles per seat. The TAB title is owned: `orca terminal create --title` and
|
|
230
|
+
`orca terminal rename --terminal <handle> --title "<seat>"` set it, and nothing a process writes to
|
|
231
|
+
its pty changes it — measured 2026-09-20 (OBS-1063) on a scratch terminal writing an OSC title every
|
|
232
|
+
2 s: the PANE title flipped to the OSC text within 8 s, the TAB title stayed, and the operator's tab
|
|
233
|
+
strip renders the TAB title. The pane title is whatever the process last wrote (oh-my-zsh, Claude
|
|
234
|
+
Code's turn summary, Codex's last prompt) and is NOT the seat's name. This is vendor-neutral by
|
|
235
|
+
construction — no per-vendor env var, no agent command, no launch-form ritual — which is what makes
|
|
236
|
+
it the same law as Herdr's tab label. So: **the overseer names its OWN seat at Setup 1** with
|
|
237
|
+
`orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (the seat
|
|
238
|
+
it was invoked in is the one seat nobody else launches), every seat it spawns is named by
|
|
239
|
+
`create --title`, and a live-label change (progress fraction, hot-state token) is `rename` again.
|
|
240
|
+
**Prove a name by the field that holds:** `tabs[].title` in
|
|
241
|
+
`orca terminal list --include-visual-layouts --json`, never the plain `list`'s `title` — that one is
|
|
242
|
+
the pane title and reads as drift on every agent seat, which is how three sessions "proved" a
|
|
243
|
+
failure that was a misread (`src/drivers/orca.ts:27-42` keys worker identity on the owned tab title
|
|
244
|
+
for the same reason). The earlier three-part launch form (`DISABLE_AUTO_TITLE` + `printf` +
|
|
245
|
+
`CLAUDE_CODE_DISABLE_TERMINAL_TITLE=1`) is withdrawn: it silenced one vendor's pane title and named
|
|
246
|
+
nothing the tab did not already hold. Brief delivery travels as a file announced by `orca terminal send --terminal <handle> --text "<announcement>" --enter --wait-submit 15 --json`. Read `result.send.prompt.stages` and treat the brief as delivered only when a `turn_started` stage is present; `accepted: true` proves input acceptance, not a started turn; never resend on `accepted` alone.
|
|
203
247
|
For Leg-2 (both hosts), a Codex reviewer under `workspace-write` must be briefed with an in-worktree verdict path such as
|
|
204
248
|
`<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
|
|
205
249
|
## Supervising tickmarkr as the executor — WHO DOES WHAT
|
|
@@ -656,6 +700,14 @@ they are left implicit:
|
|
|
656
700
|
fail-closed verdict, no daemon, no retries. A per-lane grep gate re-implements a weaker version of
|
|
657
701
|
this and passes on source text the screen never renders, which is exactly the class the acceptance
|
|
658
702
|
judge exists to reject. One command per lane, named in the lane's own brief.
|
|
703
|
+
Fix-leg laws learned the hard way (v2.5.6 R25/R26/R34):
|
|
704
|
+
- A `test:` criterion for a fix leg is written FROM the landed test's title, verbatim, never before the
|
|
705
|
+
test exists — the acceptance oracle runs the sentence as vitest `-t` and a paraphrase matches zero tests.
|
|
706
|
+
- Every fix-leg verify carries `--author <the seat that wrote it>`; without it the author is HUMAN, every
|
|
707
|
+
vendor stays eligible, and a same-vendor approval is not the cross-vendor review the criterion names.
|
|
708
|
+
- A grader change is proved on a captured REAL job log, never on a synthetic one.
|
|
709
|
+
- A public-CI red is classified in a throttled Linux container first (`docker run --cpus=2 node:22`),
|
|
710
|
+
by measurement: starvation on the base tree is not a defect on the branch.
|
|
659
711
|
|
|
660
712
|
## Pane mechanics that bite
|
|
661
713
|
|
|
@@ -1029,7 +1081,7 @@ orchestrator turn boundary.
|
|
|
1029
1081
|
|
|
1030
1082
|
### On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`) — supervision instruments
|
|
1031
1083
|
|
|
1032
|
-
At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion.
|
|
1084
|
+
At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion. A seat-liveness watcher on the ORCHESTRATOR handle is mandatory, not optional: poll `orca terminal read --terminal <handle> --screen --json` on a bounded interval, treat `terminal_handle_stale` or a missing terminal as seat death, and re-resolve by title before re-arming (OBS-1050 — the orchestrator seat exited silently and the daemon ran unsupervised to a PARTIAL run-end).
|
|
1033
1085
|
|
|
1034
1086
|
## Specialist pipeline rules
|
|
1035
1087
|
|